<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE ep-patent-document PUBLIC "-//EPO//EP PATENT DOCUMENT 1.7.1//EN" "ep-patent-document-v1-7-1.dtd">
<!-- This XML data has been generated under the supervision of the European Patent Office -->
<ep-patent-document id="EP25160557A1" file="EP25160557NWA1.xml" lang="en" country="EP" doc-number="4801066" kind="A1" date-publ="20260902" status="n" dtd-version="ep-patent-document-v1-7-1">
<SDOBI lang="en"><B000><eptags><B001EP>ATBECHDEDKESFRGBGRITLILUNLSEMCPTIESILTLVFIROMKCYALTRBGCZEEHUPLSKBAHRIS..MTNORSMESMMAKHTNMDGE........</B001EP><B005EP>J</B005EP><B007EP>0009012-RPUB02</B007EP></eptags></B000><B100><B110>4801066</B110><B120><B121>EUROPEAN PATENT APPLICATION</B121></B120><B130>A1</B130><B140><date>20260902</date></B140><B190>EP</B190></B100><B200><B210>25160557.2</B210><B220><date>20250227</date></B220><B250>en</B250><B251EP>en</B251EP><B260>en</B260></B200><B400><B405><date>20260902</date><bnum>202636</bnum></B405><B430><date>20260902</date><bnum>202636</bnum></B430></B400><B500><B510EP><classification-ipcr sequence="1"><text>H04S   1/00        20060101AFI20250807BHEP        </text></classification-ipcr></B510EP><B520EP><classifications-cpc><classification-cpc sequence="1"><text>H04S   1/007       20130101 FI20250728BHEP        </text></classification-cpc><classification-cpc sequence="2"><text>H04S2400/03        20130101 LA20250728BHEP        </text></classification-cpc><classification-cpc sequence="3"><text>H04S2420/01        20130101 LA20250728BHEP        </text></classification-cpc><classification-cpc sequence="4"><text>H04S2420/03        20130101 LA20250728BHEP        </text></classification-cpc><classification-cpc sequence="5"><text>H04S2420/07        20130101 LA20250728BHEP        </text></classification-cpc></classifications-cpc></B520EP><B540><B541>de</B541><B542>AUDIOVORRICHTUNG UND BETRIEBSVERFAHREN DAFÜR</B542><B541>en</B541><B542>AUDIO APPARATUS AND METHOD OF OPERATION THEREFOR</B542><B541>fr</B541><B542>APPAREIL AUDIO ET SON PROCÉDÉ DE FONCTIONNEMENT</B542></B540><B590><B598>2</B598></B590></B500><B700><B710><B711><snm>Koninklijke Philips N.V.</snm><iid>102120190</iid><irf>2024P00622EP</irf><adr><str>High Tech Campus 34</str><city>5656 AE Eindhoven</city><ctry>NL</ctry></adr></B711></B710><B720><B721><snm>SCHUIJERS, Erik Gosuinus Petrus</snm><adr><city>Eindhoven</city><ctry>NL</ctry></adr></B721><B721><snm>KECHICHIAN, Patrick</snm><adr><city>Eindhoven</city><ctry>NL</ctry></adr></B721><B721><snm>KOPPENS, Jeroen Gerardus Henricus</snm><adr><city>Eindhoven</city><ctry>NL</ctry></adr></B721></B720><B740><B741><snm>Philips Intellectual Property &amp; Standards</snm><iid>101808802</iid><adr><str>High Tech Campus 34</str><city>5656 AE Eindhoven</city><ctry>NL</ctry></adr></B741></B740></B700><B800><B840><ctry>AL</ctry><ctry>AT</ctry><ctry>BE</ctry><ctry>BG</ctry><ctry>CH</ctry><ctry>CY</ctry><ctry>CZ</ctry><ctry>DE</ctry><ctry>DK</ctry><ctry>EE</ctry><ctry>ES</ctry><ctry>FI</ctry><ctry>FR</ctry><ctry>GB</ctry><ctry>GR</ctry><ctry>HR</ctry><ctry>HU</ctry><ctry>IE</ctry><ctry>IS</ctry><ctry>IT</ctry><ctry>LI</ctry><ctry>LT</ctry><ctry>LU</ctry><ctry>LV</ctry><ctry>MC</ctry><ctry>ME</ctry><ctry>MK</ctry><ctry>MT</ctry><ctry>NL</ctry><ctry>NO</ctry><ctry>PL</ctry><ctry>PT</ctry><ctry>RO</ctry><ctry>RS</ctry><ctry>SE</ctry><ctry>SI</ctry><ctry>SK</ctry><ctry>SM</ctry><ctry>TR</ctry></B840><B844EP><B845EP><ctry>BA</ctry></B845EP></B844EP><B848EP><B849EP><ctry>GE</ctry></B849EP><B849EP><ctry>KH</ctry></B849EP><B849EP><ctry>MA</ctry></B849EP><B849EP><ctry>MD</ctry></B849EP><B849EP><ctry>TN</ctry></B849EP></B848EP></B800></SDOBI>
<abstract id="abst" lang="en">
<p id="pa01" num="0001">An audio apparatus generates a data signal including encoded data for the mono downmix audio signal being a downmix of a stereo signal, spatial upmix parameters, and data for a difference signal indicative of a difference between the mono downmix audio signal and a directional signal representing a point source audio component. Another apparatus comprises a receiver (301) receiving the data signal and a compensator (305) which generates a modified mono downmix audio signal by compensating the mono downmix audio signal as a function of the difference signal. The resulting modified mono downmix audio signal is rendered by a renderer (307) at a direction determined from the spatial upmix parameters by a direction determining circuit (309).
<img id="iaf01" file="imgaf001.tif" wi="115" he="97" img-content="drawing" img-format="tif"/></p>
</abstract>
<description id="desc" lang="en"><!-- EPO <DP n="1"> -->
<heading id="h0001">FIELD OF THE INVENTION</heading>
<p id="p0001" num="0001">The invention relates to an audio apparatus and method of operation therefor, and specifically, but not exclusively, to representing and rendering stereo signals using Parametric Stereo based encoding.</p>
<heading id="h0002">BACKGROUND OF THE INVENTION</heading>
<p id="p0002" num="0002">Spatial audio applications have become numerous and widespread and increasingly form part of many audiovisual experiences. New and improved spatial experiences and applications are continuously being developed which result in increased demands on the audio processing and rendering.</p>
<p id="p0003" num="0003">A lot of research and development effort has focused on providing efficient and high quality audio encoding and audio decoding for spatial audio. A frequently used spatial audio representation is multichannel audio representations, including stereo representation, and efficient encoding of such multichannel audio based on downmixing multichannel audio signals to downmix channels with fewer channels have been developed. One of the main advances in low bit-rate audio coding has been the use of parametric multichannel coding where a downmix signal is generated together with parametric data that can be used to upmix the downmix signal to recreate the multichannel audio signal.</p>
<p id="p0004" num="0004">In particular, instead of traditional mid-side or intensity coding, in parametric multichannel audio coding, a multichannel input signal is downmixed to a lower number of channels (e.g. two to one) and multichannel image (stereo) parameters are extracted. Then the downmix signal is encoded using a more traditional audio coder (e.g. a mono audio encoder). The bitstream of the downmix is multiplexed with the encoded multichannel image parameter bitstream. This bitstream is then transmitted to the decoder, where the process is inverted. First the downmix audio signal is decoded, after which the multichannel audio signal is reconstructed guided by the encoded multichannel image/ upmix parameters.</p>
<p id="p0005" num="0005">An example of stereo coding is described in <nplcit id="ncit0001" npl-type="s"><text>E. Schuijers, W. Oomen, B. den Brinker, J. Breebaart, "Advances in Parametric Coding for High-Quality Audio", 114th AES Convention, Amsterdam, The Netherlands, 2003, Preprint 5852</text></nplcit>. In the described approach, the downmixed mono signal is parametrized by exploiting the natural separation of the signal into three components (objects): transients, sinusoids, and noise. In <nplcit id="ncit0002" npl-type="s"><text>E. Schuijers, J. Breebaart, H. Pumhagen, J. Engdegård, "Low Complexity Parametric Stereo Coding", 116th AES, Berlin, Germany, 2004</text></nplcit>, Preprint 6073 more details<!-- EPO <DP n="2"> --> are provided describing how parametric stereo was realized with a low (decoder) complexity when combining it with Spectral Band Replication (SBR).</p>
<p id="p0006" num="0006">Parametric Stereo (PS) is a technology which is widely used to efficiently code a stereo signal as a mono downmix and a set of spatial parameters allowing an accurate reconstruction of the stereo image. PS has been used to substantially improve the compression efficiency for AAC and USAC at lower bit-rates, say 32kbps and below.</p>
<p id="p0007" num="0007">In addition to accurately reproducing a stereo signal, it has also been of interest to create high quality binaural rendering of (encoded) stereo signals to emulate a virtual loudspeaker playback.</p>
<p id="p0008" num="0008">Binaural rendering of content authored for multi-channel playback can be achieved by the sum of convolutions of the input channel signals with left and right Head Related Impulse Responses (HRIRs), where each HRIR pair corresponds to a measured/simulated impulse response from a loudspeaker location to the ears. This can be expressed compactly in the z-domain as: <maths id="math0001" num=""><math display="block"><msub><mi mathvariant="normal">Y</mi><mrow><mi mathvariant="normal">L</mi><mo>,</mo><mi mathvariant="normal">R</mi></mrow></msub><mfenced><mi mathvariant="normal">z</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><munder><mo>∑</mo><mrow><mo>∀</mo><mi mathvariant="normal">c</mi></mrow></munder><msub><mi mathvariant="normal">X</mi><mi mathvariant="normal">C</mi></msub><mfenced><mi mathvariant="normal">z</mi></mfenced><mo>⋅</mo><msubsup><mi mathvariant="normal">H</mi><mrow><mi mathvariant="normal">L</mi><mo>,</mo><mi mathvariant="normal">R</mi></mrow><msub><mi mathvariant="normal">φ</mi><mi mathvariant="normal">c</mi></msub></msubsup><mfenced><mi mathvariant="normal">z</mi></mfenced></mstyle></math><img id="ib0001" file="imgb0001.tif" wi="50" he="11" img-content="math" img-format="tif"/></maths> where X<sub>C</sub>(z) represents the z-transform of the time domain input signal x<sub>c</sub>[n] with channel c, Y<sub>L,R</sub>(z) represents the z-transform of the left and right time domain output signals l[n] and r[n] respectively and <maths id="math0002" num=""><math display="inline"><msubsup><mi mathvariant="normal">H</mi><mrow><mi mathvariant="normal">L</mi><mo>,</mo><mi mathvariant="normal">R</mi></mrow><msub><mi mathvariant="normal">φ</mi><mi mathvariant="normal">c</mi></msub></msubsup><mfenced><mi mathvariant="normal">z</mi></mfenced></math><img id="ib0002" file="imgb0002.tif" wi="14" he="6" img-content="math" img-format="tif" inline="yes"/></maths> is the z-transform of the HRIR of the left and right channels h<sub>l</sub>[n] and h<sub>r</sub>[n] for the angle (and distance) corresponding to loudspeaker position φ<sub>c</sub>. Approaches for rendering binaural stereo are disclosed in <patcit id="pcit0001" dnum="WO2010122455A1"><text>WO2010/122455A1</text></patcit> and <patcit id="pcit0002" dnum="WO2007031896A1"><text>WO2007/031896 A1</text></patcit>.</p>
<p id="p0009" num="0009">A particular approach for rendering stereo audio using headphones is presented in <nplcit id="ncit0003" npl-type="s"><text>J. Breebaart and E. Schuijers. "Phantom materialization: A novel method to enhance stereo audio reproduction on headphones.", IEEE transactions on audio, speech, and language processing 16.8 (2008): 1503-1511</text></nplcit>. The approach seeks to provide accurate rendering of both audio point sources and more ambient and background audio.</p>
<p id="p0010" num="0010">However, whereas current approaches for audio rendering may provide acceptable performance in many applications and scenarios, they tend to not be ideal and may exhibit suboptimal behavior in some scenarios. In particular, it may result in suboptimal perceived quality and/or a reduced user experience with e.g. perceived suboptimal spatial perception/audio scene in some cases. Complexity and/or resource usage may also be higher than desired and may in some case make the approach undesired or impractical for some implementations, such as applications based on small and cheap portable devices.</p>
<p id="p0011" num="0011">Hence, an improved approach would be advantageous. In particular an approach allowing increased flexibility, improved adaptability, improved performance, increased audio quality, improved perceived quality, an improved rendering/generation of a multichannel signal, improved spatial<!-- EPO <DP n="3"> --> perception, reduced complexity and/or resource usage, reduced computational load, facilitated implementation, improved user experience, and/or an improved spatial audio experience would be advantageous.</p>
<heading id="h0003">SUMMARY OF THE INVENTION</heading>
<p id="p0012" num="0012">Accordingly, the Invention seeks to preferably mitigate, alleviate or eliminate one or more of the above mentioned disadvantages singly or in any combination.</p>
<p id="p0013" num="0013">According to an aspect of the invention there is provided an audio apparatus comprising: a receiver arranged to receive a data signal comprising a set of spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal, encoded data for a mono downmix audio signal being a downmix of a stereo signal, and a difference signal representing a difference between the mono downmix audio signal and a directional signal, the directional signal representing a point source audio component of the stereo signal; a store comprising directional transfer functions for different directions, a directional transfer function for a given direction representing a mapping of a mono audio signal to stereo channels such that the mono audio signal is positioned in the given direction in a stereo image of the stereo channels; a decoder arranged to generate the mono downmix audio signal and the difference signal by decoding the encoded data; a compensator arranged to generate a modified mono downmix audio signal by compensating the mono downmix audio signal as a function of the difference signal; a direction determining circuit arranged to determine a first direction from the spatial upmix parameters; a first renderer arranged to perform a first rendering of the modified mono downmix audio signal to generate a first intermediate stereo signal, the first rendering being a directional rendering using a first channel transfer function retrieved from the store for the first direction; and an output circuit to generate the output stereo signal to include the first intermediate stereo signal.</p>
<p id="p0014" num="0014">The approach may provide an improved audio experience in many embodiments. For many signals and scenarios, the approach may provide improved rendering of a stereo audio signal allowing improved generation/ reconstruction of a stereo audio signal with an improved perceived audio quality. The approach may provide improved representation of an audio scene by a stereo signal. The approach may in many embodiments allow an improved and/or more attractive spatial audio experience from a stereo signal.</p>
<p id="p0015" num="0015">The approach may provide an efficient implementation and may in many embodiments allow reduced complexity and/or resource usage. The approach may in many scenarios allow a reduced computational burden while providing a perceived high quality rendering of a stereo signal, and in particular based on an efficient representation of a stereo signal using a downmix and spatial upmix parameters, such as specifically a PS encoded stereo signal.</p>
<p id="p0016" num="0016">The processing may be in time frequency segments or tiles. Each time frequency segment/tile may represent a frequency interval in a time interval. In many embodiments, the mono downmix audio signal may be divided into time segments/intervals and a frequency representation of the<!-- EPO <DP n="4"> --> signal in the time segment/interval may be provided by signal values representing different frequency segments of the signal in the time segment/interval. Some or all of the processing may be performed in the frequency domain/ frequency subbands.</p>
<p id="p0017" num="0017">The spatial upmix parameters may comprise sets of upmix parameters, each set of upmix parameters comprising at least one of: a level difference parameter indicative of a level difference between channels of the multichannel audio signal; a correlation parameter indicative of a coherence between channels of the multichannel audio signal; a timing difference parameter indicative of a timing difference between channels of the multichannel audio signal, and a phase difference parameter indicative of a phase difference between channels of the multichannel audio signal.</p>
<p id="p0018" num="0018">The first direction may be a desired/target rendering direction. A direction may be an angle and/or orientation from a listening position/in a stereo image.</p>
<p id="p0019" num="0019">The directional rendering may be a binaural rendering generating the first intermediate stereo signal as a binaural stereo signal comprising a point source positioned in the first direction, the binaural rendering comprising selecting binaural impulse response values as values for a binaural impulse response for a sound source in the first direction. The directional transfer functions may be parameterized transfer functions, and specifically may be represented in the frequency domain as weights for each of a plurality of subbands. A weight may be provided for each stereo channel. The weights may typically be complex valued.</p>
<p id="p0020" num="0020">The binaural impulse response values may be parametric values and may be frequency tile values. The binaural impulse response values may be values representing any suitable binaural impulse response in any suitable way, including HRIR, HRTF, BRIR values etc.</p>
<p id="p0021" num="0021">In many embodiments, the encoded data and the spatial upmix parameters are part of a Parametric Stereo encoding of the stereo signal.</p>
<p id="p0022" num="0022">The compensator may be arranged to generate the modified mono downmix audio signal by compensating the mono downmix audio signal as a function of the difference signal such that a difference between the modified mono downmix audio signal and the directional signal is reduced in comparison to a difference between the mono downmix audio signal and the directional signal. In some embodiments, the compensator may be arranged to generate the modified mono downmix audio signal by offsetting the mono downmix audio signal by the difference signal. The offsetting of the mono downmix audio signal may be to reduce a difference between the directional signal and the modified mono downmix audio signal. In some embodiments, the compensator may be arranged to generate the modified mono downmix audio signal by combining the mono downmix audio signal and the difference signal. The combining may be a weighted combination and specifically a weighted linear combination. In some embodiments, the compensator may be arranged to generate the modified mono downmix audio signal by summing the mono downmix audio signal and the difference signal.<!-- EPO <DP n="5"> --></p>
<p id="p0023" num="0023">The mono downmix audio signal may be any downmix that based on the spatial parameters may be upmixed to generate the stereo signal. The mono downmix audio signal and spatial parameters may together form a representation of the (output) stereo signal.</p>
<p id="p0024" num="0024">According to an optional feature of the invention, the audio apparatus further comprises: a decorrelator arranged to apply a decorrelation to at least one of the mono downmix audio signal and the modified mono downmix audio signal to generate a first decorrelated mono downmix audio signal; a second renderer arranged to perform a second rendering being a rendering of the first decorrelated mono downmix audio signal to generate a second intermediate stereo signal, the second rendering being a predetermined rendering employing a predetermined mapping of the decorrelated mono downmix audio signal to channel signals of the second intermediate stereo signal; and wherein the output circuit is arranged to combine at least the first intermediate stereo signal and the second intermediate stereo signal to generate an output stereo signal.</p>
<p id="p0025" num="0025">This may provide improved audio rendering in many embodiments. It may in many scenarios and applications allow an improved and/or more attractive spatial audio experience from a stereo signal. It may provide an advantageous approach for many scenarios, including e.g. providing an advantageous trade-off between complexity, computational resources, data rate and/or the perceived audio quality of the generated output stereo signal.</p>
<p id="p0026" num="0026">According to an optional feature of the invention, the difference signal is indicative of a difference between the directional signal and an estimate of the directional signal, the estimate of the directional signal being a predetermined function of the mono downmix audio signal and the spatial parameters; and the compensator is arranged to generate an estimate of the directional signal from the mono downmix audio signal and the spatial parameters (using the predetermined function) and to generate the modified mono downmix audio signal by compensating the predicted signal as a function of the difference signal.</p>
<p id="p0027" num="0027">This may provide improved audio rendering in many embodiments. It may in many scenarios and applications allow an improved and/or more attractive spatial audio experience from a stereo signal. It may provide an advantageous approach for many scenarios, including e.g. providing an advantageous trade-off between complexity, computational resources, data rate and/or the perceived audio quality of the generated output stereo signal.</p>
<p id="p0028" num="0028">According to an optional feature of the invention, the compensator is arranged to determine the estimate of the directional signal by scaling the mono downmix audio signal using scale factors determined from the spatial parameters.</p>
<p id="p0029" num="0029">This may provide improved audio rendering in many embodiments. It may in many scenarios and applications allow an improved and/or more attractive spatial audio experience from a stereo signal.</p>
<p id="p0030" num="0030">The scale factors may be applied at any stage in the rendering process/path including prior to or after rendering.<!-- EPO <DP n="6"> --></p>
<p id="p0031" num="0031">In some embodiments, the scaling value may be determined substantially as: <maths id="math0003" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><msqrt><mfrac><msqrt><mrow><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup><mo>+</mo><mn>4</mn><msup><mi>ICC</mi><mn>2</mn></msup><mi>IID</mi></mrow></msqrt><mrow><mi>IID</mi><mo>+</mo><mn>1</mn></mrow></mfrac></msqrt></math><img id="ib0003" file="imgb0003.tif" wi="56" he="14" img-content="math" img-format="tif"/></maths> or <maths id="math0004" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><msqrt><mfrac><mrow><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup><mo>+</mo><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi><mo>+</mo><mfenced separators=""><mn>1</mn><mo>−</mo><mi>IID</mi></mfenced><mo>⋅</mo><msqrt><mrow><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi><mo>+</mo><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup></mrow></msqrt></mrow><mrow><mn>1</mn><mo>−</mo><msup><mi>IID</mi><mn>2</mn></msup><mo>+</mo><mfenced separators=""><mi>IID</mi><mo>+</mo><mn>1</mn></mfenced><mo>⋅</mo><msqrt><mrow><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi><mo>+</mo><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup></mrow></msqrt></mrow></mfrac></msqrt></math><img id="ib0004" file="imgb0004.tif" wi="126" he="19" img-content="math" img-format="tif"/></maths> where IID is an interchannel intensity difference and ICC is an inter-channel cross-correlation as described above.</p>
<p id="p0032" num="0032">The scale factors may be determined for and applied in frequency subbands.</p>
<p id="p0033" num="0033">According to an optional feature of the invention, a data rate of encoded data for the mono downmix audio signal is no less than two, five, ten, 25, 50, or even 100 times higher than a data rate of encoded data for the difference signal.</p>
<p id="p0034" num="0034">The approach may allow an improved audio quality to data rate relationship in many scenarios and applications.</p>
<p id="p0035" num="0035">According to an optional feature of the invention, the first rendering is a binaural rendering and the directional transfer functions are binaural transfer functions.</p>
<p id="p0036" num="0036">This may provide improved audio rendering in many embodiments.</p>
<p id="p0037" num="0037">According to an optional feature of the invention, the second rendering is arranged to generate the second intermediate stereo signal using a set of directional transfer functions retrieved from the store for a set of predetermined directions.</p>
<p id="p0038" num="0038">This may provide an advantageous approach for many scenarios, including e.g. providing an advantageous trade-off between complexity, computational resources, data rate and/or the perceived audio quality of the generated output stereo signal.</p>
<p id="p0039" num="0039">In some embodiments, the set of predetermined directions consists of one predetermined direction.</p>
<p id="p0040" num="0040">This may provide a particularly advantageous trade-off between complexity and user experience/perceived audio quality in many scenarios and embodiments.</p>
<p id="p0041" num="0041">In some embodiments, the set of predetermined directions comprises a plurality of predetermined directions.<!-- EPO <DP n="7"> --></p>
<p id="p0042" num="0042">This may provide a particularly advantageous trade-off between complexity and user experience/perceived audio quality in many scenarios and embodiments.</p>
<p id="p0043" num="0043">According to an optional feature of the invention, the spatial upmix parameters and the directional transfer functions are provided for frequency subbands and the first renderer is arranged to generate subband values for subbands of the first intermediate stereo signals from subband values of the modified mono downmix audio signal based on spatial upmix parameters and directional transfer functions for the subbands.</p>
<p id="p0044" num="0044">This may provide improved audio rendering in many embodiments. It may in many scenarios and applications allow an improved and/or more attractive spatial audio experience from a stereo signal.</p>
<p id="p0045" num="0045">According to an optional feature of the invention, the direction determining circuit is arranged to determine a point source direction in a stereo image of the stereo signal from the spatial upmix parameters, and to determine the first direction by applying a mapping function to the point source direction.</p>
<p id="p0046" num="0046">This may provide an advantageous approach for many scenarios, including e.g. providing an advantageous trade-off between complexity, computational resources, data rate and/or the perceived audio quality of the generated output stereo signal.</p>
<p id="p0047" num="0047">The mapping may be non-uniform. The mapping may be from one range to a different range (e.g. from [0,90°] to [-30°,30°]). The mapping may be non-linear.</p>
<p id="p0048" num="0048">In some embodiments, the direction determining circuit may be arranged to determine an indication of the point source direction as: <maths id="math0005" num=""><math display="block"><mi>γ</mi><mo>=</mo><mi>arctan</mi><mfenced><mfrac><mrow><mn>1</mn><mo>−</mo><mi>IID</mi><mo>+</mo><msqrt><mrow><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup><mo>+</mo><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi></mrow></msqrt></mrow><mrow><mn>2</mn><mo>⋅</mo><mi>ICC</mi><mo>⋅</mo><msqrt><mi>IID</mi></msqrt></mrow></mfrac></mfenced></math><img id="ib0005" file="imgb0005.tif" wi="89" he="16" img-content="math" img-format="tif"/></maths> where IID is an interchannel intensity difference and ICC is an inter-channel cross-correlation. The point source direction may be derived by interpreting y as a relative angle between two loudspeakers (γ = 0° is one speaker, 90° is the other). In typical notation a front-left speaker is typically at 30 degrees, and front-right at -30 degrees.</p>
<p id="p0049" num="0049">According to another aspect of the invention, there is provided an audio apparatus comprising: a downmixer arranged to generate a mono downmix audio signal for a stereo signal; a parameter determiner arranged to determine spatial upmix parameters for upmixing the mono downmix audio signal to the stereo signal, the spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal; a generator arranged to determine a directional signal for the stereo signal, the directional signal representing a point source audio component of the stereo signal; a<!-- EPO <DP n="8"> --> difference signal circuit arranged to generate a difference signal indicative of a difference between the mono downmix audio signal and the directional signal; and a data signal circuit arranged to generate the audio data signal to comprise the spatial parameters, the encoded data for the mono downmix audio signal, and the difference signal.</p>
<p id="p0050" num="0050">This may provide improved representation of a stereo signal allowing audio rendering in many embodiments and scenarios. It may in many scenarios and applications allow an improved and/or more attractive spatial audio experience from a stereo signal. It may provide an advantageous approach for many scenarios, including e.g. providing an advantageous trade-off between complexity, computational resources, data rate and/or the perceived audio quality of the generated output stereo signal.</p>
<p id="p0051" num="0051">According to an optional feature of the invention, the difference signal circuit is arranged to generate an estimate/prediction of the directional signal from the mono downmix audio signal and the spatial parameters (using the predetermined function) and to generate the difference signal to indicate a difference between the estimate/prediction of the directional signal and the directional signal.</p>
<p id="p0052" num="0052">This may provide improved performance in many embodiments.</p>
<p id="p0053" num="0053">According to an optional feature of the invention, the data circuit is arranged to include the encoded data for the difference signal in an optional data field of the audio data signal.</p>
<p id="p0054" num="0054">This may provide improved performance in many embodiments.</p>
<p id="p0055" num="0055">According to another aspect of the invention, there is provided a method of operation for an audio apparatus, the method comprising: receiving a data signal comprising a set of spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal, encoded data for a mono downmix audio signal being a downmix of a stereo signal, and a difference signal representing a difference between the mono downmix audio signal and a directional signal, the directional signal representing a point source audio component of the stereo signal; providing directional transfer functions for different directions, a directional transfer function for a given direction representing a mapping of a mono audio signal to stereo channels such that the mono audio signal is positioned in the given direction in a stereo image of the stereo channels; generating the mono downmix audio signal and the difference signal by decoding the encoded data; generating a modified mono downmix audio signal by compensating the mono downmix audio signal as a function of the difference signal; determining a first direction from the spatial upmix parameters; performing a first rendering of the modified mono downmix audio signal to generate a first intermediate stereo signal, the first rendering being a directional rendering using a first channel transfer function retrieved from the store for the first direction; and generating the output stereo signal to include the first intermediate stereo signal.</p>
<p id="p0056" num="0056">According to another aspect of the invention, there is provided a method of operation for an audio apparatus, the method comprising: generating a mono downmix audio signal for a stereo signal; determining spatial upmix parameters for upmixing the mono downmix audio signal to the stereo signal, the spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal;<!-- EPO <DP n="9"> --> determining a directional signal for the stereo signal, the directional signal representing a point source audio component of the stereo signal; generating a difference signal indicative of a difference between the mono downmix audio signal and the directional signal; and generating the audio data signal to comprise the spatial parameters, the encoded data for the mono downmix audio signal, and the difference signal.</p>
<p id="p0057" num="0057">These and other aspects, features and advantages of the invention will be apparent from and elucidated with reference to the embodiment(s) described hereinafter.</p>
<heading id="h0004">BRIEF DESCRIPTION OF THE DRAWINGS</heading>
<p id="p0058" num="0058">Embodiments of the invention will be described, by way of example only, with reference to the drawings, in which
<ul id="ul0001" list-style="none" compact="compact">
<li><figref idref="f0001">FIG. 1</figref> illustrates some elements of an example of an audio distribution system;</li>
<li><figref idref="f0002">FIG. 2</figref> illustrates some elements of an example of an audio apparatus in accordance with some embodiments of the invention;</li>
<li><figref idref="f0003">FIG. 3</figref> illustrates some elements of an example of an audio render apparatus in accordance with some embodiments of the invention;</li>
<li><figref idref="f0004">FIG. 4</figref> illustrates some elements of an example of an audio render apparatus in accordance with some embodiments of the invention; and</li>
<li><figref idref="f0005">FIG. 5</figref> illustrates some elements of a possible arrangement of a processor for implementing elements of an audio apparatus in accordance with some embodiments of the invention.</li>
</ul></p>
<heading id="h0005">DETAILED DESCRIPTION OF SOME EMBODIMENTS OF THE INVENTION</heading>
<p id="p0059" num="0059"><figref idref="f0001">FIG. 1</figref> illustrates an example of an audio system wherein a stereo signal may be distributed/communicated for remote rendering. In the system, an audio apparatus referred to as the audio source device 101 generates an audio data signal including a representation of a stereo audio signal. The stereo audio signal may be one captured at the audio source device 101, may be received from another source, or indeed may e.g. be an artificially generated stereo signal (e.g. it may be a virtual audio stereo signal).</p>
<p id="p0060" num="0060">The audio source device 101 may generate the data signal to include encoded data that represents a mono downmix audio signal for the stereo signal. For example, the mono downmix audio signal may be generated as a weighted combination, and specifically as a weighted summation, of the channel signals of an input stereo signal. In many cases, the weights may be fixed and may specifically be the same for the two channel signals.</p>
<p id="p0061" num="0061">The generated mono downmix audio signal is encoded using a suitable mono audio encoding algorithm/standard to generate encoded audio data representing the mono downmix audio signal.<!-- EPO <DP n="10"> --></p>
<p id="p0062" num="0062">In addition to the mono downmix audio signal, the audio source device 101 generates spatial upmix parameters for upmixing the mono downmix audio signal to recreate the original stereo signal.</p>
<p id="p0063" num="0063">The spatial upmix parameters are generated to be indicative of/reflect relative properties of the channel signals of the stereo audio signal. In particular, the spatial upmix parameters may be indicated to include parameters that are indicative of at least one of relative intensities/levels of the stereo channels, relative (frequency domain) phases of the stereo channels, a relative time difference between the channels, and/or a correlation between the channels. Specifically, the audio source device 101 may generate spatial upmix parameters including one or more of an inter-channel intensity difference, inter-channel level difference, inter-channel time difference, inter-channel phase difference, and/or inter-channel correlation.</p>
<p id="p0064" num="0064">The data signal may specifically comprise a Parametric Stereo (PS) encoding of the stereo signal.</p>
<p id="p0065" num="0065">A classical PS downmix is calculated as: <maths id="math0006" num=""><math display="block"><mi>m</mi><mo>=</mo><mi>c</mi><mfenced separators=""><mi>l</mi><mo>+</mo><mi>r</mi></mfenced></math><img id="ib0006" file="imgb0006.tif" wi="21" he="5" img-content="math" img-format="tif"/></maths> where the parameter <i>c</i> is chosen such that the power of the stereo signal is preserved in the downmix, the power being defined using the 2-norm: <maths id="math0007" num=""><math display="block"><msup><mfenced open="‖" close="‖"><mi>m</mi></mfenced><mn>2</mn></msup><mo>=</mo><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><mo>+</mo><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></math><img id="ib0007" file="imgb0007.tif" wi="33" he="5" img-content="math" img-format="tif"/></maths> and thus e.g.: <maths id="math0008" num=""><math display="block"><mi>c</mi><mo>=</mo><msqrt><mfrac><mrow><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><mo>+</mo><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mrow><msup><mfenced open="‖" close="‖" separators=""><mi>l</mi><mo>+</mo><mi>r</mi></mfenced><mn>2</mn></msup></mfrac></msqrt></math><img id="ib0008" file="imgb0008.tif" wi="26" he="10" img-content="math" img-format="tif"/></maths></p>
<p id="p0066" num="0066">The PS parameters are specifically an Inter-channel Intensity Difference IID, an Inter-channel Correlation ICC, and in some cases an Inter-channel Phase Difference IPD parameter. These may specifically be defined as: <maths id="math0009" num=""><math display="block"><mi>IID</mi><mo>=</mo><mfrac><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mfrac></math><img id="ib0009" file="imgb0009.tif" wi="20" he="10" img-content="math" img-format="tif"/></maths> <maths id="math0010" num=""><math display="block"><mi>ICC</mi><mo>=</mo><mfrac><mfenced open="|" close="|"><mfenced open="〈" close="〉"><mi>l</mi><mi>r</mi></mfenced></mfenced><msqrt><mrow><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mrow></msqrt></mfrac></math><img id="ib0010" file="imgb0010.tif" wi="32" he="11" img-content="math" img-format="tif"/></maths><!-- EPO <DP n="11"> --> <maths id="math0011" num=""><math display="block"><mi>IPD</mi><mo>=</mo><mi>arg</mi><mfenced open="〈" close="〉"><mi>l</mi><mi>r</mi></mfenced></math><img id="ib0011" file="imgb0011.tif" wi="31" he="4" img-content="math" img-format="tif"/></maths> where the complex-valued inner product is defined as: <maths id="math0012" num=""><math display="block"><mfenced open="〈" close="〉"><mi mathvariant="normal">x</mi><mi mathvariant="normal">y</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><munder><mo>∑</mo><mrow><mo>∀</mo><mi>i</mi></mrow></munder><msub><mi>x</mi><mi>i</mi></msub><mo>⋅</mo><msubsup><mi>y</mi><mi>i</mi><mo>∗</mo></msubsup></mstyle></math><img id="ib0012" file="imgb0012.tif" wi="33" he="9" img-content="math" img-format="tif"/></maths></p>
<p id="p0067" num="0067">The spatial upmix parameters are typically generated for specific time frequency tiles, and thus specifically each parameter value is generated/provided for a given frequency subband/interval and for a given time segment/interval.</p>
<p id="p0068" num="0068">The audio source device 101 may accordingly encode a stereo signal as encoded data representing a mono downmix audio signal of the stereo signal and associated spatial upmix parameters that are indicative of relative properties of the channels (channel signals) of the stereo signal. Specifically, the audio source device 101 may be arranged to generate a data signal comprising a conventional PS encoded stereo signal.</p>
<p id="p0069" num="0069">The system of <figref idref="f0001">FIG. 1</figref> further comprises an audio apparatus which henceforth will be referred to as the audio render apparatus 103. The audio render apparatus 103 is arranged to receive the data signal generated by the audio source device 101. In the example, the audio render apparatus 103 and audio source device 101 are both coupled to a network 105 through which the data signal can be communicated and specifically through which it can be communicated from the audio source device 101 to the audio render apparatus 103. The network 105 may specifically be, or include, the Internet.</p>
<p id="p0070" num="0070">Thus, the receiver 101 receives a data signal which comprises encoded data for a mono downmix audio signal and spatial upmix parameters for upmixing the mono downmix audio signal to the stereo signal. The set of spatial upmix parameters comprises one or more parameters indicative of relative signal properties of channels of the stereo signal, and may specifically be indicative of a level/intensity difference between the channels of the stereo signal, a cross-correlation between the channels of the stereo signal, and/or a phase difference or time difference between the channels of the stereo signals. In many cases, the data signal may comprise ICC, IID and/or IPD parameters. The data signal may specifically comprise a PS (Parametric Stereo) encoded stereo signal. The audio source device 101 may be arranged to generate a data signal which comprises a stereo signal encoded in accordance with the Parametric Stereo (PS) specifications/standard. The receiver 201 may accordingly receive a representation of a stereo signal encoded by a mono downmix audio signal and spatial upmix parameters, and specifically a PS encoded stereo signal.</p>
<p id="p0071" num="0071">The audio render apparatus 103 is arranged to process the data signal to render a stereo signal. An audio render apparatus could render the stereo signal of the data signal using a conventional PS rendering approach based on a PS upmixing of the mono downmix audio signal using the PS spatial parameters. However, the audio render apparatus 103 of <figref idref="f0001">FIG. 1</figref> uses a specific approach where point<!-- EPO <DP n="12"> --> source components may be specifically considered, and where in many embodiments different/parallel paths process the received mono downmix audio signal in different ways to generate different stereo signal components which are then combined to generate the output stereo signal.</p>
<p id="p0072" num="0072">The approach is based on a signal model for the stereo signal and specifically is based on a consideration that the stereo signal can be represented by: <maths id="math0013" num=""><math display="block"><mi>l</mi><mo>=</mo><mi>cos</mi><mfenced><mi>γ</mi></mfenced><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub></msup><mi>x</mi><mo>+</mo><msub><mi>n</mi><mi>l</mi></msub></math><img id="ib0013" file="imgb0013.tif" wi="32" he="5" img-content="math" img-format="tif"/></maths> <maths id="math0014" num=""><math display="block"><mi>r</mi><mo>=</mo><mi>sin</mi><mfenced><mi>γ</mi></mfenced><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub></msup><mi>x</mi><mo>+</mo><msub><mi>n</mi><mi>r</mi></msub></math><img id="ib0014" file="imgb0014.tif" wi="33" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0073" num="0073">This signal model essentially represents a consideration that the stereo signal corresponds to a combination of a directional signal and diffuse background audio. The directional signal x is an audio component that corresponds/reflects audio that is considered to originate from an audio source that has a point source property/characteristic and which thus has a well defined spatial origin. The audio of the directional signal thus corresponds to audio that can/should be rendered to be perceived to reach the listener from a specific direction. The directional signal may for each time frequency tile represent a single audio point source and thus may for each time frequency tile have a specific source position. The directional signal may in some cases/embodiments correspond to different audio point sources in different time frequency tiles, i.e. it is not required that all time frequency tiles of the directional source represent audio from the same point source. In some cases, the directional signal may correspond to different frequency components being reflected differently in the environment and thus may (e.g. in some frequency bands) correspond to early reflections of the point source within the acoustic environment.</p>
<p id="p0074" num="0074">In the signal model, the directional signal component <i>x</i> is phase shifted using two parameters <i>ϕ<sub>l</sub></i> and <i>ϕ<sub>r</sub></i>, and is further panned/positioned in the stereo image of the original stereo channels <i>l</i> and r. The panning is to an angle represented by the panning angle <i>γ</i>. Furthermore, a diffuse signal component is represented by diffuse residual signal components <i>n<sub>l</sub></i> and <i>n<sub>r</sub></i> of the respective left and right channels.</p>
<p id="p0075" num="0075">It is noted that the signal model description does not necessarily refer to a time-domain signal, but rather can alternatively or additionally refer to individual (potentially relatively small) frequency subbands. For example, the described signal model may individually apply to each of the frequency subbands for which separate spatial upmix parameters are provided.</p>
<p id="p0076" num="0076">In the described approach, the rendering of the stereo signal is based on a consideration of such a signal model and specifically the audio render apparatus 103 is arranged to render signal components corresponding to the directional signal x and the diffuse/residual signals <i>n<sub>l</sub></i> and <i>n<sub>r</sub>.</i></p>
<p id="p0077" num="0077">Accordingly, the audio render apparatus 103 may generate a signal representing the directional audio/signal, i.e. it may generate the directional signal component x from the received signal.<!-- EPO <DP n="13"> --> A possibility for doing so might be to determine the directional signal component by processing the mono downmix audio signal based on the spatial parameters, and specifically could be to generate a directional signal by processing of a received Parametric Stereo mono downmix audio signal based on Parametric Stereo spatial parameters. Specifically, as will be described in more detail later, a directional signal may be estimated as a spectro-temporally shaped modification of the received mono signal. However, rendering of e.g. point audio sources based on such an estimated/derived directional signal tends to result in a suboptimal quality of the rendered audio.</p>
<p id="p0078" num="0078">The inventors have realized that a major cause of such degradation in quality may be due to it typically not being feasible to generate a sufficiently accurate estimate of the directional signal from the received mono downmix audio signal.</p>
<p id="p0079" num="0079">In particular, a time frequency directional signal may be estimated using two (typically complex-valued) coefficients for each time frequency interval where the coefficients are dependent on the inter-channel properties and thus are typically determined based on the spatial parameters: <maths id="math0015" num=""><math display="block"><mi>x</mi><mo>=</mo><msub><mi>g</mi><mn>1</mn></msub><mo>⋅</mo><mi>l</mi><mo>+</mo><msub><mi>g</mi><mn>2</mn></msub><mo>⋅</mo><mi>r</mi></math><img id="ib0015" file="imgb0015.tif" wi="27" he="4" img-content="math" img-format="tif"/></maths></p>
<p id="p0080" num="0080">In most cases the gains <i>g</i><sub>1</sub> and <i>g</i><sub>2</sub> can be directly expressed as a function of the Parametric Stereo parameters (IID, ICC, IPD).</p>
<p id="p0081" num="0081">In a PS encoder, the mono downmix audio signal is typically constructed as: <maths id="math0016" num=""><math display="block"><mi>m</mi><mo>=</mo><mi>c</mi><mo>⋅</mo><mfenced separators=""><mi>l</mi><mo>+</mo><mi>r</mi></mfenced></math><img id="ib0016" file="imgb0016.tif" wi="23" he="5" img-content="math" img-format="tif"/></maths> where the parameter <i>c</i> is typically a function of the PS parameters.</p>
<p id="p0082" num="0082">So, whereas for the directional signal, the coefficients <i>g</i><sub>1</sub> and <i>g</i><sub>2</sub> may be, and typically are, different, the mono downmix audio signal for a PS encoder is typically based on an equal weighting of the channel signals. As such it is typically not feasible to recreate the desired directional signal from the mono downmix audio signal as it does not contain all information.</p>
<p id="p0083" num="0083">The issue could be addressed by the encoder side generating a downmix signal that is based on using different weights for the channels of the stereo signal, and which specifically is generated to correspond to the directional signal. However, the system of <figref idref="f0001 f0002 f0003">FIGs. 1-3</figref> is arranged to use a different approach where the mono downmix audio signal is generated to be different from the directional signal, and which specifically may be generated as the mono downmix audio signal that is typical for PS encoding. Accordingly, in many embodiments, the encoder may be arranged to generate the mono downmix audio signal to have equal weighting for the channel signals of the stereo signal being downmixed. The mono downmix audio signal may specifically be a potentially scaled summation of the left and right channel of the stereo signal. However, further, the encoder is arranged to generate and include a difference signal which reflects differences between the mono downmix audio signal and the<!-- EPO <DP n="14"> --> directional signal, and in many embodiments specifically differences between a prediction of the directional signal generated from the mono downmix audio signal and the directional signal. Such a difference signal may be generated and included in the data signal which is communicated to the renderer side. Typically, the difference signal may be generated to have a lower, and often much lower, data rate than the mono downmix audio signal and thus the additional overhead of including the difference signal may be very small.</p>
<p id="p0084" num="0084">At the renderer side, the difference signal may be utilized to generate an improved estimate of the directional signal which can be rendered e.g. as a point source thereby providing an improved perceived audio experience.</p>
<p id="p0085" num="0085">The approach may allow improved rendering of the stereo signal based on the indicated signal model. Further, this may be achieved while still allowing backwards compatibility and in particular in many cases while still using restricted downmix approaches and traditional spatial parameters. In particular, improved rendering of a stereo signal may often be achieved while using a PS compatible data signal. The data signal may thus be one that can be used for the improved rendering described in the following while still allowing rendering using standard conventional PS based renderings.</p>
<p id="p0086" num="0086">In more detail, the audio source device 101 of <figref idref="f0001">FIG. 1</figref> is arranged to generate a data signal with data representing an audio stereo signal. The audio source device 101 is specifically arranged to generate the data signal to include encoded audio data for a mono downmix audio signal which represents a downmix of the stereo signal. In addition, spatial parameters that indicate relative properties between the channels of the stereo signal are included. Such spatial data may be appropriate for upmixing the mono downmix audio signal to recreate the downmix represented by the mono downmix audio signal.</p>
<p id="p0087" num="0087">As illustrated in <figref idref="f0002">FIG. 2</figref>, the audio source apparatus 101 comprises a receiver 201 which is arranged to receive audio components from which the mono downmix audio signal is generated. <b>In</b> many embodiments, the receiver 201 may directly receive a stereo signal which is to be represented by the data signal. In other embodiments, the receiver 201 may additionally or alternatively receive a number of audio components such as audio objects, mono signals, single source signals, multichannel signals etc. The audio components are fed to a downmixer 203 which proceeds to generate the mono downmix audio signal. In many embodiments and scenarios, the downmixer 203 may generate a mono downmix audio signal from a received stereo signal, e.g. simply by summing the channels of the stereo signal in accordance with a standard PS downmix approach. In other embodiments, the downmixer 203 may be arranged to generate a stereo signal from received audio components, such as e.g. by generating an intermediate stereo signal for each received audio component followed by a combination of the intermediate stereo signals. For example, a multi-channel signal may be downmixed to an intermediate stereo signal, an audio object may be panned to the stereo image of an intermediate stereo signal based on position data provided for the audio object, etc.</p>
<p id="p0088" num="0088">The audio source apparatus 101 further comprises a spatial parameter circuit 205 which is arranged to determine the spatial parameters for the stereo signal represented by the mono downmix audio<!-- EPO <DP n="15"> --> signal. Specifically, the received or locally generated stereo signal may be fed to the parameter circuit 205 which may determine the spatial parameters from an analysis/processing of the stereo signal.</p>
<p id="p0089" num="0089">The spatial parameter circuit 205 may provide sets of frequency subband spatial parameters for the stereo signal where the sets of frequency subband spatial parameters are indicative of relative signal properties of the channels of the stereo signal. The frequency subband spatial parameters are provided for individual subbands of the stereo signal.</p>
<p id="p0090" num="0090">The spatial parameters are indicative of/reflect relative properties of the channel signals of the stereo audio signal. In particular, the spatial parameters may be indicated to include parameters that are indicative of at least one of relative intensities/levels of the stereo channels, relative (frequency domain) phases of the stereo channels, a relative time difference between the channels, and/or a correlation between the channels. Specifically, the spatial parameters may include one or more of an inter-channel intensity difference, inter-channel level difference, inter-channel time difference, inter-channel phase difference, and/or inter-channel correlation.</p>
<p id="p0091" num="0091">The spatial parameters may specifically be spatial parameters as used for encoding a stereo signal using a Parametric Stereo (PS) encoding of the stereo signal.</p>
<p id="p0092" num="0092">A classical PS downmix is calculated as: <maths id="math0017" num=""><math display="block"><mi>m</mi><mo>=</mo><mi>c</mi><mfenced separators=""><mi>l</mi><mo>+</mo><mi>r</mi></mfenced></math><img id="ib0017" file="imgb0017.tif" wi="21" he="5" img-content="math" img-format="tif"/></maths> where the parameter <i>c</i> is chosen such that the power of the stereo signal is preserved in the downmix, the power being defined using the 2-norm: <maths id="math0018" num=""><math display="block"><msup><mfenced open="‖" close="‖"><mi>m</mi></mfenced><mn>2</mn></msup><mo>=</mo><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><mo>+</mo><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></math><img id="ib0018" file="imgb0018.tif" wi="33" he="5" img-content="math" img-format="tif"/></maths> and thus e.g.: <maths id="math0019" num=""><math display="block"><mi>c</mi><mo>=</mo><msqrt><mfrac><mrow><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><mo>+</mo><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mrow><msup><mfenced open="‖" close="‖" separators=""><mi>l</mi><mo>+</mo><mi>r</mi></mfenced><mn>2</mn></msup></mfrac></msqrt></math><img id="ib0019" file="imgb0019.tif" wi="26" he="10" img-content="math" img-format="tif"/></maths></p>
<p id="p0093" num="0093">The PS parameters are specifically an Inter-channel Intensity Difference IID, an Inter-channel Correlation ICC, and in some cases an Inter-channel Phase Difference IPD parameter. These may specifically be defined/determined as: <maths id="math0020" num=""><math display="block"><mi>IID</mi><mo>=</mo><mfrac><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mfrac></math><img id="ib0020" file="imgb0020.tif" wi="20" he="10" img-content="math" img-format="tif"/></maths><!-- EPO <DP n="16"> --> <maths id="math0021" num=""><math display="block"><mi>ICC</mi><mo>=</mo><mfrac><mfenced open="|" close="|"><mfenced open="〈" close="〉"><mi>l</mi><mi>r</mi></mfenced></mfenced><msqrt><mrow><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mrow></msqrt></mfrac></math><img id="ib0021" file="imgb0021.tif" wi="32" he="11" img-content="math" img-format="tif"/></maths> <maths id="math0022" num=""><math display="block"><mi>IPD</mi><mo>=</mo><mi>arg</mi><mfenced open="〈" close="〉"><mi>l</mi><mi>r</mi></mfenced></math><img id="ib0022" file="imgb0022.tif" wi="31" he="4" img-content="math" img-format="tif"/></maths> where the complex-valued inner product is defined as: <maths id="math0023" num=""><math display="block"><mfenced open="〈" close="〉"><mi mathvariant="normal">x</mi><mi mathvariant="normal">y</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><munder><mo>∑</mo><mrow><mo>∀</mo><mi>i</mi></mrow></munder><msub><mi>x</mi><mi>i</mi></msub><mo>⋅</mo><msubsup><mi>y</mi><mi>i</mi><mo>∗</mo></msubsup></mstyle></math><img id="ib0023" file="imgb0023.tif" wi="33" he="9" img-content="math" img-format="tif"/></maths></p>
<p id="p0094" num="0094">The spatial parameters are typically provided for specific time frequency tiles, and thus specifically each parameter value is generated/provided for a given frequency subband and for a given time segment.</p>
<p id="p0095" num="0095">The spatial parameter circuit 205 may determine spatial parameters that are indicative of relative properties of the channels (channel signals) of the stereo signal.</p>
<p id="p0096" num="0096">In many embodiments, the spatial parameter circuit 205 may receive the input stereo signal and process/analyze this to generate the spatial parameters. Specifically, the spatial parameter circuit 205 may calculate the IID, ICC, and IPD values in accordance with the formulas indicated above. Thus, in many embodiments, the audio render apparatus may simply receive a stereo signal and therefrom may generate spatial parameters. In such a scenario, e.g. instead of the IID, ICC and IPD, the following inner products may alternatively or additionally be used as spatial parameters: <maths id="math0024" num=""><math display="block"><mfenced open="〈" close="〉"><mi>l</mi><mi>r</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mo>∀</mo><mi>i</mi></mrow></msub><msub><mi>l</mi><mi>i</mi></msub><mo>⋅</mo><msub><mi>r</mi><mi>i</mi></msub><mo>*</mo></mstyle></math><img id="ib0024" file="imgb0024.tif" wi="35" he="5" img-content="math" img-format="tif"/></maths> <maths id="math0025" num=""><math display="block"><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><mo>=</mo><mfenced open="〈" close="〉"><mi>l</mi><mi>l</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mo>∀</mo><mi>i</mi></mrow></msub><msub><mi>l</mi><mi>i</mi></msub><mo>⋅</mo><msub><mi>l</mi><mi>i</mi></msub><mo>*</mo></mstyle></math><img id="ib0025" file="imgb0025.tif" wi="46" he="5" img-content="math" img-format="tif"/></maths> <maths id="math0026" num=""><math display="block"><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup><mo>=</mo><mfenced open="〈" close="〉"><mi>r</mi><mi>r</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mo>∀</mo><mi>i</mi></mrow></msub><msub><mi>r</mi><mi>i</mi></msub><mo>⋅</mo><msub><mi>r</mi><mi>i</mi></msub><mo>*</mo></mstyle></math><img id="ib0026" file="imgb0026.tif" wi="48" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0097" num="0097">The spatial parameters may comprise sets of spatial parameters, each set of spatial parameters comprising at least one of: a level difference parameter indicative of a level difference between channels of the multichannel audio signal; a correlation parameter indicative of a coherence between channels of the multichannel audio signal; a timing difference parameter indicative of a timing difference between channels of the multichannel audio signal, and a phase difference parameter indicative of a phase difference between channels of the multichannel audio signal.</p>
<p id="p0098" num="0098">The audio source device 101 further comprises a data signal circuit 207 which generates the audio data signal and specifically it generates the audio data signal to include data representing the mono downmix audio signal and data representing the spatial parameters. The data signal circuit 207 may specifically include an audio signal encoder arranged to encode the mono downmix audio signal. It will be appreciated that any suitable audio signal encoding algorithm and approach may be used, such as an<!-- EPO <DP n="17"> --> Advanced Audio Coding (AAC) mono encoding algorithm. Likewise, the spatial parameters may be encoded using a suitable encoding approach.</p>
<p id="p0099" num="0099">The data signal circuit 207 may generate an output data signal in accordance with any suitable format, and may specifically generate the output data signal to follow a suitable standard for an audio signal.</p>
<p id="p0100" num="0100">The audio source device 101 further includes a directional signal generator 209 which is arranged to determine a directional signal for the stereo signal. The directional signal is generated to represent/estimate a point source audio component of the stereo signal. The directional signal may be generated to seek to represent parts of the stereo signal which can be considered to have an origin with point source properties. Thus, the directional signal is generated to include audio components that correspond to audio that have a specific position/direction of origin.</p>
<p id="p0101" num="0101">The directional signal may in some cases be generated to represent/estimate audio that has a specific position in the stereo image of the stereo signal, and thus which corresponds to audio reaching the capture position (for the stereo signal) from a single direction.</p>
<p id="p0102" num="0102">The directional signal may accordingly be generated to estimate audio that is linked with audio sources that are spatially limited/small/localized and which specifically are generated by point sources. It may thus seek to extract such audio components/parts from the stereo signal to separate point source audio from audio with a spatial extension, such as e.g. background audio, ambient audio, etc.</p>
<p id="p0103" num="0103">In some cases, the directional signal generator 209 may be arranged to estimate an audio component from a single point source and specifically the directional signal may be generated to represent/estimate a single point source. In such an approach, audio from one specific direction may be identified/estimated and represented by the directional signal.</p>
<p id="p0104" num="0104">In most embodiments, however, the directional signal is generated to represent/estimate audio that has a point source origin but not necessarily from the same single point source. The directional signal may accordingly in many embodiments represent/estimate audio from a plurality of point sources.</p>
<p id="p0105" num="0105">In particular, in many embodiments, the processing to generate the directional signal may be performed in time frequency intervals/tiles and thus different frequency time intervals/tiles may result in different point sources being detected. For example, two or more point audio sources may simultaneously be active in the audio scene with different point sources being dominant in different frequency bands. Accordingly, different point sources may be detected/extracted/estimated in different frequency bands and the directional signal may be generated to represent audio from different point sources in different frequency bands.</p>
<p id="p0106" num="0106">It will be appreciated that different algorithms and approaches may be used for determining the directional signal and that any suitable algorithm may be used without detracting from the invention.</p>
<p id="p0107" num="0107">As an example, the directional signal may be generated to correspond to a principal component determined by performing a principal component analysis on the stereo signal (and<!-- EPO <DP n="18"> --> specifically on frequency domain samples of the stereo signal). In more detail, the directional signal may follow a PCA orthogonalization according to: <maths id="math0027" num=""><math display="block"><mfenced><mtable equalrows="true" equalcolumns="true"><mtr><mtd><mi>x</mi></mtd></mtr><mtr><mtd><mi>d</mi></mtd></mtr></mtable></mfenced><mo>=</mo><mfenced><mtable equalrows="true" equalcolumns="true"><mtr><mtd><mi mathvariant="italic">cosα</mi></mtd><mtd><mi mathvariant="italic">sinα</mi></mtd></mtr><mtr><mtd><mrow><mo>−</mo><mi mathvariant="italic">sinα</mi></mrow></mtd><mtd><mi mathvariant="italic">cosα</mi></mtd></mtr></mtable></mfenced><mfenced><mtable equalrows="true" equalcolumns="true"><mtr><mtd><msup><mi>e</mi><mrow><mi>j</mi><mo>⋅</mo><msub><mi>ϕ</mi><mi>l</mi></msub></mrow></msup></mtd><mtd><mn>0</mn></mtd></mtr><mtr><mtd><mn>0</mn></mtd><mtd><msup><mi>e</mi><mrow><mi>j</mi><mo>⋅</mo><msub><mi>ϕ</mi><mi>r</mi></msub></mrow></msup></mtd></mtr></mtable></mfenced><mfenced><mtable equalrows="true" equalcolumns="true"><mtr><mtd><mi>l</mi></mtd></mtr><mtr><mtd><mi>r</mi></mtd></mtr></mtable></mfenced></math><img id="ib0027" file="imgb0027.tif" wi="74" he="9" img-content="math" img-format="tif"/></maths> where the parameters <i>α, ϕ<sub>l</sub></i> and <i>ϕ<sub>r</sub></i> are chosen in such a way that the directional signal <i>x</i> (the principal component) and the residual signal d become orthogonal: <maths id="math0028" num=""><math display="block"><mfenced open="〈" close="〉"><mi>x</mi><mi>d</mi></mfenced><mo>=</mo><mn>0</mn></math><img id="ib0028" file="imgb0028.tif" wi="22" he="4" img-content="math" img-format="tif"/></maths></p>
<p id="p0108" num="0108">It can be shown that these parameters can be expressed as a function of the inter-channel properties: <maths id="math0029" num=""><math display="block"><mi>α</mi><mo>=</mo><mfrac><mn>1</mn><mn>2</mn></mfrac><mi mathvariant="italic">arctan</mi><mfenced><mfrac><mrow><mn>2</mn><mi mathvariant="italic">ICC</mi></mrow><mrow><msqrt><mi mathvariant="italic">IID</mi></msqrt><mo>−</mo><mn>1</mn><mo>/</mo><msqrt><mi mathvariant="italic">IID</mi></msqrt></mrow></mfrac></mfenced></math><img id="ib0029" file="imgb0029.tif" wi="47" he="8" img-content="math" img-format="tif"/></maths> <maths id="math0030" num=""><math display="block"><msub><mi>ϕ</mi><mi>r</mi></msub><mo>−</mo><msub><mi>ϕ</mi><mi>l</mi></msub><mo>=</mo><mi mathvariant="italic">IPD</mi></math><img id="ib0030" file="imgb0030.tif" wi="27" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0109" num="0109">So, the directional signal is determined as: <maths id="math0031" num=""><math display="block"><mi>x</mi><mo>=</mo><mi mathvariant="italic">cosα</mi><mo>⋅</mo><msup><mi>e</mi><mrow><mi>j</mi><mo>⋅</mo><msub><mi>ϕ</mi><mi>l</mi></msub></mrow></msup><mo>⋅</mo><mi>l</mi><mo>+</mo><mi mathvariant="italic">sinα</mi><mo>⋅</mo><msup><mi>e</mi><mrow><mi>j</mi><mo>⋅</mo><msub><mi>ϕ</mi><mi>r</mi></msub></mrow></msup><mo>⋅</mo><mi>r</mi></math><img id="ib0031" file="imgb0031.tif" wi="63" he="4" img-content="math" img-format="tif"/></maths></p>
<p id="p0110" num="0110">The signal to be transmitted thus becomes: <maths id="math0032" num=""><math display="block"><msub><mi>x</mi><mi mathvariant="italic">res</mi></msub><mo>=</mo><mi>x</mi><mo>−</mo><mi>m</mi></math><img id="ib0032" file="imgb0032.tif" wi="24" he="4" img-content="math" img-format="tif"/></maths></p>
<p id="p0111" num="0111"><b>In</b> this way the directional signal can be reconstructed in the decoder and used for the purpose of rendering.</p>
<p id="p0112" num="0112">The directional signal generator 209 is coupled to a difference signal circuit 211 which is arranged to generate a difference signal which is indicative of the difference between the mono downmix audio signal and the directional signal. The difference signal circuit 211 is fed both the mono downmix audio signal and the directional signal and proceeds to determine a difference signal based on a comparison between these signals. In many embodiments, the difference signal may be generated directly by subtracting the mono downmix audio signal from the directional signal or vice versa.</p>
<p id="p0113" num="0113">It will be appreciated that in other embodiments other approaches may be used to generate a signal that reflects the difference between the mono downmix audio signal and the directional signal. For example, as will be described further later, instead of transmitting the difference signal:<!-- EPO <DP n="19"> --> <maths id="math0033" num=""><math display="block"><msub><mi>x</mi><mi mathvariant="italic">res</mi></msub><mo>=</mo><mi>x</mi><mo>−</mo><mi>m</mi></math><img id="ib0033" file="imgb0033.tif" wi="24" he="4" img-content="math" img-format="tif"/></maths> the difference between the directional signal and an approximation of the directional signal from the mono signal is transmitted: <maths id="math0034" num=""><math display="block"><msub><mi>x</mi><mi mathvariant="italic">res</mi></msub><mo>=</mo><mi>x</mi><mo>−</mo><mi>x</mi><mo>′</mo></math><img id="ib0034" file="imgb0034.tif" wi="24" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0114" num="0114">The approximation of the directional signal is realized by a (time- and frequency-variant) gain parameter applied to the mono signal: <maths id="math0035" num=""><math display="block"><mi>x</mi><mo>′</mo><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><mi>m</mi></math><img id="ib0035" file="imgb0035.tif" wi="20" he="5" img-content="math" img-format="tif"/></maths> with <maths id="math0036" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><msqrt><mfrac><msup><mfenced open="‖" close="‖"><mi>x</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>m</mi></mfenced><mn>2</mn></msup></mfrac></msqrt></math><img id="ib0036" file="imgb0036.tif" wi="22" he="10" img-content="math" img-format="tif"/></maths></p>
<p id="p0115" num="0115">For the case of the PCA directional signal, it can be shown that the gain parameter can be expressed again as a function of the inter-channel properties: <maths id="math0037" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><msqrt><mfrac><msqrt><mrow><mn>4</mn><mo>⋅</mo><msup><mi mathvariant="italic">ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi mathvariant="italic">IID</mi><mo>+</mo><msup><mfenced separators=""><mi mathvariant="italic">IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup></mrow></msqrt><mrow><mi mathvariant="italic">IID</mi><mo>+</mo><mn>1</mn></mrow></mfrac></msqrt></math><img id="ib0037" file="imgb0037.tif" wi="45" he="10" img-content="math" img-format="tif"/></maths></p>
<p id="p0116" num="0116">This means that the approximation of the directional signal can be made both at encoder side as well as decoder side without the need for transmitting additional parameters. The benefit of transmitting the difference of the directional signal and the approximation of the directional signal is that this signal has a significantly smaller variance than the difference of the directional signal and the mono signal. As a result the amount of bits to transmit is significantly smaller in the case of transmitting the difference of the directional signal and the approximation of the difference signal from the mono signal and the inter-channel properties.</p>
<p id="p0117" num="0117">The difference signal is fed to the data signal circuit 207 which is arranged to include it in the output data signal. The data signal circuit 207 typically includes a suitable encoder to generate encoded data representing the difference signal. In most cases, the encoding may be a non-lossy encoding, but in some applications and embodiments the encoding of the difference signal may be a lossy encoding.<!-- EPO <DP n="20"> --></p>
<p id="p0118" num="0118">There may typically be a strong correlation between the mono downmix audio signal and the directional signal and accordingly the difference signal may have a substantially lower power and amplitude than the mono downmix audio signal (and the directional signal). This may allow the encoding of the difference signal to have a much lower data rate than the mono downmix audio signal (or indeed that would be required to encode the difference signal with a corresponding quality). Indeed, the approach can be considered to generate an audio data signal representing a stereo signal by an encoded representation of a mono downmix audio signal and associated spatial parameters, and with the audio data signal additionally including an encoding of a difference signal supporting an encoding of a directional signal. Further, the overhead required to include encoding supporting the directional signal may be kept relatively low.</p>
<p id="p0119" num="0119">In many embodiments, an encoded data rate for the difference signal may be no more than 50%, 25%, 10%, 5%, or even 1% of the encoded data rate for the mono downmix audio signal.</p>
<p id="p0120" num="0120">In many embodiments, the audio source device 101 may accordingly be arranged to estimate the difference between the mono downmix audio signal and the directional signal, encode the difference at a low bit-rate, and transmit the data as a side channel to the audio render apparatus 103. In some embodiments, the left and right stereo input signals may for example be fed to two filterbanks, e.g. hybrid QMF banks, and the resulting left and right frequency domain signals may be used for stereo/spatial parameter estimation. Furthermore, the left and right frequency domain signals may be used to form a downmix signal and a directional signal (X), optionally making use of the extracted stereo parameters. The difference between the resulting mono downmix and the directional signal can be determined and both the mono downmix audio signal (<i>M</i>) and the difference between the mono downmix audio signal and the directional signal, the difference signal (<i>X<sub>res</sub> = X -</i> M) may be fed through a synthesis filterbank, resulting in time domain signals <i>m</i> and <i>x<sub>res</sub></i> respectively. The time domain signals can then be coded using existing techniques, such as waveform coding. Finally, bit-streams, consisting of the encoded mono downmix audio signal, the encoded spatial parameters, and the encoded differential signal can then be multiplexed into a single output bit-stream/data signal.</p>
<p id="p0121" num="0121">It will be appreciated that in many embodiments, the mono downmix audio signal used to generate the difference signal may be a decoded version generated by decoding the encoded mono downmix audio signal of the output data signal. This may provide improved performance in many embodiments and may allow encoding/decoding errors to be reflected in the difference signal thereby allowing them to be compensated in a directional signal generated by the audio render apparatus 103. Such an approach may be particularly advantageous in embodiments where the decoding operation of the audio render apparatus 103 can be known and replicated by the audio source device 101.</p>
<p id="p0122" num="0122"><figref idref="f0003">FIG. 3</figref> shows examples of elements of the audio render apparatus 103.</p>
<p id="p0123" num="0123">The audio render apparatus 103 comprises a receiver 301 which is arranged to receive the data signal from the audio source device 101. Thus, the receiver 301 receives a data signal comprising<!-- EPO <DP n="21"> --> encoded data for a mono downmix audio signal of a stereo signal. In addition, the data signal includes spatial upmix parameters for upmixing the mono downmix audio signal to the stereo signal where the spatial upmix parameters are indicative of relative signal properties of channels of the stereo signal. The spatial upmix parameters may as mentioned specifically be inter-channel time, phase, level, intensity differences and/or inter-channel correlation measures. Further, the data signal includes data describing a difference signal representing/describing a difference between the mono downmix audio signal and a directional signal where the directional signal represents a point source audio component of the stereo signal.</p>
<p id="p0124" num="0124">The receiver 301 is coupled to a decoder 303 which is arranged to receive the encoded data representing the mono downmix audio signal and to decode this data to generate the mono downmix audio signal. It will be appreciated that any suitable method for encoding and decoding the mono downmix audio signal may be used and in particular that any suitable standardized encoding format and algorithm may be used.</p>
<p id="p0125" num="0125">The receiver 301 may be arranged to receive a time domain audio signal and/or a frequency domain audio signal version/representation of the mono downmix audio signal. In some cases, the received data signal may include the mono downmix audio signal in only one representation, i.e. the data signal may include only one of the frequency domain audio signal and the time domain audio signal. In such cases, the received data signal may be transformed to the other domain as appropriate. Thus, in some cases, a received data signal may include a time domain audio signal being the time domain representation of the mono downmix audio signal, and a time to frequency domain transformer may from this generate the frequency domain audio signal for the mono downmix audio signal. In some cases, a received data signal may include a frequency domain audio signal being the frequency domain representation of the mono downmix audio signal and a frequency to time domain transformer may from this generate the time domain audio signal for the mono downmix audio signal if necessary.</p>
<p id="p0126" num="0126">In particular, in some embodiments, the receiver may comprise a filter bank which is arranged to generate a frequency subband representation of a received time domain mono downmix audio signal. The receiver 301 may comprise a filter bank that is applied to the mono downmix audio signal such that it is divided into frequency subbands.</p>
<p id="p0127" num="0127">The filter bank may be Quadrature Mirror Filter (QMF) bank or may e.g. be implemented by a Fast Fourier Transform (FFT), but it will be appreciated that many other filter banks and approaches for dividing an audio signal into a plurality of subband signals are known and may be used. The filterbank may specifically be a complex-valued pseudo QMF bank, resulting in e.g. 32 or 64 complex-valued sub-band signals.</p>
<p id="p0128" num="0128">The processing is furthermore typically performed in time segments or time slots. In most embodiments, the audio signal is divided into time intervals/segments with a conversion to the frequency/subband domain by applying e.g. an FFT or QMF filtering to the samples of each signal. For example, each channel of the downmix audio signal may be divided into time segments of e.g. 2048,<!-- EPO <DP n="22"> --> 1024, or 512 samples. These signals may then be processed to generate samples for e.g. 64, 32 or 16 subbands. Thus, a set of samples may be determined for each subband of the mono downmix audio signal.</p>
<p id="p0129" num="0129">It should be noted that the number of time domain samples is not directly coupled to the number of subbands. Typically, for a so-called critically sampled filterbank of N bands, every N input samples will lead to N sub-band samples (one for every sub-band). An oversampled filterbank will produce more output samples. E.g. for every N input samples, it would generate k*N output samples, i.e., k consecutive samples for every band.</p>
<p id="p0130" num="0130">In some embodiments, the subbands are generated to have the same bandwidth but in other embodiments subbands are generated to have different bandwidths, e.g. reflecting the sensitivity of human hearing to different frequencies.</p>
<p id="p0131" num="0131">For example, the receiver 301 may employ a hybrid filterbank with logarithmic filter band center-frequency spacings that follow that of human perception similar to equivalent rectangular bandwidths (ERBs). In order to compensate for the delay of the filtering by the small filter bank, a delay may be introduced for higher frequency subbands.</p>
<p id="p0132" num="0132">As a specific example, a time-domain signal <i>x</i>[<i>n</i>] may be fed through a downsampled complex-exponential modulated QMF bank with <i>K</i> bands. Each frame of 64 time domain samples <i>x</i>[<i>n</i>] results in one slot of QMF samples <i>X</i>[<i>k</i>, <i>l</i>] with <i>k</i> = (0, ..., <i>K</i> - 1) at slot <i>l.</i> The lower slots may then be filtered by additional complex-modulated filterbanks splitting the lower bands further. The higher slots are delayed ensuring that the filtered mono downmix audio signals of the lower bands are in sync with the higher bands as the filtering introduces a delay. This finally results in a structure where for every 64 time-domain samples <i>x</i>[<i>n</i>], one slot m of hybrid QMF samples <i>Y</i>[k, <i>l</i>] is produced with <i>k</i> = (0, ... , <i>L</i> - 1) at slot <i>l,</i> e.g. with a total number of hybrid bands <i>M</i> = 77.</p>
<p id="p0133" num="0133">Thus, in many embodiments, the signals and the processing may be performed in subbands and for individual segments. Such blocks of a frequency interval/subband in a given time interval/segment will also be referred to as time frequency segments/tiles.</p>
<p id="p0134" num="0134">The mono downmix audio signal is fed to a compensator 305 which also receives the difference signal. The compensator 305 is arranged to compensate/modify the mono downmix audio signal based on the difference signal and is specifically arranged to generate a modified mono downmix audio signal by modifying the mono downmix audio signal based on the difference signal such that it more closely resembles the directional signal. The compensator 305 may compensate the mono downmix audio signal based on the difference signal to generate a modified mono downmix audio signal having for which a difference to the directional signal is reduced, and thus the modified mono downmix audio signal is generated to more closely correspond to/resemble the directional signal.</p>
<p id="p0135" num="0135">As a specific example, the compensator 305 may in some embodiments simply add the difference signal to the mono downmix audio signal to generate the modified mono downmix audio<!-- EPO <DP n="23"> --> signal. Thus, in some embodiments, the audio apparatus 101 may directly generate the data signal to include (low rate) data that describes the difference between the mono downmix audio signal and the generated directional signal. The audio render apparatus 103 may then apply this difference to the received mono downmix audio signal to generate a modified mono downmix audio signal which is a local replica of the directional signal that was generated by the audio source device 101. This local replica generated by the audio source device 101 will correspond closely to the directional signal generated at the audio source device 101, and typically will only differ by potential distortions due to encoding/decoding and communication.</p>
<p id="p0136" num="0136">As will be described in more detail later, in other embodiments, instead of transmitting the difference between the directional signal and the mono downmix audio signal , the difference between the directional signal and an estimate of the directional signal generated from the mono downmix audio signal may be transmitted. As shown above, the directional signal can be estimated from the mono signal as <i>x'</i> = <i>g<sub>x</sub></i> · m, where the parameter <i>g<sub>x</sub></i> is a function of the inter-channel properties that are transmitted to the decoder. The difference signal <i>x - x'</i> generally has a lower variance than the difference signal <i>x -</i> m, thereby requiring less bits to be transmitted.</p>
<p id="p0137" num="0137">In yet another embodiment, instead of transmitting the difference between the directional signal and the estimated directional signal, the aforementioned gain <i>g<sub>x</sub></i> may be applied to the directional signal, and the difference <i>x</i>/<i>g<sub>x</sub> -</i> m may be transmitted.</p>
<p id="p0138" num="0138">The modified mono downmix audio signal is fed to a first renderer 307 which is arranged to render the modified mono downmix audio signal to generate a first intermediate stereo signal. The rendering by the first renderer 307 (also referred to as a first rendering) is a directional rendering which renders the first intermediate stereo signal with a given direction/position in the stereo image of the first intermediate stereo signal. The first rendering may specifically render the mono downmix audio signal as a point source with a given direction/position in the stereo image.</p>
<p id="p0139" num="0139">The first renderer 307 is coupled to a direction determining circuit 309 which is arranged to determine a direction γ' which is fed to the first renderer 307 resulting in this rendering the modified mono downmix audio signal from this position/direction. Thus, the first rendering is specifically such that the modified mono downmix audio signal in the first intermediate stereo signal is perceived as a point audio source positioned in the direction corresponding to the direction γ' determined by the direction determining circuit 309. The direction γ' will also be referred to as the rendering direction or rendering angle.</p>
<p id="p0140" num="0140">The direction determining circuit 309 is arranged to determine the direction from the received spatial upmix parameters. The spatial upmix parameters provide information on the relationship between the channels of the stereo signal that is downmixed and as such provide information of the position/orientation of the audio, and specifically of a dominant signal component in the stereo image of the stereo signal. For example, for a PS encoded signal, the spatial upmix parameters provide information<!-- EPO <DP n="24"> --> of the position of the dominant signal component in the stereo signal, and specifically it provides information of an orientation angle for the dominant signal.</p>
<p id="p0141" num="0141">The direction determining circuit 309 may specifically determine the rendering direction γ' from the spatial upmix parameters. The rendering direction will typically be determined on a frequency tile basis, and specifically in frequency subbands and time segments matching those for which the spatial upmix parameters are provided.</p>
<p id="p0142" num="0142">The first renderer 307 may accordingly proceed to render the modified mono downmix audio signal such that is perceived from the given direction and it specifically achieves this directional rendering by applying a channel transfer function to the modified mono downmix audio signal with the channel transfer function generating the intermediate stereo signal from the mono downmix audio signal. The channel transfer function may specifically include a sub-transfer function for each channel, i.e. it may include one (sub)transfer function for generating a left channel signal and one (sub)transfer function for generating the right channel signal.</p>
<p id="p0143" num="0143">In many cases, the transfer function may be provided as a set of complex weights for the different subbands of a frequency representation of the modified mono downmix audio signal. The audio apparatus may perform many or all of the operations in the frequency domain and thus the transfer function may also be expressed and applied in the frequency domain. For example, for each frequency subband of the representation of the mono downmix audio signal, the transfer function may provide a complex weight for each of the output channels and a frequency representation of the first intermediate stereo signal may be generated by applying/multiplying the subband samples of the mono downmix audio signal by these weights to generate the subband samples of the first intermediate stereo signal.</p>
<p id="p0144" num="0144">The first transfer function is determined to correspond to the desired direction, i.e. it reflects the mapping from the mono downmix audio signal to the channels of the first intermediate stereo signal such that it is perceived as/corresponds to an audio source at a position in the stereo image corresponding the rendering direction/angle.</p>
<p id="p0145" num="0145">For example, in some cases, the transfer function for a given direction may correspond to a panning of the mono downmix audio signal to the given direction in the stereo image.</p>
<p id="p0146" num="0146">In many embodiments, the first rendering may be a binaural rendering and the first intermediate stereo signal may be a binaural stereo signal providing an enhanced spatial experience/perception when heard through headphones. Thus, the first renderer 307 may specifically be a binaural audio renderer which generates binaural audio signals for the left and right ear of a user. Binaural audio signals are generated to provide a desired spatial experience and are typically reproduced by headphones or earphones that specifically may be part of a headset worn by a user (the headset typically also comprises left and right eye displays).</p>
<p id="p0147" num="0147">Thus, in many embodiments, the audio rendering by the first renderer 307 is a binaural render process using suitable binaural transfer functions to provide the desired spatial effect for a user<!-- EPO <DP n="25"> --> wearing a headphone. For example, the first renderer 307 may be arranged to generate an audio component to be perceived to arrive from a specific position using binaural processing.</p>
<p id="p0148" num="0148">Binaural processing is known to be used to provide a spatial experience by virtual positioning of sound sources using individual signals for the listener's ears. With an appropriate binaural rendering processing, the signals required at the eardrums in order for the listener to perceive sound from any desired direction can be calculated, and the signals can be rendered such that they provide the desired effect. These signals are then recreated at the eardrum using either headphones or a crosstalk cancelation method (suitable for rendering over closely spaced speakers). Binaural rendering can be considered to be an approach for generating signals for the ears of a listener resulting in tricking the human auditory system into perceiving that a sound is coming from the desired positions.</p>
<p id="p0149" num="0149">The binaural rendering is based on binaural transfer functions which vary from person to person due to the acoustic properties of the head, ears and reflective surfaces, such as the shoulders. Binaural transfer functions may therefore be personalized for an optimal binaural experience. For example, binaural filters can be used to create a binaural recording simulating multiple sources at various locations. This can be realized by convolving each sound source with the pair of e.g., Head Related Impulse Responses (HRIRs) that correspond to the position of the sound source.</p>
<p id="p0150" num="0150">A well-known method to determine binaural transfer functions is binaural recording. It is a method of recording sound that uses a dedicated microphone arrangement and is intended for replay using headphones. The recording is made by either placing microphones in the ear canal of a subject or using a dummy head with built-in microphones, a bust that includes pinnae (outer ears). The use of such dummy head including pinnae provides a very similar spatial impression as if the person listening to the recordings was physically present during the recording.</p>
<p id="p0151" num="0151">By measuring e.g., the responses from a sound source at a specific location in 2D or 3D space to microphones placed in or near the human ears, the appropriate binaural filters can be determined. Based on such measurements, binaural filters reflecting the acoustic transfer functions to the user's ears can be generated. The binaural filters can be used to create a binaural recording simulating multiple sources at various locations. This can be realized e.g., by convolving each sound source with the pair of measured impulse responses for a desired position of the sound source. In order to create the illusion that a sound source is moving around the listener, a large number of binaural filters is typically required with a certain spatial resolution, e.g., 10 degrees.</p>
<p id="p0152" num="0152">The head related binaural transfer functions may be represented e.g., as Head Related Impulse Responses (HRIR), or equivalently as Head Related Transfer Functions (HRTFs) or, Binaural Room Impulse Responses (BRIRs). The (e.g., estimated or assumed) transfer function from a given position to the listener's ears (or eardrums) may for example be represented in the frequency domain in which case it is typically referred to as an HRTF or BRTF, or in the time domain in which case it is typically referred to as a HRIR or BRIR. In some scenarios, the head related binaural transfer functions are determined to include aspects or properties of the acoustic environment and specifically of the<!-- EPO <DP n="26"> --> environment in which the measurements are made, whereas in other examples only the user characteristics are considered. Examples of the first type of functions are the BRIRs and BRTFs.</p>
<p id="p0153" num="0153">The audio render apparatus 103 comprises a store 311 which stores directional transfer functions for different directions. The directional transfer function for a given direction represents the mapping of a mono audio signal to stereo channels such that the mono audio signal is positioned in the given direction in a stereo image of the stereo channels. Thus, applying the directional transfer function for a given direction to the mono downmix audio signal may generate a stereo signal representing the mono downmix audio signal as an audio source positioned in the given direction. The mapping may in some cases be a time domain mapping (such as a gain, filter or other transfer function) or may in many cases be a frequency domain mapping, such as a set of parameter values/scale values (typically complex values) for different subbands. In the latter case, a frequency domain intermediate stereo signal may be generated by for each subband multiplying the subband sample of the mono downmix audio signal with respectively a complex value for that subband for a first channel of the intermediate stereo signal and with a complex value for that subband for a second channel of the intermediate stereo signal.</p>
<p id="p0154" num="0154">For example, in examples where a panning is performed in the horizontal 2D plane, the store 311 may comprise panning parameters for different directions. For example, panning parameters for azimuth angles in a 0-360° interval may be provided for each 1° angle increment. The first renderer 307 may be coupled to the store 311 and be arranged to extract the directional transfer function for the rendering direction and then proceed to perform the rendering using the extracted directional transfer function. The rendering of the (potentially gain compensated) mono downmix audio signal may accordingly be rendered such that it is positioned/perceived in the stereo image to arrive from the rendering position.</p>
<p id="p0155" num="0155">It will be appreciated that the store 311 may not have directional transfer function stored for the desired rendering direction. In such cases, the first renderer 307 may be arranged to retrieve the nearest directional transfer function from the store 311 and use this for rendering. In such cases, the rendering direction may be considered to correspond to the direction for the retrieved directional transfer function, i.e. the rendered direction may be a quantized value γ' of the desired rendering direction determined by the direction determining circuit 309.</p>
<p id="p0156" num="0156">In other embodiments, the first renderer 307 may be arranged to estimate a desired directional transfer function for a desired rendering direction by interpolating between two directional transfer functions from the store 311 corresponding to the two rendering angles nearest to the desired rendering direction determined by the direction determining circuit 309.</p>
<p id="p0157" num="0157">In most embodiments, the first renderer 307 is as mentioned arranged to perform a binaural rendering and the directional transfer functions stored in the store 311 are binaural transfer functions. Thus, the store may store data describing binaural transfer functions for different directions. The binaural transfer functions may for example be HRTFs, BRIRs, or HRIRs. The store 311 may specifically store frequency subband complex values for each channel for each frequency subband for a<!-- EPO <DP n="27"> --> range of different frequencies. The first renderer 307 may thus perform the binaural rendering by multiplying the subband samples of the mono downmix audio signal with the corresponding subband coefficients/complex values of the selected binaural transfer function to generate subband sample values of the intermediate binaural stereo signal.</p>
<p id="p0158" num="0158">It will be appreciated that in many embodiments, the directional transfer functions may be stored as a plurality of functions linked with different directions. For example, the store 311 may be a look-up table which can receive the rendering direction as an index and provide a set of values of the directional transfer function for that direction. The directional transfer function may for example be represented by individual subband values/coefficients, or may e.g. in other embodiments be represented by e.g. parameter values defining the directional transfer function operation (e.g. coefficients for the transfer function), a mathematical description/function from which suitable values of the transfer function can be generated etc.</p>
<p id="p0159" num="0159">Thus, the audio render apparatus 103 comprises a processing path which generates an intermediate stereo signal comprising the modified mono downmix audio signal represented as an audio source at a specific position in the spatial image of the first intermediate stereo signal. The mono downmix audio signal may typically be represented as a point audio source at the given direction. The rendering is adaptive with the direction being given by the spatial parameters and thus is dynamically adapted to reflect the characteristics of the stereo signal.</p>
<p id="p0160" num="0160">Further, the spatially definite and directional rendering is specifically of a directional signal representing point source audio. Thus, an improved spatial rendering of point source audio can be achieved with point source audio typically being rendered with a higher audio quality.</p>
<p id="p0161" num="0161">The generated intermediate stereo signal is fed to an output circuit 313 which is arranged to generate an output stereo signal which includes the intermediate stereo signal. As will be described more in the following, the output stereos signal may be generated to include other signal components, including audio components representing non-point source audio, such as background or ambient sounds.</p>
<p id="p0162" num="0162">In particular, the audio render apparatus 103 comprises a second processing path which generates a second intermediate stereo signal. The mono downmix audio signal is fed to a decorrelator 315 which is arranged to apply a decorrelation to the mono downmix audio signal to generate a first decorrelated mono downmix audio signal. It will be appreciated that a large number of different algorithms and functions for decorrelating an audio signal is known to the skilled person, and that any suitable approach or algorithm may be used without detracting from the invention.</p>
<p id="p0163" num="0163">The first decorrelated mono downmix audio signal is fed to a second renderer 317 which is arranged to perform a second rendering being a rendering of the first decorrelated mono downmix audio signal to generate a second intermediate stereo signal. However, in contrast to the first rendering process, the second rendering process is a predetermined rendering which is not dependent on the spatial parameters, and which typically is not depending on properties of the stereo signal. The second rendering may typically be a diffuse rendering seeking to generate the second intermediate stereo signal to provide a<!-- EPO <DP n="28"> --> perception of a more diffuse and spatially less definite audio source. The second rendering is specifically a predetermined rendering employing a predetermined mapping of the decorrelated mono downmix audio signal to channel signals of the second intermediate stereo signal.</p>
<p id="p0164" num="0164">As a specific example, the second rendering may typically generate the second intermediate stereo signal by simply mapping the first decorrelated mono downmix audio signal to two phase inverse signals, i.e. the second intermediate stereo signal may be generated with the first decorrelated mono downmix audio signal being mapped to both channels but with a 180° phase offset between them (the first decorrelated mono downmix audio signal may specifically be inverted for one of the channels). For example, in some embodiments, the first decorrelated mono downmix audio signal may be mapped to the right and left signals of the second intermediate stereo signal but with the mapping being 180° out of phase for the two channels of the second intermediate stereo signal.</p>
<p id="p0165" num="0165">The first renderer 307 and second renderer 317 are coupled to a combiner 313 which is arranged to combine at least the first intermediate stereo signal and the second intermediate stereo signal to generate an output stereo signal. In many embodiments, the combiner 313 may be arranged to combine/sum the samples/values of the individual channels of the first and second intermediate stereo signals to generate the samples/values of the output stereo signal. In many cases, the combination may be performed by combining/summing subband values of the intermediate stereo signals. In other embodiments, the combination may be performed in the time domain by combining/summing time domain values of the intermediate stereo signals.</p>
<p id="p0166" num="0166">In many embodiments, the combination of the intermediate stereo signals may be by a (possibly weighted) combination/summation of corresponding channel signals for the first intermediate stereo signal and the second intermediate stereo signal.</p>
<p id="p0167" num="0167">The audio render apparatus 103 accordingly generates an output stereo signal which is the combination of a directional rendering putting an audio source at a desired position as determined from the received spatial upmix parameters, and of a predetermined rendering providing a more diffuse and decorrelated perception of the corresponding audio source. The approach provides two parallel rendering processes/paths for the mono downmix audio signal with the rendered results being combined to generate the output stereo signal.</p>
<p id="p0168" num="0168">In addition to the described flexible and adaptable generation of the output signal to provide an output stereo signal that includes both a directionally rendered component and a more diffuse/predeterminedly rendered component, the audio render apparatus 103 is arranged to flexibly and adaptively adapt/control the relative level between these components.</p>
<p id="p0169" num="0169">An example of such an approach is illustrated in <figref idref="f0003">FIG. 3</figref>. In addition to the audio render apparatus of <figref idref="f0002">FIG. 2</figref>, this apparatus includes a second path for rendering the diffuse/ non-directional component.</p>
<p id="p0170" num="0170">In many embodiments, the audio source device 101 may be arranged to generate multiple signals representing non-directional signal components. In particular, the decorrelator 315 may be<!-- EPO <DP n="29"> --> arranged to generate a second decorrelated mono downmix audio signal which is then rendered by the second renderer 317 generating a third intermediate stereo signal. The third intermediate stereo signal is then combined with the first intermediate stereo signal and the second intermediate stereo signal.</p>
<p id="p0171" num="0171">The second decorrelated mono downmix audio signal is generated to have low correlation with both the mono downmix audio signal and the first decorrelated mono downmix audio signal, and typically will be uncorrelated with both of these. Accordingly, the output stereo signal includes two components for the residual signal. The two signals may be rendered differently such that an improved diffuseness is perceived. For example, the second decorrelated mono downmix audio signal may be rendered from one position (e.g. fully in a first channel of the output stereo signal or from a first virtual speaker position for a binaural signal) with the first decorrelated mono downmix audio signal being rendered from a second position (e.g. fully in a second channel of the output stereo signal or from a second virtual speaker position for a binaural signal).</p>
<p id="p0172" num="0172">The approach may typically provide an improved performance and perception relative to what can be achieved with a single decorrelated signal representing the non-directional component. In particular, the approach may more accurately represent the signal model indicated previously with different and uncorrelated diffuse signal components <i>n<sub>l</sub></i> and <i>n<sub>r</sub>.</i></p>
<p id="p0173" num="0173">The predetermined rendering of the second renderer 317 may as previously mentioned simply be achieved by rendering the corresponding decorrelated signal in one channel of the corresponding intermediate stereo signal, and with no signal being included in the other channel. For example, the second decorrelated mono downmix audio signal may be rendered in the left channel of the second intermediate stereo signal and the first decorrelated mono downmix audio signal may be rendered in the right channel of the third intermediate stereo signal.</p>
<p id="p0174" num="0174">In some embodiments where binaural processing is used, each of the decorrelated signals may be rendered from a specific position, such as each decorrelated signal being rendered from a different virtual position, such as for example from different virtual positions.</p>
<p id="p0175" num="0175">In some embodiments, the rendering for a decorrelated signal, such as the rendering of the first decorrelated mono downmix audio signal, may be to position the signal at a specific position.</p>
<p id="p0176" num="0176">In many embodiments, the rendering of a decorrelated mono downmix audio signal may be performed by the renderer 317 retrieving a set of directional transfer functions from the store 311 and rendering the decorrelated mono downmix audio signal using the retrieved transfer function(s).</p>
<p id="p0177" num="0177">In many embodiments, the audio render apparatus 103 may be arranged to extract a directional transfer function for a single predetermined direction and render the decorrelated mono downmix audio signal using this directional transfer function. Accordingly, (each of) the decorrelated mono downmix audio signal(s) may be rendered from one predetermined direction/position, such as a direction/position corresponding to a virtual speaker position.</p>
<p id="p0178" num="0178">An example of subband parametric rendering may e.g. result in left and right signals:<!-- EPO <DP n="30"> --> <maths id="math0038" num=""><math display="block"><mi>l</mi><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><msub><mi>m</mi><mi>d</mi></msub><mo>⋅</mo><msub><mi>G</mi><mi>l</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced></mrow></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>1</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mi>l</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>l</mi></msub></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>l</mi></msub></mfenced></mrow></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>2</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mi>l</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>r</mi></msub></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>r</mi></msub></mfenced></mrow></msup></math><img id="ib0038" file="imgb0038.tif" wi="164" he="6" img-content="math" img-format="tif"/></maths> <maths id="math0039" num=""><math display="block"><mtable columnalign="left"><mtr><mtd><mi>r</mi><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><msub><mi>m</mi><mi>d</mi></msub><mo>⋅</mo><msub><mi>G</mi><mi>r</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced></mrow></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>1</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mi>r</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>l</mi></msub></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>l</mi></msub></mfenced></mrow></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>2</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mi>r</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>r</mi></msub></mfenced></mtd></mtr><mtr><mtd><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub><mfenced open="[" close="]"><msub><mi>β</mi><mi>r</mi></msub></mfenced></mrow></msup></mtd></mtr></mtable></math><img id="ib0039" file="imgb0039.tif" wi="151" he="13" img-content="math" img-format="tif"/></maths> where <i>G<sub>l</sub>, G<sub>r</sub>, ϕ<sub>l</sub>, ϕ<sub>r</sub></i> form the parametric HRIRs, <i>f</i>(<i>γ</i>) is a mapping function converting the estimated angles (orientation direction y) to HRIR direction angles, <i>β<sub>l</sub></i> and <i>β<sub>r</sub></i> are two pre-determined angles and <i>H</i><sub>1</sub>{.} and <i>H</i><sub>2</sub>{.} are two mutually independent decorrelators, m is the received mono downmix audio signal and <i>m<sub>d</sub></i> is the directional signal generated by the compensator 305, and <i>g<sub>x</sub></i> and <i>g<sub>n</sub></i> are suitable scaling factors.</p>
<p id="p0179" num="0179">In some embodiments, the renderer 217 may retrieve directional transfer functions for a plurality of predetermined directions and it may use multiple directional transfer functions in performing the predetermined rendering. For example, different directional transfer functions may be used for different frequency subbands. This may provide a more diffuse perception with the audio being generated such that it is perceived from different directions for different subbands thereby resulting in a perception of a more distributed and spread audio source.</p>
<p id="p0180" num="0180">Such approaches may be used both in embodiments in which a single decorrelated mono downmix audio signal is generated and rendered, or indeed in cases where multiple decorrelated mono downmix audio signals are generated and rendered. In the latter case, the sets of predetermined directions for the different decorrelated mono downmix audio signals are different in order to enhance the perceived diffuseness of the non-directional signal component.</p>
<p id="p0181" num="0181">The renderer 317 may generate the second intermediate stereo signal using a first set of directional transfer functions retrieved from the store 311 for a first set of predetermined directions, and may generate the third intermediate stereo signal using a second set of directional transfer functions retrieved from the store 311 for a second set of predetermined directions where the first set of set of predetermined directions is different from the second set of predetermined directions.</p>
<p id="p0182" num="0182">In particular, the directional transfer functions may be binaural transfer functions and the renderer 317 may be arranged to perform binaural rendering to generate the first intermediate stereo signal using binaural impulse response values for a first set of predetermined directions and may be arranged to perform binaural rendering to generate the first intermediate stereo signal using binaural impulse response values for a second set of predetermined directions where the first set of predetermined directions are different from the second set of predetermined directions.</p>
<p id="p0183" num="0183">In many cases, the use of multiple directional transfer functions may be achieved by using directional transfer functions for different directions in different frequency subbands.</p>
<p id="p0184" num="0184">Thus, instead of rendering the diffuse/ non-directional signals using fixed angles, e.g. mimicking a virtual stereo speaker setup, the diffuse signals may also be rendered using composite, e.g.<!-- EPO <DP n="31"> --> pre-calculated HRIRs for many sources/directions, e.g. spread over a (part of a) circle, or (part of) a sphere. <maths id="math0040" num=""><math display="block"><mtable columnalign="left"><mtr><mtd><mi>l</mi><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><msub><mi>m</mi><mi>d</mi></msub><mo>⋅</mo><msub><mi>G</mi><mi>l</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced></mrow></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>1</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mrow><mi>l</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub><mo>⋅</mo><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mrow><mi>l</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>2</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mrow><mi>l</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub></mtd></mtr><mtr><mtd><mo>⋅</mo><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mrow><mi>l</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub></msup></mtd></mtr></mtable></math><img id="ib0040" file="imgb0040.tif" wi="154" he="13" img-content="math" img-format="tif"/></maths> <maths id="math0041" num=""><math display="block"><mtable columnalign="left"><mtr><mtd><mi>r</mi><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><msub><mi>m</mi><mi>d</mi></msub><mo>⋅</mo><msub><mi>G</mi><mi>r</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub><mfenced open="[" close="]" separators=""><mi>f</mi><mfenced><mi>γ</mi></mfenced></mfenced></mrow></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>1</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mrow><mi>r</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub><mo>⋅</mo><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mrow><mi>r</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub></msup><mo>+</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><msub><mi>H</mi><mn>2</mn></msub><mfenced open="{" close="}"><mi>m</mi></mfenced><mo>⋅</mo><msub><mi>G</mi><mrow><mi>r</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub></mtd></mtr><mtr><mtd><mo>⋅</mo><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mrow><mi>r</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub></msup></mtd></mtr></mtable></math><img id="ib0041" file="imgb0041.tif" wi="157" he="13" img-content="math" img-format="tif"/></maths> where e.g.: <maths id="math0042" num=""><math display="block"><msub><mi>G</mi><mrow><mi>l</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub><mo>=</mo><msub><mi>g</mi><mi mathvariant="italic">norm</mi></msub><mo>⋅</mo><mfenced open="|" close="|" separators=""><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mi>β</mi><mo>∈</mo><msub><mi mathvariant="normal">B</mi><mi>l</mi></msub></mrow></msub><msub><mi>G</mi><mi>l</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mstyle><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mrow></msup></mfenced></math><img id="ib0042" file="imgb0042.tif" wi="69" he="6" img-content="math" img-format="tif"/></maths> <maths id="math0043" num=""><math display="block"><msub><mi>ϕ</mi><mrow><mi>l</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub><mo>=</mo><mo>∠</mo><mfenced open="{" close="}" separators=""><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mi>β</mi><mo>∈</mo><msub><mi mathvariant="normal">B</mi><mi>l</mi></msub></mrow></msub><msub><mi>G</mi><mi>l</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mstyle><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mrow></msup></mfenced></math><img id="ib0043" file="imgb0043.tif" wi="60" he="6" img-content="math" img-format="tif"/></maths> <maths id="math0044" num=""><math display="block"><msub><mi>G</mi><mrow><mi>r</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub><mo>=</mo><msub><mi>g</mi><mi mathvariant="italic">norm</mi></msub><mo>⋅</mo><mfenced open="|" close="|" separators=""><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mi>β</mi><mo>∈</mo><msub><mi mathvariant="normal">B</mi><mi>r</mi></msub></mrow></msub><msub><mi>G</mi><mi>r</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mstyle><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mrow></msup></mfenced></math><img id="ib0044" file="imgb0044.tif" wi="71" he="6" img-content="math" img-format="tif"/></maths> <maths id="math0045" num=""><math display="block"><msub><mi>ϕ</mi><mrow><mi>r</mi><mo>,</mo><mi mathvariant="italic">comp</mi></mrow></msub><mo>=</mo><mo>∠</mo><mfenced open="{" close="}" separators=""><mstyle displaystyle="true"><msub><mo>∑</mo><mrow><mi>β</mi><mo>∈</mo><msub><mi mathvariant="normal">B</mi><mi>r</mi></msub></mrow></msub><msub><mi>G</mi><mi>r</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mstyle><mo>⋅</mo><msup><mi>e</mi><mrow><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub><mfenced open="[" close="]"><mi>β</mi></mfenced></mrow></msup></mfenced></math><img id="ib0045" file="imgb0045.tif" wi="61" he="6" img-content="math" img-format="tif"/></maths> with B<i><sub>l</sub></i> being a set of angles at which the left diffuse signal is to be rendered, B<i><sub>r</sub></i> a set of angles at which the right diffuse signal is to be rendered, and <i>g<sub>norm</sub></i> a normalisation factor.</p>
<p id="p0185" num="0185">In some embodiments, the diffuse signal component may be directly rendered onto left and right channels without any HRIR processing.</p>
<p id="p0186" num="0186">In some embodiments where only a single decorrelation path is used, the left and right diffuse signals may be approximated by an out-of-phase approximation, such as <maths id="math0046" num=""><math display="block"><msubsup><mi>n</mi><mi>l</mi><mo>′</mo></msubsup><mo>=</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><mi>H</mi><mfenced open="{" close="}"><mi>m</mi></mfenced></math><img id="ib0046" file="imgb0046.tif" wi="27" he="5" img-content="math" img-format="tif"/></maths> <maths id="math0047" num=""><math display="block"><msubsup><mi>n</mi><mi>r</mi><mo>′</mo></msubsup><mo>=</mo><mo>−</mo><msub><mi>g</mi><mi>n</mi></msub><mo>⋅</mo><mi>H</mi><mfenced open="{" close="}"><mi>m</mi></mfenced></math><img id="ib0047" file="imgb0047.tif" wi="31" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0187" num="0187">As described, the operation may typically be performed in a frequency domain representation and most or all of the described processing may be performed on a subband basis. In particular, the spatial upmix parameters may be provided for frequency subbands and the first renderer 307 may be arranged to generate subband values for subbands of the first intermediate signals from subband values of the mono downmix audio signal based on spatial upmix parameters and directional transfer functions for the subbands. The processing may be performed in time intervals/segments with a<!-- EPO <DP n="32"> --> time interval/segment of the output stereo signal being generated for the corresponding time interval/segment of the mono downmix audio signal.</p>
<p id="p0188" num="0188">Similarly, in many embodiments, the second renderer is arranged to generate subband values for subbands of the second intermediate signals from subband values of the decorrelated mono downmix audio signal, and specifically based on spatial upmix parameters and directional transfer functions for the subbands (if these are used).</p>
<p id="p0189" num="0189">The output signal may specifically be a binaural stereo audio signal and the audio render apparatus 103 may be arranged to perform a binaural rendering where an improved point source audio rendering can be achieved by generating a local directional signal from the received mono downmix audio signal and a received difference signal (<i>X̂<sub>res</sub></i>)<i>.</i> Specifically, a received bit-stream/data signal may be de-multiplexed into a bit-stream for the mono downmix audio signal, the spatial parameters, and the difference signal. The mono downmix audio signal and the differential signal may be decoded to the time domain and transformed to the frequency domain. The differential signal may be added to the mono downmix audio signal to generate the directional signal <i>X̂</i> = <i>X̂'</i> + <i>X̂<sub>res</sub>.</i> The resulting signal may be rendered, and specifically may be rendered as an audio point source. This point source audio may further (optionally) be supplemented by rendering other audio that may specifically represent non-point source audio, such as background or ambient sounds. These signal components may be generated from decorrelated signals generated from the mono downmix audio signal.</p>
<p id="p0190" num="0190">The audio apparatus may accordingly be arranged to generate an output stereo signal, and often an output binaural stereo signal from the received mono downmix audio signal, spatial upmix parameters, and a difference signal. The audio apparatus specifically implements two different rendering paths with one being a directional (binaural) rendering of a directional (e.g. a dominant) signal component which is generated from the mono downmix audio signal and the difference signal. The other rendering path may employ a predetermined rendering/mapping of a decorrelated audio signal generated from the mono downmix audio signal. The rendering of the output stereo signal is typically not a conventional adaptive upmixing of the received and decorrelated mono signals, and is specifically not a conventional 2x2 matrix upmixing of the mono signal and a decorrelated signal, but rather is a direct generation of a stereo signal by parallel processing of respectively the mono downmix audio signal and one or more decorrelated versions of this, with the former rendering being directional dependent on the spatial upmix parameters and the latter rendering being a predetermined rendering.</p>
<p id="p0191" num="0191">The processing seeks to render direct/dominant/directional point source components using a direct rendering with a direction that is given by the spatial parameters. The rendering employs a directionally dependent transfer function to the left and right stereo output signal for that purpose. The approach further seeks to render a residual/remaining signal component as a more diffuse signal, and specifically it uses a predetermined rendering where a decorrelated signal is mapped directly to the channels of the output binaural signal using a transfer function. The mapping is predetermined and may<!-- EPO <DP n="33"> --> specifically be such that it allows a more diffuse and non-directional perception of this signal component. The rendering process thus uses fundamentally different approaches to provide different signal components in the output binaural signal, but does so without specifically decomposing the mono downmix audio signal into a dominant and diffuse/residual signal component with these subsequently being individually rendered. Rather a direct rendering of respectively the mono downmix audio signal and a decorrelated version thereof is performed to generate the binaural output signal.</p>
<p id="p0192" num="0192">The approach allows low complexity and computationally efficient rendering of a stereo signal encoded as a downmix and spatial upmix parameters, such as a PS encoded signal. It may further allow high performance rendering with a perceived improved audio quality. In many cases, a substantially improved spatial perception and user experience may be achieved, and indeed can be provided using headphones. The approach may in many cases provide a user perception of an audio scene where individual/dominant audio sources are well defined at specific directions/positions in a stereo image whereas other sources (e.g. ambient or background audio sources) are perceived more diffuse.</p>
<p id="p0193" num="0193">The approach may typically allow a very efficient operation and rendering with reduced complexity. A particular advantage of the approach is that it does not require a decomposition of the received mono downmix audio signal into different components with different specific properties.</p>
<p id="p0194" num="0194">In many embodiments, the difference signal is generated to directly describe or define the difference between the directional signal and the mono downmix audio signal, such as by directly corresponding to the difference between these.</p>
<p id="p0195" num="0195">However, in some embodiments, the difference signal may be generated to be indicative of a difference between the directional signal and a prediction/estimate of the directional signal where the prediction/estimate of the directional signal is determined as a function of (at least) the mono downmix audio signal and the spatial parameters. The function may typically be a predetermined function that when applied to the mono downmix audio signal and the spatial parameters will provide an output of an estimated/predicted directional signal. The predetermined function may in many embodiments be applied in the frequency domain. The estimated directional signal may then be compared to the generated directional signal to generate the difference signal. Specifically, the difference signal may be generated as the difference between the estimated/predicted directional signal and the directional signal.</p>
<p id="p0196" num="0196">At the audio render apparatus 103, the compensator 305 may be arranged to perform the corresponding operation to generate a local estimate of the directional signal. It can apply the, typically predetermined, function to the received mono downmix audio signal and spatial parameters to generate the local replica of the difference signal. The audio source device 101 and the audio render apparatus 103 may be arranged to apply identical functions resulting in (at least substantially) the same estimated directional signal being generated at the audio render apparatus 103 as is generated by the audio source device 101. The function may be predetermined and may for example be defined/specified by a suitable audio encoding standard, or it may e.g. otherwise be known/shared by the audio source device 101 and the audio render apparatus 103. In some embodiments, a fixed predetermined function may be specified<!-- EPO <DP n="34"> --> by the audio source device 101 and included in the data signal such that the audio source device 101 can reconstruct the function and apply it to the received mono downmix audio signal as part of the rendering process.</p>
<p id="p0197" num="0197">The compensator 305 may proceed to generate the modified mono downmix audio signal by compensating the predicted signal as a function of the difference signal. Specifically, in many embodiments, the compensator 305 may generate the modified mono downmix audio signal by adding the difference signal to the generated local estimate of the directional signal.</p>
<p id="p0198" num="0198">In many embodiments, instead of directly calculating the difference between the mono signal and the directional signal, the audio source device 101 may generate an estimate of the directional signal which may be used to reduce the differences that are to be represented by the difference signal thereby allowing a lower data rate for a given audio quality. In such cases, the difference between the directional signal x and the approximation/estimate x' may specifically be encoded by the difference signal: <maths id="math0048" num=""><math display="block"><msub><mover accent="true"><mi>X</mi><mo>^</mo></mover><mi mathvariant="italic">res</mi></msub><mo>=</mo><mi>X</mi><mo>−</mo><mi>X</mi><mo>′</mo></math><img id="ib0048" file="imgb0048.tif" wi="25" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0199" num="0199">The complementary operation may be performed by the audio render apparatus 103 using the modified differential signal (<i>X̂<sub>res</sub></i>)<i>.</i> Using the spatial parameters, such as specifically PS parameters, an estimate of the directional signal <i>X̂'</i> is made and the differential signal <i>X̂<sub>res</sub></i> is added, resulting in the modified mono downmix audio signal corresponding to the directional signal estimate <i>X̂</i> = <i>X̂' + X̂<sub>res</sub>.</i> The modified mono downmix audio signal corresponding to the resulting directional signal estimate <i>X̂</i> may be rendered by a directional rendering. Further, decorrelated signals may be generated from the modified mono downmix audio signal/directional signal estimate <i>X̂</i> (or possibly directly from the received mono downmix audio signal) and rendered as more diffuse and spatially non-specific audio.</p>
<p id="p0200" num="0200">A specific example of a block diagram of an implementation of the audio renderer device 103 which follows such an approach is illustrated in <figref idref="f0004">FIG.4</figref>. In the example, the received bit-stream/data signal is de-multiplexed into a bit-stream for the mono downmix audio signal (the lower path), the stereo parameters (the middle path) and the difference signal (the upper path). The mono downmix audio signal and the difference signal are decoded to the time domain and transformed to the frequency domain by filter banks FB. Using the spatial (e.g. PS) parameters, an estimate of the directional signal <i>X̂'</i> is made. The difference signal is added to the directional signal <i>X̂',</i> resulting in the directional signal <i>X̂</i> = <i>X̂'</i> + <i>X̂<sub>res</sub>.</i> From the directional signal, (or e.g. directly from the mono downmix audio signal) one or more residual components may be estimated. Finally, rendering of the components is performed as previously described after which left and right binaural signals are synthesized by a conversion to the time domain using complementary filter banks FB<sup>-1</sup>.<!-- EPO <DP n="35"> --></p>
<p id="p0201" num="0201">It will be appreciated that different approaches, algorithms, and functions may be used to estimate the directional signal from the mono downmix audio signal using the spatial parameters.</p>
<p id="p0202" num="0202">For example, a rudimentary estimate of the directional signal from the mono signal may follow from: <i>x' = ICC ·</i> m. This example may follow the consideration that for fully decorrelated signals all of the signal power of the original stereo signal is captured by the directional signal, whereas for fully decorrelated signals none of the signal power of the original stereo signal is captured by the directional signal.</p>
<p id="p0203" num="0203">In a particular approach, the directional signal is estimated as a spectro-temporally shaped version of the received mono downmix audio signal, where the shaping is done in such a way that it approximates the signal power of the directional signal of the signal model as previously described, i.e. where the original stereo signal is considered to represent a main component x and residual/more diffuse component n (which typically has a component for each channel, i.e. n can be considered a stereo component): <maths id="math0049" num=""><math display="block"><mi>l</mi><mo>=</mo><mi>cos</mi><mfenced><mi>γ</mi></mfenced><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mi>l</mi></msub></msup><mi>x</mi><mo>+</mo><msub><mi>n</mi><mi>l</mi></msub></math><img id="ib0049" file="imgb0049.tif" wi="32" he="5" img-content="math" img-format="tif"/></maths> <maths id="math0050" num=""><math display="block"><mi>r</mi><mo>=</mo><mi>sin</mi><mfenced><mi>γ</mi></mfenced><msup><mi>e</mi><msub><mi mathvariant="italic">jϕ</mi><mi>r</mi></msub></msup><mi>x</mi><mo>+</mo><msub><mi>n</mi><mi>r</mi></msub></math><img id="ib0050" file="imgb0050.tif" wi="33" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0204" num="0204">The audio render apparatus 103 may seek to generate an estimate of the directional signal component x by scaling of the mono downmix audio signal (with the scaling typically being separate for different frequency band). The estimated direct signal component may be represented by: <maths id="math0051" num=""><math display="block"><mi>x</mi><mo>′</mo><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><mi>m</mi><mo>=</mo><msqrt><mfrac><msup><mfenced open="‖" close="‖"><mi>x</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>m</mi></mfenced><mn>2</mn></msup></mfrac></msqrt><mi>m</mi></math><img id="ib0051" file="imgb0051.tif" wi="41" he="12" img-content="math" img-format="tif"/></maths> where the resulting approximation x' is the estimate of the directional signal in a given subband and g<sub>x</sub> is determined from the spatial parameters, and specifically with <maths id="math0052" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><mi>f</mi><mfenced><mi mathvariant="italic">IID</mi><mi mathvariant="italic">ICC</mi><mi mathvariant="italic">IPD</mi></mfenced></math><img id="ib0052" file="imgb0052.tif" wi="40" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0205" num="0205">In this case, the directional signal may (in the frequency/subband domain) be determined as: <maths id="math0053" num=""><math display="block"><mover accent="true"><mi>X</mi><mo>^</mo></mover><mo>=</mo><mover accent="true"><mi>X</mi><mo>^</mo></mover><mo>′</mo><mo>+</mo><msub><mover accent="true"><mi>X</mi><mo>^</mo></mover><mi mathvariant="italic">res</mi></msub><mo>=</mo><msub><mi>g</mi><mi>x</mi></msub><mo>⋅</mo><mover accent="true"><mi>M</mi><mo>^</mo></mover><mo>+</mo><msub><mover accent="true"><mi>X</mi><mo>^</mo></mover><mi mathvariant="italic">res</mi></msub><mo>.</mo></math><img id="ib0053" file="imgb0053.tif" wi="56" he="5" img-content="math" img-format="tif"/></maths><!-- EPO <DP n="36"> --></p>
<p id="p0206" num="0206">The spatial parameters are indicative of the relative properties of the channel signals for the stereo signal and, as has been realized by the inventors, this may also provide information on the relative levels for respectively a directional component (and for a non-directional component). The compensator 305 may specifically determine a scaling factor that scales the effective gain of subbands of the mono downmix audio signal to generate the directional signal estimate.</p>
<p id="p0207" num="0207">In many embodiments, the compensator 305 is accordingly arranged to generate the gains dependent on the spatial parameters. In particular, the gain for the directional rendering may be set in dependence on a relative power level of a directional component of the stereo signal relative to a power level of the mono downmix audio signal. The spatial upmix parameters provide information on the relative properties of the channel signals of the stereo signal, and specifically may provide information on both the interchannel levels/intensity differences as well as on the interchannel correlation. Accordingly, the spatial upmix parameters can be considered to provide information on the directional signal component x and the diffuse/residual component n. Accordingly, the gains <i>g<sub>x</sub></i> may be estimated/calculated from the received spatial parameters.</p>
<p id="p0208" num="0208">The gains may in many embodiments be determined to ensure power preservation. For the indicated signal model of the stereo signal, the gains for an actual point source with no other audio being present may be 1 for the directional rendering path and 0 for the predetermined/diffuse rendering path. For a completely diffuse signal, the gains would be the opposite, i.e. 1 for the predetermined rendering path and 0 for the directional rendering path. <b>In</b> practice, when two diffuse signals are considered, one for the left channel and one for the right channel, the gains would typically be <maths id="math0054" num=""><math display="inline"><mn>1</mn><mo>/</mo><msqrt><mn>2</mn></msqrt></math><img id="ib0054" file="imgb0054.tif" wi="10" he="6" img-content="math" img-format="tif" inline="yes"/></maths>.</p>
<p id="p0209" num="0209">The gains for estimating the directional signal may in particular in many embodiments advantageously be determined in line with one or more of the following: <maths id="math0055" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><msqrt><mfrac><msqrt><mrow><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup><mo>+</mo><mn>4</mn><msup><mi>ICC</mi><mn>2</mn></msup><mi>IID</mi></mrow></msqrt><mrow><mi>IID</mi><mo>+</mo><mn>1</mn></mrow></mfrac></msqrt></math><img id="ib0055" file="imgb0055.tif" wi="56" he="14" img-content="math" img-format="tif"/></maths> <maths id="math0056" num=""><math display="block"><msub><mi>g</mi><mi>x</mi></msub><mo>=</mo><msqrt><mfrac><mrow><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup><mo>+</mo><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi><mo>+</mo><mfenced separators=""><mn>1</mn><mo>−</mo><mi>IID</mi></mfenced><mo>⋅</mo><msqrt><mrow><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi><mo>+</mo><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup></mrow></msqrt></mrow><mrow><mn>1</mn><mo>−</mo><msup><mi>IID</mi><mn>2</mn></msup><mo>+</mo><mfenced separators=""><mi>IID</mi><mo>+</mo><mn>1</mn></mfenced><mo>⋅</mo><msqrt><mrow><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi><mo>+</mo><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup></mrow></msqrt></mrow></mfrac></msqrt></math><img id="ib0056" file="imgb0056.tif" wi="126" he="19" img-content="math" img-format="tif"/></maths> where IID is an interchannel intensity difference and ICC is an inter-channel cross-correlation, and specifically<!-- EPO <DP n="37"> --> <maths id="math0057" num=""><math display="block"><mi>IID</mi><mo>=</mo><mfrac><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mfrac></math><img id="ib0057" file="imgb0057.tif" wi="20" he="10" img-content="math" img-format="tif"/></maths> <maths id="math0058" num=""><math display="block"><mi>ICC</mi><mo>=</mo><mfrac><mfenced open="|" close="|"><mfenced open="〈" close="〉"><mi>l</mi><mi>r</mi></mfenced></mfenced><msqrt><mrow><msup><mfenced open="‖" close="‖"><mi>l</mi></mfenced><mn>2</mn></msup><msup><mfenced open="‖" close="‖"><mi>r</mi></mfenced><mn>2</mn></msup></mrow></msqrt></mfrac></math><img id="ib0058" file="imgb0058.tif" wi="32" he="11" img-content="math" img-format="tif"/></maths> and <maths id="math0059" num=""><math display="block"><mfenced open="〈" close="〉"><mi mathvariant="normal">x</mi><mi mathvariant="normal">y</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><munder><mo>∑</mo><mrow><mo>∀</mo><mi>i</mi></mrow></munder><msub><mi>x</mi><mi>i</mi></msub><mo>⋅</mo><msubsup><mi>y</mi><mi>i</mi><mo>∗</mo></msubsup></mstyle></math><img id="ib0059" file="imgb0059.tif" wi="33" he="9" img-content="math" img-format="tif"/></maths></p>
<p id="p0210" num="0210">Thus, in many embodiments, one or more of the gains may be determined based on received interchannel intensity differences, and interchannel correlations (being part of the spatial upmix parameters).</p>
<p id="p0211" num="0211">Different approaches for determining the rendering direction from the spatial upmix parameters may be used in different embodiments. In particular, the signal model as indicated above is based on directional component x being at a direction γ in the stereo image of the stereo signal, henceforth also referred to as the orientation direction. In many embodiments, the direction determining circuit 309 may determine the orientation direction y and then determine the (desired) rendering direction γ' from the orientation direction y. Indeed, in some embodiments or scenarios, the rendering direction γ' may simply be set equal to the orientation direction y.</p>
<p id="p0212" num="0212">The determination of the orientation direction y may be based on the signal model indicated above. The spatial upmix parameters provide information on the relative properties of the channel signals of the stereo signal and specifically they may provide information on both the interchannel levels/intensity differences as well as on the interchannel correlation. Accordingly, the spatial upmix parameters can be considered to provide information on the directional signal component x and on the position of this in the stereo image of the stereo signal, i.e. the spatial upmix parameters provide information on the orientation direction y allowing this to be determined from the provided parameter values.</p>
<p id="p0213" num="0213">The direction determining circuit 309 may determine the orientation direction y as a direction to a directional signal component in a stereo image of the stereo signal from the spatial upmix parameters, and to map this to a direction in a stereo image of the output stereo signal. The directional signal component may be a dominant signal component. The direction determining circuit 309 may be arranged to determine the orientation direction y as a direction of a dominant sound source in the stereo signal where the direction of the dominant sound source is represented by the spatial upmix parameters.<!-- EPO <DP n="38"> --></p>
<p id="p0214" num="0214">The directional signal component may specifically be a signal component (estimated/determined) to originate from a point source. Specifically, the direction determining circuit 309 may be arranged to determine the orientation direction y as a direction for which a single point source audio source will result in spatial upmix parameter values matching the spatial upmix parameters of the data signal.</p>
<p id="p0215" num="0215">In some embodiments, the direction determining circuit may be arranged to determine the first direction in line with: <maths id="math0060" num=""><math display="block"><mi>γ</mi><mo>=</mo><mi>arctan</mi><mspace width="1ex"/><mfenced><mfrac><mrow><mn>1</mn><mo>−</mo><mi>IID</mi><mo>+</mo><msqrt><mrow><msup><mfenced separators=""><mi>IID</mi><mo>−</mo><mn>1</mn></mfenced><mn>2</mn></msup><mo>+</mo><mn>4</mn><mo>⋅</mo><msup><mi>ICC</mi><mn>2</mn></msup><mo>⋅</mo><mi>IID</mi></mrow></msqrt></mrow><mrow><mn>2</mn><mo>⋅</mo><mi>ICC</mi><mo>⋅</mo><msqrt><mi>IID</mi></msqrt></mrow></mfrac></mfenced></math><img id="ib0060" file="imgb0060.tif" wi="89" he="16" img-content="math" img-format="tif"/></maths> where IID is an interchannel intensity difference and ICC is an inter-channel cross-correlation, and specifically with these given by the equations provided above in connection with the equations for determining gains. The determination of the orientation direction y above results from an assumption that the signals <i>x, n<sub>g</sub></i> and <i>n<sub>r</sub></i> of the signal model are mutually decorrelated, and that the power of the left and right diffuse signals <i>n<sub>l</sub></i> and <i>n<sub>r</sub></i> are equal.</p>
<p id="p0216" num="0216">The direction determining circuit 309 may, as previously mentioned, in some embodiments be used directly as the rendering direction γ', i.e. γ = γ'. However, in many embodiments, a mapping may be included which for at least some values of the orientation direction y may result in a dfferent rendering direction γ'.</p>
<p id="p0217" num="0217">Thus, in many embodiments, the direction determining circuit 309 may be arranged to apply a mapping function to the orientation direction y to determine the rendering direction γ'.</p>
<p id="p0218" num="0218">For example, the mapping may map the position in the stereo image of the original stereo signal as represented by the orientation direction y to a desired position in the stereo image of the output stereo signal as represented by the rendering direction γ'. In many cases, where the output stereo signal is a binaural signal, the mapping may include a consideration/ determination of a distance to the audio sources. For example, a range of the orientation direction y in the interval of [0,180°] may be mapped to a location between two virtual stereo speakers in the audio scene created by the binaural rendering. Such speakers may for example be positioned at angles of -30° and +30° relative to a center direction for the binaural signal. Thus, in such situations, the direction determining circuit 309 may include a mapping between an orientation direction y in the range of [0,180°] to a rendering direction γ' in the range of [-30°,+30°].</p>
<p id="p0219" num="0219">Thus, in some embodiments, the directional component (the mono downmix audio signal) may be rendered to a virtual angle in the range of a virtual loudspeaker angle range generated by a<!-- EPO <DP n="39"> --> binaural rendering. The rendered directional component may be combined with a diffuse rendering of the residual signal(s).</p>
<p id="p0220" num="0220">In many embodiments, the direction determining circuit 309 may be arranged to map an orientation direction γ representing an angle in one interval/range to a rendering direction γ' representing an angle in a different interval/range.</p>
<p id="p0221" num="0221">In the previous examples, the rendering has been based on one intermediate stereo signal representing the diffuse signal component. However, in many embodiments, there may be two (or possibly more) parallel paths for the rendering of the residual/non-directional signal components.</p>
<p id="p0222" num="0222">The audio source device 101 is arranged to include data representing the difference signal in the data signal. Typically, the difference signal will be suitable encoded, e.g. using an audio encoding algorithm, or in many embodiments using a non-lossy encoding.</p>
<p id="p0223" num="0223">The data signal may be generated to include the different types of data in a suitable structure and with suitable data overhead (such as headers etc).</p>
<p id="p0224" num="0224">In some embodiments, the data signal circuit 207 is arranged to include the encoded data for the difference signal in suitable data fields of the generated bitstream, and in particular it may include the encoded data for the difference signal in one or more optional data fields of the audio data signal. An optional data field may be a field that can include data which is optional for the data signal. For example, some traditional data structures for PS data signals include data fields in which proprietary or non-PS data can be included and in some embodiments the difference signal data may be included in such data fields. In such cases, a conventional PS decoder/renderer may simply ignore the difference data and proceed to perform a conventional PS decoding and rendering. However, an audio render apparatus as previously described may instead extract the difference data and proceed to perform a rendering based on a generated directional signal as previously described. The approach may accordingly provide improved rendering for suitably capable renderers while providing backwards compatibility to support legacy decoders/renderers.</p>
<p id="p0225" num="0225">The processing may be performed in subbands and may be performed in time segments. The processing in each subband may for some (any) or all steps be performed separately/independently in each subband (with respect to the processing in other subbands). The processing in each time segment may for some (any) or all steps be performed separately/independently in each time segment (with respect to the processing in other time segments).</p>
<p id="p0226" num="0226">The processing may be time interval/segment based with all processing being performed for each time segment. Equivalently, the signal(s) for each segment may be considered a signal (and in particular signals of different time segments, may be considered different signals).</p>
<p id="p0227" num="0227">The audio apparatus(s) may specifically be implemented in one or more suitably programmed processors. An example of a suitable processor is provided in the following.</p>
<p id="p0228" num="0228"><figref idref="f0005">FIG. 5</figref> is a block diagram illustrating an example processor 500 according to embodiments of the disclosure. Processor 500 may be used to implement one or more processors<!-- EPO <DP n="40"> --> implementing an apparatus as previously described or elements thereof (including in particular one more artificial neural network). Processor 500 may be any suitable processor type including, but not limited to, a microprocessor, a microcontroller, a Digital Signal Processor (DSP), a Field ProGrammable Array (FPGA) where the FPGA has been programmed to form a processor, a Graphical Processing Unit (GPU), an Application Specific Integrated Circuit (ASIC) where the ASIC has been designed to form a processor, or a combination thereof.</p>
<p id="p0229" num="0229">The processor 500 may include one or more cores 502. The core 502 may include one or more Arithmetic Logic Units (ALU) 504. In some embodiments, the core 502 may include a Floating Point Logic Unit (FPLU) 506 and/or a Digital Signal Processing Unit (DSPU) 508 in addition to or instead of the ALU 504.</p>
<p id="p0230" num="0230">The processor 500 may include one or more registers 512 communicatively coupled to the core 502. The registers 512 may be implemented using dedicated logic gate circuits (e.g., flip-flops) and/or any memory technology. In some embodiments the registers 512 may be implemented using static memory. The register may provide data, instructions and addresses to the core 502.</p>
<p id="p0231" num="0231">In some embodiments, processor 500 may include one or more levels of cache memory 510 communicatively coupled to the core 502. The cache memory 510 may provide computer-readable instructions to the core 502 for execution. The cache memory 510 may provide data for processing by the core 502. In some embodiments, the computer-readable instructions may have been provided to the cache memory 510 by a local memory, for example, local memory attached to the external bus 516. The cache memory 510 may be implemented with any suitable cache memory type, for example, Metal-Oxide Semiconductor (MOS) memory such as Static Random Access Memory (SRAM), Dynamic Random Access Memory (DRAM), and/or any other suitable memory technology.</p>
<p id="p0232" num="0232">The processor 500 may include a controller 514, which may control input to the processor 500 from other processors and/or components included in a system and/or outputs from the processor 500 to other processors and/or components included in the system. Controller 514 may control the data paths in the ALU 504, FPLU 506 and/or DSPU 508. Controller 514 may be implemented as one or more state machines, data paths and/or dedicated control logic. The gates of controller 514 may be implemented as standalone gates, FPGA, ASIC or any other suitable technology.</p>
<p id="p0233" num="0233">The registers 512 and the cache 510 may communicate with controller 514 and core 502 via internal connections 520A, 520B, 520C and 520D. Internal connections may be implemented as a bus, multiplexer, crossbar switch, and/or any other suitable connection technology.</p>
<p id="p0234" num="0234">Inputs and outputs for the processor 500 may be provided via a bus 516, which may include one or more conductive lines. The bus 516 may be communicatively coupled to one or more components of processor 500, for example the controller 514, cache 510, and/or register 512. The bus 516 may be coupled to one or more components of the system.</p>
<p id="p0235" num="0235">The bus 516 may be coupled to one or more external memories. The external memories may include Read Only Memory (ROM) 532. ROM 532 may be a masked ROM, Electronically<!-- EPO <DP n="41"> --> Programmable Read Only Memory (EPROM) or any other suitable technology. The external memory may include Random Access Memory (RAM) 533. RAM 533 may be a static RAM, battery backed up static RAM, Dynamic RAM (DRAM) or any other suitable technology. The external memory may include Electrically Erasable Programmable Read Only Memory (EEPROM) 535. The external memory may include Flash memory 534. The External memory may include a magnetic storage device such as disc 536. In some embodiments, the external memories may be included in a system.</p>
<p id="p0236" num="0236">The invention can be implemented in any suitable form including hardware, software, firmware, or any combination of these. The invention may optionally be implemented at least partly as computer software running on one or more data processors and/or digital signal processors. The elements and components of an embodiment of the invention may be physically, functionally and logically implemented in any suitable way. Indeed, the functionality may be implemented in a single unit, in a plurality of units or as part of other functional units. As such, the invention may be implemented in a single unit or may be physically and functionally distributed between different units, circuits and processors.</p>
<p id="p0237" num="0237">Although the present invention has been described in connection with some embodiments, it is not intended to be limited to the specific form set forth herein. Rather, the scope of the present invention is limited only by the accompanying claims. Additionally, although a feature may appear to be described in connection with particular embodiments, one skilled in the art would recognize that various features of the described embodiments may be combined in accordance with the invention. In the claims, the term comprising does not exclude the presence of other elements or steps.</p>
<p id="p0238" num="0238">Furthermore, although individually listed, a plurality of means, elements, circuits or method steps may be implemented by e.g. a single circuit, unit or processor. Additionally, although individual features may be included in different claims, these may possibly be advantageously combined, and the inclusion in different claims does not imply that a combination of features is not feasible and/or advantageous. Also, the inclusion of a feature in one category of claims does not imply a limitation to this category but rather indicates that the feature is equally applicable to other claim categories as appropriate. Furthermore, the order of features in the claims do not imply any specific order in which the features must be worked and in particular the order of individual steps in a method claim does not imply that the steps must be performed in this order. Rather, the steps may be performed in any suitable order. In addition, singular references do not exclude a plurality. Thus references to "a", "an", "first", "second" etc. do not preclude a plurality. Reference signs in the claims are provided merely as a clarifying example shall not be construed as limiting the scope of the claims in any way.</p>
</description>
<claims id="claims01" lang="en"><!-- EPO <DP n="42"> -->
<claim id="c-en-0001" num="0001">
<claim-text>An audio apparatus comprising:
<claim-text>a receiver (301) arranged to receive a data signal comprising a set of spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal, encoded data for a mono downmix audio signal being a downmix of a stereo signal, and a difference signal representing a difference between the mono downmix audio signal and a directional signal, the directional signal representing a point source audio component of the stereo signal;</claim-text>
<claim-text>a store (311) comprising directional transfer functions for different directions, a directional transfer function for a given direction representing a mapping of a mono audio signal to stereo channels such that the mono audio signal is positioned in the given direction in a stereo image of the stereo channels;</claim-text>
<claim-text>a decoder (303) arranged to generate the mono downmix audio signal and the difference signal by decoding the encoded data;</claim-text>
<claim-text>a compensator (305) arranged to generate a modified mono downmix audio signal by compensating the mono downmix audio signal as a function of the difference signal;</claim-text>
<claim-text>a direction determining circuit (309) arranged to determine a first direction from the spatial upmix parameters;</claim-text>
<claim-text>a first renderer (307) arranged to perform a first rendering of the modified mono downmix audio signal to generate a first intermediate stereo signal, the first rendering being a directional rendering using a first channel transfer function retrieved from the store for the first direction; and</claim-text>
<claim-text>an output circuit (313) to generate the output stereo signal to include the first intermediate stereo signal.</claim-text></claim-text></claim>
<claim id="c-en-0002" num="0002">
<claim-text>The audio apparatus of claim 1 further comprising:
<claim-text>a decorrelator (315) arranged to apply a decorrelation to at least one of the mono downmix audio signal and the modified mono downmix audio signal to generate a first decorrelated mono downmix audio signal;</claim-text>
<claim-text>a second renderer (317) arranged to perform a second rendering being a rendering of the first decorrelated mono downmix audio signal to generate a second intermediate stereo signal, the second rendering being a predetermined rendering employing a predetermined mapping of the decorrelated mono downmix audio signal to channel signals of the second intermediate stereo signal; and</claim-text>
<claim-text>wherein the output circuit (313) is arranged to combine at least the first intermediate stereo signal and the second intermediate stereo signal to generate an output stereo signal.</claim-text><!-- EPO <DP n="43"> --></claim-text></claim>
<claim id="c-en-0003" num="0003">
<claim-text>The audio apparatus of any previous claim wherein the difference signal is indicative of a difference between the directional signal and an estimate of the directional signal, the estimate of the directional signal being a predetermined function of the mono downmix audio signal and the spatial parameters; and the compensator (305) is arranged to generate an estimate of the directional signal from the mono downmix audio signal and the spatial parameters and to generate the modified mono downmix audio signal by compensating the predicted signal as a function of the difference signal.</claim-text></claim>
<claim id="c-en-0004" num="0004">
<claim-text>The audio apparatus of claim 3 wherein the compensator (305) is arranged to determine the estimate of the directional signal by scaling the mono downmix audio signal using scale factors determined from the spatial parameters.</claim-text></claim>
<claim id="c-en-0005" num="0005">
<claim-text>The audio apparatus of any previous claim wherein a data rate of encoded data for the mono downmix audio signal is no less than ten times higher than a data rate of encoded data for the difference signal.</claim-text></claim>
<claim id="c-en-0006" num="0006">
<claim-text>The audio apparatus of any previous claim wherein the first rendering is a binaural rendering and the directional transfer functions are binaural transfer functions.</claim-text></claim>
<claim id="c-en-0007" num="0007">
<claim-text>The audio apparatus of any previous claim wherein the second rendering is arranged to generate the second intermediate stereo signal using a set of directional transfer functions retrieved from the store for a set of predetermined directions.</claim-text></claim>
<claim id="c-en-0008" num="0008">
<claim-text>The audio apparatus of any previous claim wherein the spatial upmix parameters and the directional transfer functions are provided for frequency subbands and the first renderer is arranged to generate subband values for subbands of the first intermediate stereo signals from subband values of the modified mono downmix audio signal based on spatial upmix parameters and directional transfer functions for the subbands.</claim-text></claim>
<claim id="c-en-0009" num="0009">
<claim-text>The audio apparatus of any previous claim wherein the direction determining circuit (309) is arranged to determine a point source direction in a stereo image of the stereo signal from the spatial upmix parameters, and to determine the first direction by applying a mapping function to the point source direction.</claim-text></claim>
<claim id="c-en-0010" num="0010">
<claim-text>An audio apparatus comprising:
<claim-text>a downmixer (203) arranged to generate a mono downmix audio signal for a stereo signal;<!-- EPO <DP n="44"> --></claim-text>
<claim-text>a parameter determiner (205) arranged to determine spatial upmix parameters for upmixing the mono downmix audio signal to the stereo signal, the spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal;</claim-text>
<claim-text>a generator (209) arranged to determine a directional signal for the stereo signal, the directional signal representing a point source audio component of the stereo signal;</claim-text>
<claim-text>a difference signal circuit (211) arranged to generate a difference signal indicative of a difference between the mono downmix audio signal and the directional signal; and</claim-text>
<claim-text>a data signal circuit (207) arranged to generate the audio data signal to comprise the spatial parameters, the encoded data for the mono downmix audio signal, and the difference signal.</claim-text></claim-text></claim>
<claim id="c-en-0011" num="0011">
<claim-text>The apparatus of claim 10 wherein the difference signal circuit (211) is arranged to generate an estimate of the directional signal from the mono downmix audio signal and the spatial parameters and to generate the difference signal to indicate a difference between the estimate of the directional signal and the directional signal.</claim-text></claim>
<claim id="c-en-0012" num="0012">
<claim-text>The apparatus of claim 10 or 11 wherein the data circuit (207) is arranged to include the encoded data for the difference signal in an optional data field of the audio data signal.</claim-text></claim>
<claim id="c-en-0013" num="0013">
<claim-text>A method of operation for an audio apparatus, the method comprising:
<claim-text>receiving a data signal comprising a set of spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal, encoded data for a mono downmix audio signal being a downmix of a stereo signal, and a difference signal representing a difference between the mono downmix audio signal and a directional signal, the directional signal representing a point source audio component of the stereo signal;</claim-text>
<claim-text>providing directional transfer functions for different directions, a directional transfer function for a given direction representing a mapping of a mono audio signal to stereo channels such that the mono audio signal is positioned in the given direction in a stereo image of the stereo channels;</claim-text>
<claim-text>generating the mono downmix audio signal and the difference signal by decoding the encoded data;</claim-text>
<claim-text>generating a modified mono downmix audio signal by compensating the mono downmix audio signal as a function of the difference signal;</claim-text>
<claim-text>determining a first direction from the spatial upmix parameters;</claim-text>
<claim-text>performing a first rendering of the modified mono downmix audio signal to generate a first intermediate stereo signal, the first rendering being a directional rendering using a first channel transfer function retrieved from the store for the first direction; and</claim-text>
<claim-text>generating the output stereo signal to include the first intermediate stereo signal.</claim-text><!-- EPO <DP n="45"> --></claim-text></claim>
<claim id="c-en-0014" num="0014">
<claim-text>A method of operation for an audio apparatus, the method comprising:<br/>
generating a mono downmix audio signal for a stereo signal;
<claim-text>determining spatial upmix parameters for upmixing the mono downmix audio signal to the stereo signal, the spatial upmix parameters being indicative of relative signal properties of channels of the stereo signal;</claim-text>
<claim-text>determining a directional signal for the stereo signal, the directional signal representing a point source audio component of the stereo signal;</claim-text>
<claim-text>generating a difference signal indicative of a difference between the mono downmix audio signal and the directional signal; and</claim-text>
<claim-text>generating the audio data signal to comprise the spatial parameters, the encoded data for the mono downmix audio signal, and the difference signal.</claim-text></claim-text></claim>
<claim id="c-en-0015" num="0015">
<claim-text>A computer program product comprising computer program code means adapted to perform all the steps of claims 13 or 14 when said program is run on a computer.</claim-text></claim>
</claims>
<drawings id="draw" lang="en"><!-- EPO <DP n="46"> -->
<figure id="f0001" num="1"><img id="if0001" file="imgf0001.tif" wi="112" he="171" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="47"> -->
<figure id="f0002" num="2"><img id="if0002" file="imgf0002.tif" wi="152" he="128" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="48"> -->
<figure id="f0003" num="3"><img id="if0003" file="imgf0003.tif" wi="89" he="224" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="49"> -->
<figure id="f0004" num="4"><img id="if0004" file="imgf0004.tif" wi="127" he="236" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="50"> -->
<figure id="f0005" num="5"><img id="if0005" file="imgf0005.tif" wi="118" he="222" img-content="drawing" img-format="tif"/></figure>
</drawings>
<search-report-data id="srep" lang="en" srep-office="EP" date-produced=""><doc-page id="srep0001" file="srep0001.tif" wi="160" he="240" type="tif"/><doc-page id="srep0002" file="srep0002.tif" wi="158" he="240" type="tif"/></search-report-data><search-report-data date-produced="20250730" id="srepxml" lang="en" srep-office="EP" srep-type="ep-sr" status="n"><!--
 The search report data in XML is provided for the users' convenience only. It might differ from the search report of the PDF document, which contains the officially published data. The EPO disclaims any liability for incorrect or incomplete data in the XML for search reports.
 -->

<srep-info><file-reference-id>2024P00622EP</file-reference-id><application-reference><document-id><country>EP</country><doc-number>25160557.2</doc-number></document-id></application-reference><applicant-name><name>Koninklijke Philips N.V.</name></applicant-name><srep-established srep-established="yes"/><srep-invention-title title-approval="yes"/><srep-abstract abs-approval="yes"/><srep-figure-to-publish figinfo="by-applicant"><figure-to-publish><fig-number>2</fig-number></figure-to-publish></srep-figure-to-publish><srep-info-admin><srep-office><addressbook><text>DH</text></addressbook></srep-office><date-search-report-mailed><date>20250813</date></date-search-report-mailed></srep-info-admin></srep-info><srep-for-pub><srep-fields-searched><minimum-documentation><classifications-ipcr><classification-ipcr><text>H04S</text></classification-ipcr></classifications-ipcr></minimum-documentation></srep-fields-searched><srep-citations><citation id="sr-cit0001"><patcit dnum="WO2010122455A1" id="sr-pcit0001" url="http://v3.espacenet.com/textdoc?DB=EPODOC&amp;IDX=WO2010122455&amp;CY=ep"><document-id><country>WO</country><doc-number>2010122455</doc-number><kind>A1</kind><name>KONINKL PHILIPS ELECTRONICS NV [NL]; SCHUIJERS ERIK G P [NL] ET AL.</name><date>20101028</date></document-id></patcit><category>A</category><rel-claims>1-15</rel-claims><rel-passage><passage>* page 10, line 31 - page 26, line 19; figures 1-4 *</passage></rel-passage></citation><citation id="sr-cit0002"><nplcit id="sr-ncit0001" npl-type="s"><article><author><name>JEROEN BREEBAART ET AL</name></author><atl>Phantom Materialization: A Novel Method to Enhance Stereo Audio Reproduction on Headphones</atl><serial><sertitle>IEEE TRANSACTIONS ON AUDIO, SPEECH AND LANGUAGE PROCESSING, IEEE, US</sertitle><pubdate>20081101</pubdate><vid>16</vid><ino>8</ino><doi>10.1109/TASL.2008.2002983</doi><issn>1558-7916</issn></serial><location><pp><ppf>1503</ppf><ppl>1511</ppl></pp></location><refno>XP011236282</refno></article></nplcit><category>A</category><rel-claims>1-15</rel-claims><rel-passage><passage>* the whole document *</passage></rel-passage></citation><citation id="sr-cit0003"><patcit dnum="WO2009046909A1" id="sr-pcit0002" url="http://v3.espacenet.com/textdoc?DB=EPODOC&amp;IDX=WO2009046909&amp;CY=ep"><document-id><country>WO</country><doc-number>2009046909</doc-number><kind>A1</kind><name>KONINKL PHILIPS ELECTRONICS NV [NL]; DOLBY SWEDEN AB [SE] ET AL.</name><date>20090416</date></document-id></patcit><category>A</category><rel-claims>1-15</rel-claims><rel-passage><passage>* page 13, line 18 - page 26, line 20; figure 4 *</passage></rel-passage></citation><citation id="sr-cit0004"><patcit dnum="WO2007031896A1" id="sr-pcit0003" url="http://v3.espacenet.com/textdoc?DB=EPODOC&amp;IDX=WO2007031896&amp;CY=ep"><document-id><country>WO</country><doc-number>2007031896</doc-number><kind>A1</kind><name>KONINKL PHILIPS ELECTRONICS NV [NL]; BREEBAART DIRK J [NL]</name><date>20070322</date></document-id></patcit><category>A</category><rel-claims>1-15</rel-claims><rel-passage><passage>* page 14, line 17 - page 17, line 26; figures 6-8 *</passage></rel-passage></citation></srep-citations><srep-admin><examiners><primary-examiner><name>Navarri, Massimo</name></primary-examiner></examiners><srep-office><addressbook><text>The Hague</text></addressbook></srep-office><date-search-completed><date>20250730</date></date-search-completed></srep-admin><!--							The annex lists the patent family members relating to the patent documents cited in the above mentioned European search report.							The members are as contained in the European Patent Office EDP file on							The European Patent Office is in no way liable for these particulars which are merely given for the purpose of information.							For more details about this annex : see Official Journal of the European Patent Office, No 12/82						--><srep-patent-family><patent-family><priority-application><document-id><country>WO</country><doc-number>2010122455</doc-number><kind>A1</kind><date>20101028</date></document-id></priority-application><family-member><document-id><country>CN</country><doc-number>102414743</doc-number><kind>A</kind><date>20120411</date></document-id></family-member><family-member><document-id><country>EP</country><doc-number>2422344</doc-number><kind>A1</kind><date>20120229</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>2012525051</doc-number><kind>A</kind><date>20121018</date></document-id></family-member><family-member><document-id><country>KR</country><doc-number>20120006060</doc-number><kind>A</kind><date>20120117</date></document-id></family-member><family-member><document-id><country>RU</country><doc-number>2011147119</doc-number><kind>A</kind><date>20130527</date></document-id></family-member><family-member><document-id><country>TW</country><doc-number>201106343</doc-number><kind>A</kind><date>20110216</date></document-id></family-member><family-member><document-id><country>US</country><doc-number>2012039477</doc-number><kind>A1</kind><date>20120216</date></document-id></family-member><family-member><document-id><country>WO</country><doc-number>2010122455</doc-number><kind>A1</kind><date>20101028</date></document-id></family-member></patent-family><patent-family><priority-application><document-id><country>WO</country><doc-number>2009046909</doc-number><kind>A1</kind><date>20090416</date></document-id></priority-application><family-member><document-id><country>AU</country><doc-number>2008309951</doc-number><kind>A1</kind><date>20090416</date></document-id></family-member><family-member><document-id><country>BR</country><doc-number>PI0816618</doc-number><kind>A2</kind><date>20150310</date></document-id></family-member><family-member><document-id><country>CA</country><doc-number>2701360</doc-number><kind>A1</kind><date>20090416</date></document-id></family-member><family-member><document-id><country>CN</country><doc-number>101933344</doc-number><kind>A</kind><date>20101229</date></document-id></family-member><family-member><document-id><country>EP</country><doc-number>2198632</doc-number><kind>A1</kind><date>20100623</date></document-id></family-member><family-member><document-id><country>ES</country><doc-number>2461601</doc-number><kind>T3</kind><date>20140520</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>5391203</doc-number><kind>B2</kind><date>20140115</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>2010541510</doc-number><kind>A</kind><date>20101224</date></document-id></family-member><family-member><document-id><country>KR</country><doc-number>20100063113</doc-number><kind>A</kind><date>20100610</date></document-id></family-member><family-member><document-id><country>MY</country><doc-number>150381</doc-number><kind>A</kind><date>20131231</date></document-id></family-member><family-member><document-id><country>PL</country><doc-number>2198632</doc-number><kind>T3</kind><date>20140829</date></document-id></family-member><family-member><document-id><country>RU</country><doc-number>2010112887</doc-number><kind>A</kind><date>20111120</date></document-id></family-member><family-member><document-id><country>TW</country><doc-number>200926876</doc-number><kind>A</kind><date>20090616</date></document-id></family-member><family-member><document-id><country>US</country><doc-number>2010246832</doc-number><kind>A1</kind><date>20100930</date></document-id></family-member><family-member><document-id><country>WO</country><doc-number>2009046909</doc-number><kind>A1</kind><date>20090416</date></document-id></family-member></patent-family><patent-family><priority-application><document-id><country>WO</country><doc-number>2007031896</doc-number><kind>A1</kind><date>20070322</date></document-id></priority-application><family-member><document-id><country>BR</country><doc-number>PI0615899</doc-number><kind>A2</kind><date>20110531</date></document-id></family-member><family-member><document-id><country>CN</country><doc-number>101263742</doc-number><kind>A</kind><date>20080910</date></document-id></family-member><family-member><document-id><country>EP</country><doc-number>1927266</doc-number><kind>A1</kind><date>20080604</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>5587551</doc-number><kind>B2</kind><date>20140910</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>5698189</doc-number><kind>B2</kind><date>20150408</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>2009508157</doc-number><kind>A</kind><date>20090226</date></document-id></family-member><family-member><document-id><country>JP</country><doc-number>2012181556</doc-number><kind>A</kind><date>20120920</date></document-id></family-member><family-member><document-id><country>KR</country><doc-number>20080047446</doc-number><kind>A</kind><date>20080528</date></document-id></family-member><family-member><document-id><country>KR</country><doc-number>20150008932</doc-number><kind>A</kind><date>20150123</date></document-id></family-member><family-member><document-id><country>TW</country><doc-number>I415111</doc-number><kind>B</kind><date>20131111</date></document-id></family-member><family-member><document-id><country>US</country><doc-number>2008205658</doc-number><kind>A1</kind><date>20080828</date></document-id></family-member><family-member><document-id><country>WO</country><doc-number>2007031896</doc-number><kind>A1</kind><date>20070322</date></document-id></family-member></patent-family></srep-patent-family></srep-for-pub></search-report-data>
<ep-reference-list id="ref-list">
<heading id="ref-h0001"><b>REFERENCES CITED IN THE DESCRIPTION</b></heading>
<p id="ref-p0001" num=""><i>This list of references cited by the applicant is for the reader's convenience only. It does not form part of the European patent document. Even though great care has been taken in compiling the references, errors or omissions cannot be excluded and the EPO disclaims all liability in this regard.</i></p>
<heading id="ref-h0002"><b>Patent documents cited in the description</b></heading>
<p id="ref-p0002" num="">
<ul id="ref-ul0001" list-style="bullet">
<li><patcit id="ref-pcit0001" dnum="WO2010122455A1"><document-id><country>WO</country><doc-number>2010122455</doc-number><kind>A1</kind></document-id></patcit><crossref idref="pcit0001">[0008]</crossref></li>
<li><patcit id="ref-pcit0002" dnum="WO2007031896A1"><document-id><country>WO</country><doc-number>2007031896</doc-number><kind>A1</kind></document-id></patcit><crossref idref="pcit0002">[0008]</crossref></li>
</ul></p>
<heading id="ref-h0003"><b>Non-patent literature cited in the description</b></heading>
<p id="ref-p0003" num="">
<ul id="ref-ul0002" list-style="bullet">
<li><nplcit id="ref-ncit0001" npl-type="s"><article><author><name>E. SCHUIJERS</name></author><author><name>W. OOMEN</name></author><author><name>B. DEN BRINKER</name></author><author><name>J. BREEBAART</name></author><atl>Advances in Parametric Coding for High-Quality Audio</atl><serial><sertitle>114th AES Convention</sertitle><pubdate><sdate>20030000</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0001">[0005]</crossref></li>
<li><nplcit id="ref-ncit0002" npl-type="s"><article><author><name>E. SCHUIJERS</name></author><author><name>J. BREEBAART</name></author><author><name>H. PUMHAGEN</name></author><author><name>J. ENGDEGÅRD</name></author><atl>Low Complexity Parametric Stereo Coding</atl><serial><sertitle>116th AES</sertitle><pubdate><sdate>20040000</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0002">[0005]</crossref></li>
<li><nplcit id="ref-ncit0003" npl-type="s"><article><author><name>J. BREEBAART</name></author><author><name>E. SCHUIJERS</name></author><atl>Phantom materialization: A novel method to enhance stereo audio reproduction on headphones</atl><serial><sertitle>IEEE transactions on audio, speech, and language processing</sertitle><pubdate><sdate>20080000</sdate><edate/></pubdate><vid>16</vid><ino>8</ino></serial><location><pp><ppf>1503</ppf><ppl>1511</ppl></pp></location></article></nplcit><crossref idref="ncit0003">[0009]</crossref></li>
</ul></p>
</ep-reference-list>
</ep-patent-document>
