<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE ep-patent-document PUBLIC "-//EPO//EP PATENT DOCUMENT 1.5//EN" "ep-patent-document-v1-5.dtd">
<ep-patent-document id="EP08734700B1" file="EP08734700NWB1.xml" lang="en" country="EP" doc-number="2269371" kind="B1" date-publ="20180131" status="n" dtd-version="ep-patent-document-v1-5">
<SDOBI lang="en"><B000><eptags><B001EP>ATBECHDEDKESFRGBGRITLILUNLSEMCPTIESILTLVFIRO..CY..TRBGCZEEHUPLSK..HRIS..MTNO........................</B001EP><B003EP>*</B003EP><B005EP>J</B005EP><B007EP>BDM Ver 0.1.63 (23 May 2017) -  2100000/0</B007EP></eptags></B000><B100><B110>2269371</B110><B120><B121>EUROPEAN PATENT SPECIFICATION</B121></B120><B130>B1</B130><B140><date>20180131</date></B140><B190>EP</B190></B100><B200><B210>08734700.1</B210><B220><date>20080320</date></B220><B240><B241><date>20101020</date></B241><B242><date>20130321</date></B242></B240><B250>en</B250><B251EP>en</B251EP><B260>en</B260></B200><B400><B405><date>20180131</date><bnum>201805</bnum></B405><B430><date>20110105</date><bnum>201101</bnum></B430><B450><date>20180131</date><bnum>201805</bnum></B450><B452EP><date>20170912</date></B452EP></B400><B500><B510EP><classification-ipcr sequence="1"><text>G06K   9/00        20060101AFI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="2"><text>G06K   9/32        20060101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="3"><text>H04N   7/01        20060101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="4"><text>G06K   9/46        20060101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="5"><text>H04N   5/14        20060101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="6"><text>H04N   5/44        20110101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="7"><text>H04N  21/21        20110101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="8"><text>H04N  21/44        20110101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="9"><text>H04N  21/4402      20110101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="10"><text>H04N  21/41        20110101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="11"><text>H04N  21/462       20110101ALI20170830BHEP        </text></classification-ipcr><classification-ipcr sequence="12"><text>H04N  21/84        20110101ALI20170830BHEP        </text></classification-ipcr></B510EP><B540><B541>de</B541><B542>VERFAHREN ZUR ADAPTIERUNG VON BILDER AN KLEINE BILDANZEIGEGERÄTE</B542><B541>en</B541><B542>A METHOD OF ADAPTING VIDEO IMAGES TO SMALL SCREEN SIZES</B542><B541>fr</B541><B542>PROCÉDÉ D'ADAPTATION D'IMAGES VIDÉO À DE PETITES TAILLES D'ÉCRAN</B542></B540><B560><B561><text>WO-A-2006/056311</text></B561><B561><text>US-A1- 2005 203 927</text></B561><B561><text>US-A1- 2006 139 371</text></B561><B561><text>US-A1- 2006 215 753</text></B561><B561><text>US-A1- 2006 239 645</text></B561></B560></B500><B700><B720><B721><snm>DEIGMÖLLER, Jörg</snm><adr><str>Mainzer Strasse 46</str><city>55257 Budenheim</city><ctry>DE</ctry></adr></B721><B721><snm>STOLL, Gerhard</snm><adr><str>Ahornweg 21</str><city>85406 Zolling</city><ctry>DE</ctry></adr></B721><B721><snm>NEUSCHMIED, Helmut</snm><adr><str>Messendorfberg 92</str><city>A-8042 Graz</city><ctry>AT</ctry></adr></B721><B721><snm>KRIECHBAUM, Andreas</snm><adr><str>Waltendorfer Gürtel 13a/2/210</str><city>A-8010 Graz</city><ctry>AT</ctry></adr></B721><B721><snm>DOS SANTOS CARDOSO, José Bernardo</snm><adr><str>Rua Eng. José Ferreira Pinto Basto</str><city>P-3810-106 Aveiro</city><ctry>PT</ctry></adr></B721><B721><snm>OLIVEIRA DE CARVALHO, Fausto, José</snm><adr><str>Rua Eng. José Ferreira Pinto Basto</str><city>P-3810-106 Aveiro</city><ctry>PT</ctry></adr></B721><B721><snm>SALGADO DE ALEM, Roger</snm><adr><str>Rua Eng. José Ferreira Pinto Basto</str><city>P-3810-106 Aveiro</city><ctry>PT</ctry></adr></B721><B721><snm>HUET, Benoit</snm><adr><str>Cidex 416
Descente de l'Aire de Boules</str><city>F-06330 Roquefort Les Pins</city><ctry>FR</ctry></adr></B721><B721><snm>MERIALDO, Bernard</snm><adr><str>969 avenue de Pierrefeu</str><city>F-06560 Valbonne</city><ctry>FR</ctry></adr></B721><B721><snm>TRICHET, Remi</snm><adr><str>1800 N. New Hampshire Ave.</str><city>90027 Los Angeles, CA</city><ctry>US</ctry></adr></B721></B720><B730><B731><snm>Institut für Rundfunktechnik GmbH</snm><iid>100147087</iid><irf>IRT082</irf><adr><str>Floriansmühlstrasse 60</str><city>80939 München</city><ctry>DE</ctry></adr></B731><B731><snm>Joanneum Research Forschungsgesellschaft Mbh 
Institute Of Information Systems</snm><iid>101138404</iid><irf>IRT082</irf><adr><str>Steyrergasse 17</str><city>8010 Graz</city><ctry>AT</ctry></adr></B731><B731><snm>Portugal Telecom Inovacao, Sa</snm><iid>101138405</iid><irf>IRT082</irf><adr><str>Rua Eng. José Ferreira Pinto Basto</str><city>3810-106 Aveiro</city><ctry>PT</ctry></adr></B731></B730><B740><B741><snm>Dini, Roberto</snm><sfx>et al</sfx><iid>100024688</iid><adr><str>Metroconsult S.r.l. 
Via Sestriere 100</str><city>10060 None (TO)</city><ctry>IT</ctry></adr></B741></B740></B700><B800><B840><ctry>AT</ctry><ctry>BE</ctry><ctry>BG</ctry><ctry>CH</ctry><ctry>CY</ctry><ctry>CZ</ctry><ctry>DE</ctry><ctry>DK</ctry><ctry>EE</ctry><ctry>ES</ctry><ctry>FI</ctry><ctry>FR</ctry><ctry>GB</ctry><ctry>GR</ctry><ctry>HR</ctry><ctry>HU</ctry><ctry>IE</ctry><ctry>IS</ctry><ctry>IT</ctry><ctry>LI</ctry><ctry>LT</ctry><ctry>LU</ctry><ctry>LV</ctry><ctry>MC</ctry><ctry>MT</ctry><ctry>NL</ctry><ctry>NO</ctry><ctry>PL</ctry><ctry>PT</ctry><ctry>RO</ctry><ctry>SE</ctry><ctry>SI</ctry><ctry>SK</ctry><ctry>TR</ctry></B840><B860><B861><dnum><anum>EP2008002266</anum></dnum><date>20080320</date></B861><B862>en</B862></B860><B870><B871><dnum><pnum>WO2009115101</pnum></dnum><date>20090924</date><bnum>200939</bnum></B871></B870><B880><date>20110105</date><bnum>201101</bnum></B880></B800></SDOBI>
<description id="desc" lang="en"><!-- EPO <DP n="1"> -->
<p id="p0001" num="0001">The invention to which this application relates is a method of adapting video images to small screen sizes, in particular to small screen sizes of portable handheld terminals. Mobile TV (Mobile Television) is a growing and certainly promising market. It allows the reception of Television signals on small portable devices like cell phones, smartphones or PDAs (Personal Digital Assistant). The display on the screen of those small portable devices does not provide such a detailed image as it is known from stationary TV sets at home (currently SDTV, Standard Definition Television). Irrespective of such essential difference of viewing conditions, the same contents are mainly displayed on the screens of both, mobile and stationary TV systems. However, producing a separate programme for mobile TV would cause a huge expenditure of human sources as well as an increase of costs which broadcasters hardly can bring up. To overcome such uncomfortable situation some proposals were made to adapt video contents having a high image resolution to smaller displays by cropping parts out. Such proposals are dealing with the automatic detection of regions of interest (ROI) based on feature extraction with common video analysis methods. The detected regions of interest in a video signal are used to find an adequate crop (cutting) area and to compose a new image containing all relevant information adapted to displays of handheld devices.</p>
<p id="p0002" num="0002">However, such known cropping systems are inadequately dealing with a wide range of contents since they are missing semantically knowledge and thus general defined methods.</p>
<p id="p0003" num="0003"><patcit id="pcit0001" dnum="WO2006056311A"><text>WO2006/056311</text></patcit> discloses a method of automatic navigation between a digital image and a region of interest of this image. The method enables automatic navigation towards a region of interest without manual control, and simply by transmitting a tilting movement to the terminal in the direction of the region of interest.</p>
<p id="p0004" num="0004">It is the object of the present invention to improve a cropping system by obtaining the coverage of a wide range of contents for smaller sized displays of handheld devices. The above object is solved by a method starting from a metadata aggregation and the corresponding video, e.g. in post-production, programme exchange and archiving,wherein
<ol id="ol0001" compact="compact" ol-style="">
<li>(a) the video is passed through to a video analysis to deliver video, e.g. by use of motion detection, morphology filters, edge detection, etc.,</li>
<li>(b) the separated video and metadata are combined to extract important features in a context wherein important information from the metadata is categorised and is used to initialise a dynamically fitted chain of feature extraction steps adapted to the delivered</li>
</ol><!-- EPO <DP n="2"> --></p>
<p id="p0005" num="0005">It is the object of the present invention to improve a cropping system by obtaining the coverage of a wide range of contents for smaller sized displays of handheld devices.</p>
<p id="p0006" num="0006">The above object is solved by the features of the appended claims which define the scope of the present invention.<!-- EPO <DP n="3"> --> Adantageously, the invention provides a feature extraction in video signals with the aid of available metadata to crop important image regions and adapt them on displays with lower resolution.</p>
<p id="p0007" num="0007">Specific embodiments of the invention are now described with reference to the accompanying drawings wherein
<dl id="dl0001">
<dt>Fig. 1</dt><dd>illustrates a schematic block diagram of the overall system performing the method of the invention;</dd>
<dt>Figs. 2 to 5</dt><dd>illustrate the various blocks shown in the system of <figref idref="f0001">Fig. 1</figref>;</dd>
<dt>Fig. 6</dt><dd>illustrates an example of initialised feature extraction methods to detect a Region of Interest (ROI), and</dd>
<dt>Fig. 7:</dt><dd>is a comparison of an original and a cropped image.</dd>
</dl></p>
<p id="p0008" num="0008">The invention is aiming at file-based production formats (based on a shift from tape records to tapeless records) which are allowing the usage of various metadata for post-production, programme exchange and archiving. Such metadata are included in a container format containing video data and metadata. Such metadata include content-related- information which describes the type of genre as well as specific information related to details of the production procedure. The generated metadata are made available in a container format containing video and metadata. Such container format allows a multiplex of different data in a synchronised way, either as file or stream. The combination of metadata information with known feature extraction methods is resulting in the inventive method which is individually adaptable to a wide range of contents.<!-- EPO <DP n="4"> --></p>
<p id="p0009" num="0009">The overall system according to <figref idref="f0001">Fig. 1</figref> is illustrating a block diagram comprising the three blocks 1, 2 and 3. The video and metadata are inputted into block 1. The metadata can be aggregated from one or several sources. In a further step, the collection of data is parsed and important information is categorized in a useful structure. The resulting data is sent to block 2 and partly to block 3. The video content is passed via "Video"-output line to block 2. Block 2 is the feature extraction module performing the step of shot detection and the step of feature extration and following object tracking as is described in more detail with reference to <figref idref="f0003">Fig. 3</figref>. The feature extraction as performed by block 2 results in n extracted ROIs which are fed to block 3. Block 3 is the cropping module producing the cropping area to be displayed on smaller sized displays of handheld devices. This module can be placed either on production side or in the end device.</p>
<p id="p0010" num="0010">Block 1 performes the aggregation and parsing of metadata as shown in detail in <figref idref="f0002">Fig. 2</figref>. The video is passed through to the video analysis (see <figref idref="f0001">Figure 1</figref>), while the metadata is parsed (analysed) and important information is categorised in a useful structure. Metadata is a content related description using an easy file structure, e.g. XML (Extensible Markup Language). Here, it is roughly distinguished in descriptive data, technical data and optional data. Descriptive data is a content related description. This information can be either static or dynamic. Dynamic means data changing in time is synchronised to the video content, e.g. description of a person appearing in the video. Static data is a description which is valid for the entire video, e.g. type of genre. On the other hand, technical data is related to the format of the essence and can also be static or dynamic. It describes the format of the embedded video. Optional metadata does not describe production-specific technical or descriptive metadata but can give necessary information for the adaption process, e.g. where the copping will be done (on production side or on the end device) or properties of the final video (resolution fame rate, et.).<!-- EPO <DP n="5"> --></p>
<p id="p0011" num="0011">All three metadata types, namely technical, descriptive and optional data are provided to the feature extraction modul (Block 2).</p>
<p id="p0012" num="0012">Block 2 which is the feature extraction module is shown in detail in <figref idref="f0003">Fig. 3</figref>. The video and metadata delivered by the demultiplexing module (block 1) is combined to extract important features in a context. For this, the categorised metadata are used to initialise a dynamically fitted chain of feature extractions adapted to the delivered video content. Those can be motion detection (e.g. Block Matching), morphology filters (e.g. Erosion), edge detection (e.g. Sobel operator), etc. As additional feature extraction, a visual attention model is implemented and used. Such a visual attention system emulates the visual system of human beings. It detects salient low level features (bottom-up features), like main orientation, colours or intensity and combine them similar to the procedure of the human eye.</p>
<p id="p0013" num="0013">Each genre type has a different combination of feature extraction methods and different parameters, which are dynamically controllable by metadata or other information obtained by extracted features. This is depicted in block 2 by a matrix allocating a genre type with specific feature extraction methods. Following, the detected features are weighted by importance, e.g. by their contextual position or size. Relevant and related features are then combined to a ROI and delivered to the tracking tool. The tracking tool identifies the new position and deformation of each initialised ROI in consecutive frames and returns this information to the feature extraction. By this, a permanent communication between feature extraction and tracking tool is guaranteed. This can be used to suppress areas for feature extraction which are already tracked. Finally, one or several ROIs are extracted. The weighting of each feature depends on the context of the present video content. It comes to the decision by an algorithm aggregating and processing all available feature extraction data and metadata. This allocations deliver decision citerions what should be an integral part and how it should be arranged in a new composed image.<!-- EPO <DP n="6"> --></p>
<p id="p0014" num="0014">To explain the feature extraction performed in block 2 in more detail, a short example shown in <figref idref="f0005">Fig. 5</figref> and treating a showjumping scene depicts a possible combination of different feature extraction methods. As already mentioned, the used methods are initialised and combined by available metadata. The most important metadata information is which type of genre is present. Here, that information is used to apply special video analysis methods to detect the position of the horse. <figref idref="f0005">Fig. 5</figref> roughly explains a possible process to get the position and size of the horse and rider. Basic prerequisite in this case is that showjumping is produced with static foreground (horse) and moving background. This leads to an approach to calculate the offset of moving background between two consecutive frames (depicted with f<sub>0</sub> and f<sub>1</sub> in <figref idref="f0002">Figure 2</figref>). Knowing the offset, the latter frame can be repositioned by it and subtracted from the previous one. The results are dark areas where background matches and bright areas where pixels differ from the background. After applying some filters to gain the difference between dark and bright, clearly bring out a rough shape of the horse and rider (shown at the bottom of <figref idref="f0005">Fig. 5</figref>). Once detected, it would be desirable to keep this ROI as long as it is visible in the following frames. For this, the tracking application is initialised receiving the initialised detected horse and matches it in consecutive frames. Updated tracking positions in subsequent frames are returned from the tracking module to the Feature Extraction Module (block 2).</p>
<p id="p0015" num="0015">Block 3 and 4 (<figref idref="f0004">Figures 4</figref> and <figref idref="f0005">5</figref>) depict the cropping modules in detail more detail. The cropping modules mainly have the function to crop a well composed image part. For this, all received ROIs, classified by importance, are used to aid the decision of positioning the cropped area. Besides simply choosing an area for cropping, it has to be considered whether an anamorphic video is present (16:9 aspect ratio horizontally clinched to 4:3) and square or non-square pixels composes the image. Dependent of the image format of the target display, these possibilities must be considered and adapted to avoid image distortions. The<!-- EPO <DP n="7"> --> cropping pocess is accomplished on the transmitter side (block 3) or on the receiving device itself (block 4). Both possibilities use the same procedure. The only difference is the way to feed information about the requirements of the end devices. On transmission side, this is done by the optional metadata which also describe the requirements of the video format for the distribution. On the end device, this information is available by the device itself. This has the advantage that the entire original video plus the ROI information is available and thus the adaption can be individually done. Compared to the option doing the processing on transmission side, the cropping area is once defined and provided to all end devices.</p>
<p id="p0016" num="0016">In addition to the cropping parameters as mentioned above, viewing conditions for the different displays have to be considered. By this, a benchmark defines which size the cropped area should have compared to the original image. Such a benchmark can be determined by a comparison of viewing distances for both display resolution. Those considerations may change the size and shape of the cropped area again and has to be adapted once more. After coming to a decision of a properly cropped area considering all content-related and technical issues, the image has to be scaled to the size of the target display.</p>
<p id="p0017" num="0017">As shown above, the example of extracting features for showjumping (<figref idref="f0006">Fig. 6</figref>) is a specially-tailored method and would not work properly for other types of content, e.g. soccer. Therefore, the presented approach requires metadata to choose the right extraction method for the present type of genre. In the end, it is desirable to adapt video content like depicted in <figref idref="f0007">Figure 7</figref>.</p>
<p id="p0018" num="0018">The proposed methodology describes a workflow controlled by metadata. By this, a specially-tailored feature extraction and cropping method can be applied to increase the reliability of video analysis and aesthetic of the composed image.<!-- EPO <DP n="8"> --></p>
<p id="p0019" num="0019">The video analysis and cropping example of showjumping explained above is just for demonstration purposes of one possible workflow more in detail. They are not part of the patent application. Moreover, the scope of application is not limited to tv productions. The invention can be generally used where video cropping is required and metadata in a known structure is available, e.g. for web streaming or local stored videos.</p>
</description>
<claims id="claims01" lang="en"><!-- EPO <DP n="9"> -->
<claim id="c-en-01-0001" num="0001">
<claim-text>A method of adapting video images to small screen sizes, in particular to small screen sizes of portable handheld terminals, said method starting from a metadata aggregation and the corresponding video, wherein
<claim-text>(a) the video is passed through to a video analysis to deliver video while metadata is parsed,</claim-text>
<claim-text>(b) the separated video and metadata are combined to extract important features in a context wherein important information from the metadata is categorised and is used to initialise a dynamically fitted chain of feature extraction steps adapted to the delivered video content,<br/>
wherein said metadata are distinguished at least in descriptive data and technical data, wherein said descriptive data are a content related description which can be either static or dynamic data, said dynamic data being data changing in time and synchronised to the video content, and said static data being a description which is valid for the entire video, said static descriptive data including type of genre, and said technical data being related to the format of the embedded video which can also be static or dynamic,<br/>
wherein a combination of feature extraction methods is selected according to said type of genre,</claim-text>
<claim-text>(c) extracted important features are combined to define regions of interest (ROI) which are searched in consecutive video frames by object tracking, said object tracking identifies the new position and deformation of each initialised ROI in consecutive video frames and returns this information to the feature extraction thereby obtaining a permanent communication between said feature extraction and said object tracking, wherein said features are extracted based on said selected feature extraction methods,</claim-text>
<claim-text>(d) one or several ROIs are extracted and inputted video frame by video frame into a cropping step</claim-text>
<claim-text>(e) based on weighting information of each feature which depends on the context of the present video content, a well composed image part is cropped by classifying said supplied ROIs by importance, and</claim-text>
<claim-text>(f) said cropped image area(s) are scaled to the desired small screen size.</claim-text></claim-text></claim>
<claim id="c-en-01-0002" num="0002">
<claim-text>A method according to claim 1, wherein<br/>
said metadata are further distinguished in optional data.</claim-text></claim>
<claim id="c-en-01-0003" num="0003">
<claim-text>A method according to claim 1 or 2 , wherein said technical data are used to detect of scene changes (shots) in the video images.</claim-text></claim>
<claim id="c-en-01-0004" num="0004">
<claim-text>A method according to one of claims 1 to 3, wherein said permanent<!-- EPO <DP n="10"> --> communication between said feature extraction steps and said object tracking step is used to suppress areas for feature extraction which are already tracked.</claim-text></claim>
<claim id="c-en-01-0005" num="0005">
<claim-text>A method according to one of claims 1 to 4, wherein the extracted important features are weighted by importance, e.g. by their position or size, wherein relevant and related features are combined to a weighted region of interest (ROI).</claim-text></claim>
<claim id="c-en-01-0006" num="0006">
<claim-text>A method according to one of claims 1 to 5, wherein said classifying of said supplied ROIs in said cropping step examines whether an anamorphic video is present (16:9 aspect ratio horizontally clinched to 4:3) and square or non-square pixels composes the image, and wherein in the scaling of the image format to the targeted small screen size, the examined parameters are considered and adapted to avoid an image distortion.</claim-text></claim>
<claim id="c-en-01-0007" num="0007">
<claim-text>A method according to one of claims 1 to 6, wherein classifying of said supplied ROIs in said cropping step examines viewing conditions for the different displays thereby determining a benchmark which size the cropped area should have compared to the original image, such determination is made by a comparison of viewing distances for both display resolution.</claim-text></claim>
<claim id="c-en-01-0008" num="0008">
<claim-text>A method according to any of the previous claims, wherein said method starts from a metadata aggregation and the corresponding video, used in post-production or programme exchange or archiving.</claim-text></claim>
</claims>
<claims id="claims02" lang="de"><!-- EPO <DP n="11"> -->
<claim id="c-de-01-0001" num="0001">
<claim-text>Verfahren zum Anpassen von Videobildern an kleine Bildschirmgrößen, insbesondere an kleine Bildschirmgrößen von tragbaren Handterminals, wobei das Verfahren von einer Metadatenaggregation und dem entsprechenden Video startet, wobei
<claim-text>(a) das Video eine Videoanalyse durchläuft, um Video zu liefern, während Metadaten geparst werden,</claim-text>
<claim-text>(b) das getrennte Video und die Metadaten kombiniert werden, um wichtige Merkmale in einem Kontext zu extrahieren, wobei wichtige Informationen von den Metadaten kategorisiert werden und verwendet werden, um eine dynamisch angepasste Kette von Merkmalsextraktionsschritten, die an den gelieferten Videoinhalt angepasst sind, zu initialisieren,<br/>
wobei die Metadaten zumindest in beschreibende Daten und technische Daten unterschieden werden,<br/>
wobei die beschreibenden Daten eine inhaltsbezogene Beschreibung sind, die entweder statische oder dynamische Daten sein können, wobei die dynamischen Daten Daten sind, die sich mit der Zeit ändern und die mit dem Videoinhalt synchronisiert sind, und die statischen Daten eine Beschreibung sind, die für das gesamte Video gültig ist, wobei die statischen beschreibenden Daten einen Genre-Typ beinhalten, und wobei die technischen Daten im Zusammenhang stehen mit dem Format des eingebetteten Videos, welches ebenfalls statisch oder dynamisch sein kann,<br/>
wobei eine Kombination von Merkmalsextraktionsverfahren entsprechend dem Genre-Typ ausgewählt wird,</claim-text>
<claim-text>(c) extrahierte wichtige Merkmale, kombiniert werden, um Bereiche von Interesse (ROI) zu definieren, die in aufeinanderfolgenden Videorahmen mittels einer Objektverfolgung gesucht werden, wobei die Objektverfolgung die neue Position und Deformation von jedem initialisierten ROI in aufeinanderfolgenden Videorahmen identifiziert und diese Information an die Merkmalsextraktion zurückgibt, wodurch eine permanente Kommunikation zwischen der Merkmalsextraktion und der Objektverfolgung erhalten wird, wobei die Merkmale basierend auf den ausgewählten Merkmalsextraktionsverfahren extrahiert werden,</claim-text>
<claim-text>(d) ein oder mehrere ROIs extrahiert werden und Videorahmen für Videorahmen in einen Beschneidungsschritt eingegeben werden,</claim-text>
<claim-text>(e) basierend auf einer Gewichtungsinformation für jedes Merkmal, die von dem Kontext des aktuellen Videoinhalts abhängt, ein wohlkomponierter Bildteil durch Klassifizieren der gelieferten ROIs entsprechend ihrer Wichtigkeit zugeschnitten wird, und<!-- EPO <DP n="12"> --></claim-text>
<claim-text>(f) der/die zugeschnittene(n) Bildbereich(e) auf die gewünschte kleine Bildschirmgröße skaliert wird/werden.</claim-text></claim-text></claim>
<claim id="c-de-01-0002" num="0002">
<claim-text>Verfahren nach Anspruch 1, wobei<br/>
die Metadaten des Weiteren in optionale Daten unterschieden werden.</claim-text></claim>
<claim id="c-de-01-0003" num="0003">
<claim-text>Verfahren nach Anspruch 1 oder 2, wobei die technischen Daten verwendet werden, um Szenenänderungen (Aufnahmen) in den Videobildern zu detektieren.</claim-text></claim>
<claim id="c-de-01-0004" num="0004">
<claim-text>Verfahren nach einem der Ansprüche 1 bis 3, wobei die permanente Kommunikation zwischen den Merkmalsextraktionsschritten und dem Objektverfolgungsschritt verwendet wird, um Bereiche für die Merkmalsextraktion zu unterdrücken, die bereits getrackt werden.</claim-text></claim>
<claim id="c-de-01-0005" num="0005">
<claim-text>Verfahren nach einem der Ansprüche 1 bis 4, wobei die extrahierten wichtigen Merkmale entsprechend ihrer Wichtigkeit gewichtet werden, z.B., durch ihre Position oder Größe, wobei relevante und zugehörige Merkmale zu einem gewichteten Bereich von Interesse (ROI) kombiniert werden.</claim-text></claim>
<claim id="c-de-01-0006" num="0006">
<claim-text>Verfahren nach einem der Ansprüche 1 bis 5, wobei das Klassifizieren der gelieferten ROIs in dem Zuschneideschritt untersucht, ob ein anamorphes Video vorhanden ist (16:9 Aspektverhältnis horizontal auf 4:3 gestaucht) und das Bild aus quadratischen oder nicht-quadratischen Pixeln komponiert ist, und wobei in dem Skalieren des Bildformats auf die anvisierte kleine Bildschirmgröße, die untersuchten Parameter berücksichtigt werden und angepasst werden, um eine Bildverzerrung zu vermeiden.</claim-text></claim>
<claim id="c-de-01-0007" num="0007">
<claim-text>Verfahren nach einem der Ansprüche 1 bis 6, wobei das Klassifizieren der bereitgestellten ROIs in dem Zuschneideschritt Betrachtungsbedingungen für die unterschiedlichen Displays untersucht, wodurch ein Benchmark bestimmt wird, welche Größe der zugeschnittene Bereich im Vergleich zu dem Originalbild haben sollte, wobei diese Bestimmung durch einen Vergleich von Betrachtungsabständen für beide Anzeigeauflösungen gemacht wird.</claim-text></claim>
<claim id="c-de-01-0008" num="0008">
<claim-text>Verfahren nach einem der vorangegangenen Ansprüche, wobei das Verfahren von einer Metadatenaggregation und dem korrespondierenden Video startet, welche in einer Postproduktion oder einem Programmaustausch oder in einer Archivierung verwendet werden.</claim-text></claim>
</claims>
<claims id="claims03" lang="fr"><!-- EPO <DP n="13"> -->
<claim id="c-fr-01-0001" num="0001">
<claim-text>Procédé d'adaptation d'images vidéo à des tailles de petits écrans, en particulier à des tailles de petits écrans de terminaux portables de poche, ledit procédé commençant à partir d'une agrégation de métadonnées et de la vidéo correspondante, dans lequel
<claim-text>(a) la vidéo passe par une analyse vidéo pour diffuser une vidéo tandis que les métadonnées sont analysées syntaxiquement,</claim-text>
<claim-text>(b) la vidéo et les métadonnées séparées sont combinées pour extraire des traits caractéristiques importants dans un contexte dans lequel les informations importantes des métadonnées sont catégorisées et sont utilisées pour initialiser une chaîne à ajustement dynamique d'étapes d'extraction de traits caractéristiques adaptée au contenu vidéo diffusé,<br/>
dans lequel lesdites métadonnées se distinguent au moins en données descriptives et en données techniques,<br/>
dans lequel lesdites données descriptives sont une description liée au contenu qui peut être des données statiques ou dynamiques, lesdites données dynamiques étant des données changeant dans le temps et synchronisées au contenu vidéo, et lesdites données statiques étant une description qui est valable pour la totalité de la vidéo, lesdites données descriptives incluant un type de genre, et lesdites données techniques étant liées au format de la vidéo incorporée qui peut également être statique ou dynamique,<br/>
dans lequel une combinaison de procédé d'extraction de traits caractéristiques est choisie selon ledit type de genre,<!-- EPO <DP n="14"> --></claim-text>
<claim-text>(c) des traits caractéristiques importants extraits sont combinés pour définir des régions d'intérêt (ROI) qui sont recherchées dans des trames vidéo consécutives par suivi d'objet, ledit suivi d'objet identifie la nouvelle position et déformation de chaque ROI initialisée dans des trames vidéo consécutives et renvoie cette information à l'extraction de traits caractéristiques obtenant ainsi une communication constante entre ladite extraction de traits caractéristiques et ledit suivi d'objet, dans lequel lesdits traits caractéristiques sont extraits d'après lesdits procédés d'extraction de traits caractéristiques sélectionnés,</claim-text>
<claim-text>(d) une ou plusieurs ROI sont extraites et entrées trame vidéo par trame vidéo dans une étape de recadrage</claim-text>
<claim-text>(e) d'après des informations de pondération de chaque trait caractéristique qui dépend du contexte du contenu vidéo actuel, une partie d'image bien composée est recadrée en classant lesdites ROI fournies par importance, et</claim-text>
<claim-text>(f) ladite ou lesdites zone (s) d'image recadrée(s) est (sont) mise(s) à la taille de petit écran souhaitée.</claim-text></claim-text></claim>
<claim id="c-fr-01-0002" num="0002">
<claim-text>Procédé selon la revendication 1, dans lequel
<claim-text>lesdites métadonnées se distinguent en outre en données facultatives.</claim-text></claim-text></claim>
<claim id="c-fr-01-0003" num="0003">
<claim-text>Procédé selon la revendication 1 ou 2, dans lequel lesdites données techniques sont utilisées pour détecter des changements de scène (plans) dans les images vidéo.</claim-text></claim>
<claim id="c-fr-01-0004" num="0004">
<claim-text>Procédé selon l'une des revendications 1 à 3, dans lequel ladite communication constante entre lesdites<!-- EPO <DP n="15"> --> étapes d'extraction de traits caractéristiques et ladite étape de suivi d'objet est utilisée pour supprimer des zones d'extraction de traits caractéristiques qui sont déjà suivies.</claim-text></claim>
<claim id="c-fr-01-0005" num="0005">
<claim-text>Procédé selon l'une des revendications 1 à 4, dans lequel les traits caractéristiques importants extraits sont pondérés par importance, par exemple par leur position ou taille, dans lequel des traits caractéristiques pertinents et liés sont combinés à une région d'intérêt (ROI) pondérée.</claim-text></claim>
<claim id="c-fr-01-0006" num="0006">
<claim-text>Procédé selon l'une des revendications 1 à 5, dans lequel ledit classement desdites ROI fournies à ladite étape de recadrage examine si une vidéo anamorphosée est présente (rapport de côté de 16 : 9 résolu à l'horizontale à 4 : 3) et des pixels carrés ou non carrés composent l'image, et dans lequel à la mise à l'échelle du format d'image à la taille de petit écran ciblée, les paramètres examinés sont pris en considération et adaptés pour éviter une distorsion d'image.</claim-text></claim>
<claim id="c-fr-01-0007" num="0007">
<claim-text>Procédé selon l'une des revendications 1 à 6, dans lequel le classement desdites ROI fournies à ladite étape de recadrage examine des conditions de visualisation pour les différents afficheurs déterminant ainsi en point de repère quelle taille la zone recadrée doit avoir comparée à l'image d'origine, une telle détermination se fait par une comparaison de distances de visualisation pour les deux résolutions d'afficheur.</claim-text></claim>
<claim id="c-fr-01-0008" num="0008">
<claim-text>Procédé selon l'une quelconque des revendications précédentes, dans lequel ledit procédé commence à partir<!-- EPO <DP n="16"> --> d'une agrégation de métadonnées et de la vidéo correspondante, utilisé en post-production ou en échange ou archivage de programmes.</claim-text></claim>
</claims>
<drawings id="draw" lang="en"><!-- EPO <DP n="17"> -->
<figure id="f0001" num="1"><img id="if0001" file="imgf0001.tif" wi="165" he="229" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="18"> -->
<figure id="f0002" num="2"><img id="if0002" file="imgf0002.tif" wi="165" he="224" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="19"> -->
<figure id="f0003" num="3"><img id="if0003" file="imgf0003.tif" wi="165" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="20"> -->
<figure id="f0004" num="4"><img id="if0004" file="imgf0004.tif" wi="156" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="21"> -->
<figure id="f0005" num="5"><img id="if0005" file="imgf0005.tif" wi="151" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="22"> -->
<figure id="f0006" num="6"><img id="if0006" file="imgf0006.tif" wi="129" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="23"> -->
<figure id="f0007" num="7"><img id="if0007" file="imgf0007.tif" wi="139" he="227" img-content="drawing" img-format="tif"/></figure>
</drawings>
<ep-reference-list id="ref-list">
<heading id="ref-h0001"><b>REFERENCES CITED IN THE DESCRIPTION</b></heading>
<p id="ref-p0001" num=""><i>This list of references cited by the applicant is for the reader's convenience only. It does not form part of the European patent document. Even though great care has been taken in compiling the references, errors or omissions cannot be excluded and the EPO disclaims all liability in this regard.</i></p>
<heading id="ref-h0002"><b>Patent documents cited in the description</b></heading>
<p id="ref-p0002" num="">
<ul id="ref-ul0001" list-style="bullet">
<li><patcit id="ref-pcit0001" dnum="WO2006056311A"><document-id><country>WO</country><doc-number>2006056311</doc-number><kind>A</kind></document-id></patcit><crossref idref="pcit0001">[0003]</crossref></li>
</ul></p>
</ep-reference-list>
</ep-patent-document>
