substtml.c 30.4 KB
Newer Older
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22
/*****************************************************************************
 * substtml.c : TTML subtitles decoder
 *****************************************************************************
 * Copyright (C) 2015 VLC authors and VideoLAN
 *
 * Authors: Hugo Beauzée-Luyssen <hugo@beauzee.fr>
 *          Sushma Reddy <sushma.reddy@research.iiit.ac.in>
 *
 * This program is free software; you can redistribute it and/or modify it
 * under the terms of the GNU Lesser General Public License as published by
 * the Free Software Foundation; either version 2.1 of the License, or
 * (at your option) any later version.
 *
 * This program is distributed in the hope that it will be useful,
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
 * GNU Lesser General Public License for more details.
 *
 * You should have received a copy of the GNU Lesser General Public License
 * along with this program; if not, write to the Free Software Foundation,
 * Inc., 51 Franklin Street, Fifth Floor, Boston MA 02110-1301, USA.
 *****************************************************************************/
Sushma Reddy's avatar
Sushma Reddy committed
23

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
24 25 26 27 28 29 30 31
#ifdef HAVE_CONFIG_H
# include "config.h"
#endif

#include <vlc_common.h>
#include <vlc_plugin.h>
#include <vlc_modules.h>
#include <vlc_codec.h>
Sushma Reddy's avatar
Sushma Reddy committed
32 33 34
#include <vlc_xml.h>
#include <vlc_stream.h>
#include <vlc_text_style.h>
35
#include <vlc_charset.h>
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
36 37 38

#include "substext.h"

Sushma Reddy's avatar
Sushma Reddy committed
39 40
#include <ctype.h>

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
41 42 43 44 45 46 47 48 49
#define ALIGN_TEXT N_("Subtitle justification")
#define ALIGN_LONGTEXT N_("Set the justification of subtitles")

/*****************************************************************************
 * Module descriptor.
 *****************************************************************************/
static int  OpenDecoder   ( vlc_object_t * );
static void CloseDecoder  ( vlc_object_t * );

Sushma Reddy's avatar
Sushma Reddy committed
50 51
static text_segment_t *ParseTTMLSubtitles( decoder_t *, subpicture_updater_sys_t *, char * );

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
52 53 54 55 56 57 58 59 60 61 62 63 64 65
vlc_module_begin ()
    set_capability( "decoder", 10 )
    set_shortname( N_("TTML decoder"))
    set_description( N_("TTML subtitles decoder") )
    set_callbacks( OpenDecoder, CloseDecoder )
    set_category( CAT_INPUT )
    set_subcategory( SUBCAT_INPUT_SCODEC )
    add_integer( "ttml-align", 0, ALIGN_TEXT, ALIGN_LONGTEXT, false )
vlc_module_end ();

/*****************************************************************************
 * Local prototypes
 *****************************************************************************/

Sushma Reddy's avatar
Sushma Reddy committed
66 67 68 69 70 71 72 73 74
typedef struct
{
    char*           psz_styleid;
    text_style_t*   font_style;
    int             i_align;
    int             i_margin_h;
    int             i_margin_v;
    int             i_margin_percent_h;
    int             i_margin_percent_v;
75 76
    int             i_direction;
    bool            b_direction_set;
Sushma Reddy's avatar
Sushma Reddy committed
77 78
}  ttml_style_t;

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
79 80
struct decoder_sys_t
{
Sushma Reddy's avatar
Sushma Reddy committed
81 82 83
    int                     i_align;
    ttml_style_t**          pp_styles;
    size_t                  i_styles;
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
84 85
};

86 87 88 89 90 91 92 93
enum
{
    UNICODE_BIDI_LTR = 0,
    UNICODE_BIDI_RTL = 1,
    UNICODE_BIDI_EMBEDDED = 2,
    UNICODE_BIDI_OVERRIDE = 4,
};

94 95 96 97 98 99 100 101
static int tagnamecmp( char const* tagname, char const* needle )
{
    if( !strncasecmp( "tt:", tagname, 3 ) )
        tagname += 3;

    return strcasecmp( tagname, needle );
}

102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118
static void MergeTTMLStyle( ttml_style_t *p_dst, const ttml_style_t *p_src)
{
    text_style_Merge( p_dst->font_style, p_src->font_style, false );
    if( !( p_dst->i_align & SUBPICTURE_ALIGN_MASK ) )
        p_dst->i_align |= p_src->i_align;

    if( !p_dst->i_margin_h )
        p_dst->i_margin_h = p_src->i_margin_h;

    if( !p_dst->i_margin_v )
        p_dst->i_margin_v = p_src->i_margin_v;

    if( !p_dst->i_margin_percent_h )
        p_dst->i_margin_percent_h = p_src->i_margin_percent_h;

    if( !p_dst->i_margin_percent_v )
        p_dst->i_margin_percent_v = p_src->i_margin_percent_v;
119 120 121 122 123 124

    if( !p_dst->b_direction_set )
    {
        p_dst->i_direction = p_src->i_direction;
        p_dst->b_direction_set = p_src->b_direction_set;
    }
125 126
}

127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150
static ttml_style_t* DuplicateStyle( ttml_style_t* p_style_src )
{
    ttml_style_t* p_style = calloc( 1, sizeof( *p_style ) );
    if( unlikely( p_style == NULL ) )
        return NULL;

    *p_style = *p_style_src;
    p_style->psz_styleid = strdup( p_style_src->psz_styleid );
    if( unlikely( p_style->psz_styleid == NULL ) )
    {
        free( p_style );
        return NULL;
    }

    p_style->font_style = text_style_Duplicate( p_style_src->font_style );
    if( unlikely( p_style->font_style == NULL ) )
    {
        free( p_style->psz_styleid );
        free( p_style );
        return NULL;
    }
    return p_style;
}

151 152 153 154 155 156 157
static void CleanupStyle( ttml_style_t* p_ttml_style )
{
    text_style_Delete( p_ttml_style->font_style );
    free( p_ttml_style->psz_styleid );
    free( p_ttml_style );
}

Sushma Reddy's avatar
Sushma Reddy committed
158 159
static ttml_style_t *FindTextStyle( decoder_t *p_dec, const char *psz_style )
{
160
    decoder_sys_t  *p_sys = p_dec->p_sys;
Sushma Reddy's avatar
Sushma Reddy committed
161 162 163 164

    for( size_t i = 0; i < p_sys->i_styles; i++ )
    {
        if( !strcmp( p_sys->pp_styles[i]->psz_styleid, psz_style ) )
165 166
            return DuplicateStyle( p_sys->pp_styles[i] );

Sushma Reddy's avatar
Sushma Reddy committed
167 168 169 170
    }
    return NULL;
}

171
typedef struct  style_stack_t
Sushma Reddy's avatar
Sushma Reddy committed
172
{
173 174 175
    ttml_style_t*  p_style;
    struct style_stack_t* p_next;
} style_stack_t ;
Sushma Reddy's avatar
Sushma Reddy committed
176 177 178

static bool PushStyle( style_stack_t **pp_stack, ttml_style_t* p_style )
{
179 180
    style_stack_t* p_entry = malloc( sizeof( *p_entry ) );
    if( unlikely( p_entry == NULL ) )
Sushma Reddy's avatar
Sushma Reddy committed
181 182 183 184 185 186 187 188 189
        return false;
    p_entry->p_style = p_style;
    p_entry->p_next = *pp_stack;
    *pp_stack = p_entry;
    return true;
}

static void PopStyle( style_stack_t** pp_stack )
{
190
    if( *pp_stack == NULL )
Sushma Reddy's avatar
Sushma Reddy committed
191 192
        return;
    style_stack_t* p_next = (*pp_stack)->p_next;
193
    CleanupStyle( (*pp_stack)->p_style );
Sushma Reddy's avatar
Sushma Reddy committed
194 195 196 197 198 199
    free( *pp_stack );
    *pp_stack = p_next;
}

static void ClearStack( style_stack_t* p_stack )
{
200
    while( p_stack != NULL )
Sushma Reddy's avatar
Sushma Reddy committed
201 202
    {
        style_stack_t* p_next = p_stack->p_next;
203
        CleanupStyle( p_stack->p_style );
Sushma Reddy's avatar
Sushma Reddy committed
204 205 206 207 208 209 210
        free( p_stack );
        p_stack = p_next;
    }
}

static text_style_t* CurrentStyle( style_stack_t* p_stack )
{
211
    if( p_stack == NULL )
Sushma Reddy's avatar
Sushma Reddy committed
212
        return text_style_Create( STYLE_NO_DEFAULTS );
213

Sushma Reddy's avatar
Sushma Reddy committed
214 215 216
    return text_style_Duplicate( p_stack->p_style->font_style );
}

217
static ttml_style_t* ParseTTMLStyle( decoder_t *p_dec, xml_reader_t* p_reader, const char* psz_node_name )
Sushma Reddy's avatar
Sushma Reddy committed
218 219 220 221 222
{
    decoder_sys_t* p_sys = p_dec->p_sys;
    ttml_style_t *p_ttml_style = NULL;
    ttml_style_t *p_base_style = NULL;

223 224 225
    p_ttml_style = calloc( 1, sizeof( ttml_style_t ) );
    if( unlikely( !p_ttml_style ) )
        return NULL;
Sushma Reddy's avatar
Sushma Reddy committed
226 227 228 229 230

    p_ttml_style->font_style = text_style_Create( STYLE_NO_DEFAULTS );
    if( unlikely( !p_ttml_style->font_style ) )
    {
        free( p_ttml_style );
231
        return NULL;
Sushma Reddy's avatar
Sushma Reddy committed
232 233 234 235 236 237
    }

    const char *attr, *val;

    while( (attr = xml_ReaderNextAttr( p_reader, &val ) ) )
    {
238 239
        /* searching previous styles for inheritence */
        if( !strcasecmp( attr, "style" ) || !strcasecmp( attr, "region" ) )
Sushma Reddy's avatar
Sushma Reddy committed
240
        {
241
            if( !tagnamecmp( psz_node_name, "style" ) || !tagnamecmp( psz_node_name, "region" ) )
242 243 244 245 246
            {
                for( size_t i = 0; i < p_sys->i_styles; i++ )
                {
                    if( !strcasecmp( p_sys->pp_styles[i]->psz_styleid, val ) )
                    {
247
                        p_base_style = p_sys->pp_styles[i];
248 249 250 251 252 253 254 255 256 257 258 259 260 261 262
                        break;
                    }
                }
            }
            /*
            * In p nodes, style attribute has this format :
            * style="style1 style2 style3" where style1 and style2 are
            * style applied on the parents of p in that order.
            *
            * In span node, we can apply several styles in the same order than
            * in p nodes with the same inheritance order.
            *
            * In order to preserve this style predominance, we merge the styles
            * in the from right to left ( the right one being predominant ) .
            */
263
            else if( !tagnamecmp( psz_node_name, "p" ) || !tagnamecmp( psz_node_name, "span" ) )
264 265 266 267
            {
                char *tmp;
                char *value = strdup( val );
                if( unlikely( value == NULL ) )
268 269
                {
                    CleanupStyle( p_ttml_style );
270
                    return NULL;
271
                }
272 273

                char *token = strtok_r( value , " ", &tmp );
274 275 276 277 278 279 280 281 282 283

                if( token == NULL )
                {
                    msg_Warn( p_dec, "No IDREF specified in attribute "
                                     "'%s' on tag '%s', ignoring.", attr,
                                     psz_node_name );
                    free( value );
                    continue;
                }

284 285 286
                ttml_style_t* p_style = FindTextStyle( p_dec, token );
                if( p_style == NULL )
                {
287
                    msg_Warn( p_dec, "IDREF '%s' in '%s' not found", token, attr );
288 289 290 291 292 293 294 295 296
                    free( value );
                    break;
                }

                while( ( token = strtok_r( NULL, " ", &tmp) ) != NULL )
                {
                    ttml_style_t* p_next_style = FindTextStyle( p_dec, token );
                    if( p_next_style == NULL )
                    {
297
                        msg_Warn( p_dec, "IDREF '%s' in '%s' not found", token, attr );
298 299 300 301 302 303 304 305 306 307 308 309
                        break;
                    }
                    MergeTTMLStyle( p_next_style, p_style );
                    CleanupStyle( p_style );
                    p_style = p_next_style;
                }
                MergeTTMLStyle( p_style, p_ttml_style );
                free( value );
                CleanupStyle( p_ttml_style );
                p_ttml_style = p_style;
            }
            else
Sushma Reddy's avatar
Sushma Reddy committed
310
            {
311 312
                ttml_style_t* p_style = FindTextStyle( p_dec, val );
                if( p_style == NULL )
Sushma Reddy's avatar
Sushma Reddy committed
313
                {
314
                    msg_Warn( p_dec, "IDREF '%s' in '%s' not found", val, attr );
Sushma Reddy's avatar
Sushma Reddy committed
315 316
                    break;
                }
317 318 319
                MergeTTMLStyle( p_style , p_ttml_style );
                CleanupStyle( p_ttml_style );
                p_ttml_style = p_style;
Sushma Reddy's avatar
Sushma Reddy committed
320 321
            }
        }
322
        else if( !strcasecmp( "xml:id", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
323 324 325 326
        {
            free( p_ttml_style->psz_styleid );
            p_ttml_style->psz_styleid = strdup( val );
        }
327
        else if( !strcasecmp ( "tts:fontFamily", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
328 329 330
        {
            free( p_ttml_style->font_style->psz_fontname );
            p_ttml_style->font_style->psz_fontname = strdup( val );
331 332 333 334 335
            if( unlikely( p_ttml_style->font_style->psz_fontname == NULL ) )
            {
                CleanupStyle( p_ttml_style );
                return NULL;
            }
Sushma Reddy's avatar
Sushma Reddy committed
336
        }
337 338 339 340 341 342 343
        else if( !strcasecmp( "tts:opacity", attr ) )
        {
            p_ttml_style->font_style->i_background_alpha = atoi( val );
            p_ttml_style->font_style->i_font_alpha = atoi( val );
            p_ttml_style->font_style->i_features |= STYLE_HAS_BACKGROUND_ALPHA | STYLE_HAS_FONT_ALPHA;
        }
        else if( !strcasecmp( "tts:fontSize", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
344
        {
345
            char* psz_end = NULL;
346
            float size = us_strtof( val, &psz_end );
347 348 349 350
            if( *psz_end == '%' )
                p_ttml_style->font_style->f_font_relsize = size;
            else
                p_ttml_style->font_style->i_font_size = (int)( size + 0.5 );
Sushma Reddy's avatar
Sushma Reddy committed
351
        }
352
        else if( !strcasecmp( "tts:color", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
353
        {
354
            unsigned int i_color = vlc_html_color( val, NULL );
Sushma Reddy's avatar
Sushma Reddy committed
355 356 357 358
            p_ttml_style->font_style->i_font_color = (i_color & 0xffffff);
            p_ttml_style->font_style->i_font_alpha = (i_color & 0xFF000000) >> 24;
            p_ttml_style->font_style->i_features |= STYLE_HAS_FONT_COLOR | STYLE_HAS_FONT_ALPHA;
        }
359
        else if( !strcasecmp( "tts:backgroundColor", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
360
        {
361
            unsigned int i_color = vlc_html_color( val, NULL );
Sushma Reddy's avatar
Sushma Reddy committed
362 363 364 365
            p_ttml_style->font_style->i_background_color = i_color & 0xFFFFFF;
            p_ttml_style->font_style->i_background_alpha = (i_color & 0xFF000000) >> 24;
            p_ttml_style->font_style->i_features |= STYLE_HAS_BACKGROUND_COLOR
                                                      | STYLE_HAS_BACKGROUND_ALPHA;
366
            p_ttml_style->font_style->i_style_flags |= STYLE_BACKGROUND;
Sushma Reddy's avatar
Sushma Reddy committed
367
        }
368
        else if( !strcasecmp( "tts:textAlign", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
369
        {
370
            if( !strcasecmp ( "left", val ) )
371
                p_ttml_style->i_align = SUBPICTURE_ALIGN_TOP | SUBPICTURE_ALIGN_LEFT;
372
            else if( !strcasecmp ( "right", val ) )
373
                p_ttml_style->i_align = SUBPICTURE_ALIGN_TOP | SUBPICTURE_ALIGN_RIGHT;
374
            else if( !strcasecmp ( "center", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
375
                p_ttml_style->i_align = SUBPICTURE_ALIGN_BOTTOM;
376
            else if( !strcasecmp ( "start", val ) )
377
                p_ttml_style->i_align = SUBPICTURE_ALIGN_BOTTOM | SUBPICTURE_ALIGN_LEFT;
378
            else if( !strcasecmp ( "end", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
379 380
                p_ttml_style->i_align = SUBPICTURE_ALIGN_BOTTOM | SUBPICTURE_ALIGN_RIGHT;
        }
381
        else if( !strcasecmp( "tts:fontStyle", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
382
        {
383
            if( !strcasecmp ( "italic", val ) || !strcasecmp ( "oblique", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
384 385 386 387 388
                p_ttml_style->font_style->i_style_flags |= STYLE_ITALIC;
            else
                p_ttml_style->font_style->i_style_flags &= ~STYLE_ITALIC;
            p_ttml_style->font_style->i_features |= STYLE_HAS_FLAGS;
        }
389
        else if( !strcasecmp ( "tts:fontWeight", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
390
        {
391
            if( !strcasecmp ( "bold", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
392 393 394 395 396
                p_ttml_style->font_style->i_style_flags |= STYLE_BOLD;
            else
                p_ttml_style->font_style->i_style_flags &= ~STYLE_BOLD;
            p_ttml_style->font_style->i_features |= STYLE_HAS_FLAGS;
        }
397
        else if( !strcasecmp ( "tts:textDecoration", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
398
        {
399
            if( !strcasecmp ( "underline", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
400
                p_ttml_style->font_style->i_style_flags |= STYLE_UNDERLINE;
401
            else if( !strcasecmp ( "noUnderline", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
402
                p_ttml_style->font_style->i_style_flags &= ~STYLE_UNDERLINE;
403
            if( !strcasecmp ( "lineThrough", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
404
                p_ttml_style->font_style->i_style_flags |= STYLE_STRIKEOUT;
405
            else if( !strcasecmp ( "noLineThrough", val ) )
Sushma Reddy's avatar
Sushma Reddy committed
406 407 408
                p_ttml_style->font_style->i_style_flags &= ~STYLE_STRIKEOUT;
            p_ttml_style->font_style->i_features |= STYLE_HAS_FLAGS;
        }
409
        else if( !strcasecmp ( "tts:origin", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
410 411
        {
            const char *psz_token = val;
412
            while( isspace( *psz_token ) )
Sushma Reddy's avatar
Sushma Reddy committed
413 414 415
                psz_token++;

            const char *psz_separator = strchr( psz_token, ' ' );
416
            if( psz_separator == NULL )
Sushma Reddy's avatar
Sushma Reddy committed
417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432
            {
                msg_Warn( p_dec, "Invalid origin attribute: \"%s\"", val );
                continue;
            }
            const char *psz_percent_sign = strchr( psz_token, '%' );

            if( psz_percent_sign != NULL && psz_percent_sign < psz_separator )
            {
                p_ttml_style->i_margin_h = 0;
                p_ttml_style->i_margin_percent_h = atoi( psz_token );
            }
            else
            {
                p_ttml_style->i_margin_h = atoi( psz_token );
                p_ttml_style->i_margin_percent_h = 0;
            }
433
            while( isspace( *psz_separator ) )
Sushma Reddy's avatar
Sushma Reddy committed
434 435 436 437 438 439 440 441 442 443 444 445 446 447
                psz_separator++;
            psz_token = psz_separator;
            psz_percent_sign = strchr( psz_token, '%' );
            if( psz_percent_sign != NULL )
            {
                p_ttml_style->i_margin_v = 0;
                p_ttml_style->i_margin_percent_v = atoi( val );
            }
            else
            {
                p_ttml_style->i_margin_v = atoi( val );
                p_ttml_style->i_margin_percent_v = 0;
            }
        }
448
        else if( !strcasecmp( "tts:textOutline", attr ) )
Sushma Reddy's avatar
Sushma Reddy committed
449 450 451 452 453 454
        {
            char *value = strdup( val );
            char* psz_saveptr = NULL;
            char* token = strtok_r( value, " ", &psz_saveptr );
            // <color>? <length> <length>?
            bool b_ok = false;
455
            unsigned int color = vlc_html_color( token, &b_ok );
456
            if( b_ok )
Sushma Reddy's avatar
Sushma Reddy committed
457 458 459 460 461 462 463
            {
                p_ttml_style->font_style->i_outline_color = color & 0xFFFFFF;
                p_ttml_style->font_style->i_outline_alpha = (color & 0xFF000000) >> 24;
                token = strtok_r( NULL, " ", &psz_saveptr );
            }
            char* psz_end = NULL;
            int i_outline_width = strtol( token, &psz_end, 10 );
464
            if( psz_end != token )
Sushma Reddy's avatar
Sushma Reddy committed
465 466 467 468 469 470
            {
                // Assume unit is pixel, and ignore border radius
                p_ttml_style->font_style->i_outline_width = i_outline_width;
            }
            free( value );
        }
471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505
        else if( !strcasecmp( "tts:direction", attr ) )
        {
            if( !strcasecmp( "rtl", val ) )
            {
                p_ttml_style->i_direction |= UNICODE_BIDI_RTL;
                p_ttml_style->b_direction_set = true;
            }
            else if( !strcasecmp( "ltr", val ) )
            {
                p_ttml_style->i_direction |= UNICODE_BIDI_LTR;
                p_ttml_style->b_direction_set = true;
            }
        }
        else if( !strcasecmp( "tts:unicodeBidi", attr ) )
        {
                if( !strcasecmp( "bidiOverride", val ) )
                    p_ttml_style->i_direction |= UNICODE_BIDI_OVERRIDE & ~UNICODE_BIDI_EMBEDDED;
                else if( !strcasecmp( "embed", val ) )
                    p_ttml_style->i_direction |= UNICODE_BIDI_EMBEDDED & ~UNICODE_BIDI_OVERRIDE;
        }
        else if( !strcasecmp( "tts:writingMode", attr ) )
        {
            if( !strcasecmp( "rl", val ) || !strcasecmp( "rltb", val ) )
            {
                p_ttml_style->i_direction = UNICODE_BIDI_RTL | UNICODE_BIDI_OVERRIDE;
                p_ttml_style->i_align = SUBPICTURE_ALIGN_BOTTOM | SUBPICTURE_ALIGN_RIGHT;
                p_ttml_style->b_direction_set = true;
            }
            else if( !strcasecmp( "lr", val ) || !strcasecmp( "lrtb", val ) )
            {
                p_ttml_style->i_direction = UNICODE_BIDI_LTR | UNICODE_BIDI_OVERRIDE;
                p_ttml_style->i_align = SUBPICTURE_ALIGN_BOTTOM | SUBPICTURE_ALIGN_LEFT;
                p_ttml_style->b_direction_set = true;
            }
        }
Sushma Reddy's avatar
Sushma Reddy committed
506
    }
507
    if( p_base_style != NULL )
Sushma Reddy's avatar
Sushma Reddy committed
508
    {
509
        MergeTTMLStyle( p_ttml_style, p_base_style );
Sushma Reddy's avatar
Sushma Reddy committed
510
    }
511
    if( p_ttml_style->psz_styleid == NULL )
Sushma Reddy's avatar
Sushma Reddy committed
512
    {
513 514
        CleanupStyle( p_ttml_style );
        return NULL;
Sushma Reddy's avatar
Sushma Reddy committed
515
    }
516
    return p_ttml_style;
Sushma Reddy's avatar
Sushma Reddy committed
517 518 519 520
}

static void ParseTTMLStyles( decoder_t* p_dec )
{
521
    stream_t* p_stream = vlc_stream_MemoryNew( p_dec, (uint8_t*)p_dec->fmt_in.p_extra, p_dec->fmt_in.i_extra, true );
Sushma Reddy's avatar
Sushma Reddy committed
522 523 524 525 526 527
    if( unlikely( p_stream == NULL ) )
        return ;

    xml_reader_t* p_reader = xml_ReaderCreate( p_dec, p_stream );
    if( unlikely( p_reader == NULL ) )
    {
528
        vlc_stream_Delete( p_stream );
Sushma Reddy's avatar
Sushma Reddy committed
529 530
        return ;
    }
531 532
    const char* psz_node_name;
    int i_type = xml_ReaderNextNode( p_reader, &psz_node_name );
Sushma Reddy's avatar
Sushma Reddy committed
533

534
    if( i_type == XML_READER_STARTELEM && !tagnamecmp( psz_node_name, "tt" ) )
Sushma Reddy's avatar
Sushma Reddy committed
535
    {
536
        int i_type = xml_ReaderNextNode( p_reader, &psz_node_name );
537

538
        while( i_type != XML_READER_STARTELEM || tagnamecmp( psz_node_name, "head" ) )
539 540
            i_type = xml_ReaderNextNode( p_reader, &psz_node_name );

Sushma Reddy's avatar
Sushma Reddy committed
541 542
        do
        {
543
            /* region and style tag are respectively inside layout and styling tags */
544
            if( !tagnamecmp( psz_node_name, "styling" ) || !tagnamecmp( psz_node_name, "layout" ) )
Sushma Reddy's avatar
Sushma Reddy committed
545
            {
546
                i_type = xml_ReaderNextNode( p_reader, &psz_node_name );
547
                while( i_type != XML_READER_ENDELEM )
Sushma Reddy's avatar
Sushma Reddy committed
548
                {
549 550 551 552 553 554 555 556 557
                    ttml_style_t* p_ttml_style = ParseTTMLStyle( p_dec, p_reader, psz_node_name );
                    if ( p_ttml_style == NULL )
                    {
                        xml_ReaderDelete( p_reader );
                        vlc_stream_Delete( p_stream );
                        return;
                    }
                    decoder_sys_t* p_sys = p_dec->p_sys;
                    TAB_APPEND( p_sys->i_styles, p_sys->pp_styles, p_ttml_style );
558
                    i_type = xml_ReaderNextNode( p_reader, &psz_node_name );
Sushma Reddy's avatar
Sushma Reddy committed
559 560
                }
            }
561
            i_type = xml_ReaderNextNode( p_reader, &psz_node_name );
562
        }while( i_type != XML_READER_ENDELEM || tagnamecmp( psz_node_name, "head" ) );
Sushma Reddy's avatar
Sushma Reddy committed
563 564
    }
    xml_ReaderDelete( p_reader );
565
    vlc_stream_Delete( p_stream );
Sushma Reddy's avatar
Sushma Reddy committed
566 567 568 569 570 571 572 573 574
}

static text_segment_t *ParseTTMLSubtitles( decoder_t *p_dec, subpicture_updater_sys_t *p_update_sys, char *psz_subtitle )
{
    stream_t*       p_sub = NULL;
    xml_reader_t*   p_xml_reader = NULL;
    text_segment_t* p_first_segment = NULL;
    text_segment_t* p_current_segment = NULL;
    style_stack_t*  p_style_stack = NULL;
575
    ttml_style_t*   p_style = NULL;
Sushma Reddy's avatar
Sushma Reddy committed
576

577
    p_sub = vlc_stream_MemoryNew( p_dec, (uint8_t*)psz_subtitle, strlen( psz_subtitle ), true );
Sushma Reddy's avatar
Sushma Reddy committed
578 579 580 581 582 583
    if( unlikely( p_sub == NULL ) )
        return NULL;

    p_xml_reader = xml_ReaderCreate( p_dec, p_sub );
    if( unlikely( p_xml_reader == NULL ) )
    {
584
        vlc_stream_Delete( p_sub );
Sushma Reddy's avatar
Sushma Reddy committed
585 586 587 588 589 590 591
        return NULL;
    }

    const char *node;
    int i_type;

    i_type = xml_ReaderNextNode( p_xml_reader, &node );
592
    while( i_type != XML_READER_NONE && i_type > 0 )
Sushma Reddy's avatar
Sushma Reddy committed
593
    {
594 595 596 597
        /*
        * We parse the styles and put them on the style stack
        * until we reach a text node.
        */
598
        if( i_type == XML_READER_STARTELEM && ( !tagnamecmp( node, "p") || !tagnamecmp( node, "span" ) ) )
Sushma Reddy's avatar
Sushma Reddy committed
599
        {
600 601 602 603 604 605 606 607
            p_style = ParseTTMLStyle( p_dec, p_xml_reader, node );
            if( unlikely( p_style == NULL ) )
                goto fail;

            if( p_style_stack != NULL && p_style_stack->p_style != NULL )
                 MergeTTMLStyle( p_style, p_style_stack->p_style );

            if( PushStyle( &p_style_stack, p_style ) == false )
608 609
            {
                CleanupStyle( p_style );
610
                goto fail;
611
            }
612 613 614 615 616 617 618 619 620

        }
        else if( i_type == XML_READER_TEXT )
        {
            /*
            * Once we have a text node, we create a segment, apply the
            * latest style put on the style stack and fill it with the
            * content of the node.
            */
Sushma Reddy's avatar
Sushma Reddy committed
621
            text_segment_t* p_segment = text_segment_New( NULL );
622
            if( unlikely( p_segment == NULL ) )
Sushma Reddy's avatar
Sushma Reddy committed
623
                goto fail;
624 625 626

            p_segment->psz_text = strdup( node );
            if( unlikely( p_segment->psz_text == NULL ) )
Sushma Reddy's avatar
Sushma Reddy committed
627
            {
628 629
                text_segment_Delete( p_segment );
                goto fail;
Sushma Reddy's avatar
Sushma Reddy committed
630
            }
631 632 633 634

            vlc_xml_decode( p_segment->psz_text );
            if( p_segment->style == NULL && p_style_stack == NULL )
            {
Sushma Reddy's avatar
Sushma Reddy committed
635
                p_segment->style = text_style_Create( STYLE_NO_DEFAULTS );
636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653
            }
            else if( p_segment->style == NULL )
            {
                p_segment->style = CurrentStyle( p_style_stack );
                if( p_segment->style->f_font_relsize && !p_segment->style->i_font_size )
                    p_segment->style->i_font_size = (int)( ( p_segment->style->f_font_relsize * STYLE_DEFAULT_FONT_SIZE / 100 ) + 0.5 );

                if( p_style_stack->p_style->i_margin_h )
                    p_update_sys->x = p_style_stack->p_style->i_margin_h;
                else
                    p_update_sys->x = p_style_stack->p_style->i_margin_percent_h;

                if( p_style_stack->p_style->i_margin_v )
                    p_update_sys->y = p_style_stack->p_style->i_margin_v;
                else
                    p_update_sys->y = p_style_stack->p_style->i_margin_percent_v;

                p_update_sys->align |= p_style_stack->p_style->i_align;
654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676
                /*
                * For bidirectionnal support, we use different enum
                * to recognize different cases, en then we add the
                * corresponding unicode character to the text of
                * the text_segment.
                */
                int i_direction = p_style_stack->p_style->i_direction;
                static const struct
                {
                    const char* psz_uni_start;
                    const char* psz_uni_end;
                }p_bidi[] = {
                { "\u2066", "\u2069" },
                { "\u2067", "\u2069" },
                { "\u202A", "\u202C" },
                { "\u202B", "\u202C" },
                { "\u202D", "\u202C" },
                { "\u202E", "\u202C" },
                };
                if( p_style_stack->p_style->b_direction_set )
                {
                    char* psz_text = NULL;
                    if( asprintf( &psz_text, "%s%s%s", p_bidi[i_direction].psz_uni_start, p_segment->psz_text, p_bidi[i_direction].psz_uni_end ) < 0 )
677 678
                    {
                        text_segment_Delete( p_segment );
679
                        goto fail;
680
                    }
681 682 683 684

                    free( p_segment->psz_text );
                    p_segment->psz_text = psz_text;
                }
685 686
            }
            if( p_first_segment == NULL )
Sushma Reddy's avatar
Sushma Reddy committed
687 688 689 690
            {
                p_first_segment = p_segment;
                p_current_segment = p_segment;
            }
691
            else if( p_current_segment->psz_text != NULL )
Sushma Reddy's avatar
Sushma Reddy committed
692 693 694 695
            {
                p_current_segment->p_next = p_segment;
                p_current_segment = p_segment;
            }
696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712
            else
            {
                /*
                * If p_first_segment isn't NULL but p_current_segment->psz_text is NULL
                * this means that something went wrong in the decoding of the
                * first segment text:
                *
                * Indeed, to allocate p_first_segment ( aka non NULL ), we must have
                * - i_type == XML_READER_TEXT
                * - passed the allocation of p_segment->psz_text without any error
                *
                * This would mean that vlc_xml_decode failed and p_first_segment->psz_text
                * is NULL.
                */
                text_segment_Delete( p_segment );
                goto fail;
            }
Sushma Reddy's avatar
Sushma Reddy committed
713
        }
714
        else if( i_type == XML_READER_ENDELEM && !tagnamecmp( node, "span" ) )
Sushma Reddy's avatar
Sushma Reddy committed
715
        {
716 717
            if( p_style_stack->p_next )
                PopStyle( &p_style_stack);
Sushma Reddy's avatar
Sushma Reddy committed
718
        }
719
        else if( i_type == XML_READER_ENDELEM && !tagnamecmp( node, "p" ) )
Sushma Reddy's avatar
Sushma Reddy committed
720
        {
721 722
            PopStyle( &p_style_stack );
            p_current_segment->p_next = NULL;
Sushma Reddy's avatar
Sushma Reddy committed
723
        }
724
        else if( i_type == XML_READER_STARTELEM && !strcasecmp( node, "br" ) )
Sushma Reddy's avatar
Sushma Reddy committed
725
        {
726
            if( p_current_segment != NULL && p_current_segment->psz_text != NULL )
Sushma Reddy's avatar
Sushma Reddy committed
727 728
            {
                char* psz_text = NULL;
729
                if( asprintf( &psz_text, "%s\n", p_current_segment->psz_text ) != -1 )
Sushma Reddy's avatar
Sushma Reddy committed
730 731 732 733 734 735 736 737 738 739
                {
                    free( p_current_segment->psz_text );
                    p_current_segment->psz_text = psz_text;
                }
            }
        }
        i_type = xml_ReaderNextNode( p_xml_reader, &node );
    }
    ClearStack( p_style_stack );
    xml_ReaderDelete( p_xml_reader );
740
    vlc_stream_Delete( p_sub );
Sushma Reddy's avatar
Sushma Reddy committed
741 742 743 744 745 746 747

    return p_first_segment;

fail:
    text_segment_ChainDelete( p_first_segment );
    ClearStack( p_style_stack );
    xml_ReaderDelete( p_xml_reader );
748
    vlc_stream_Delete( p_sub );
Sushma Reddy's avatar
Sushma Reddy committed
749 750 751
    return NULL;
}

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
752 753 754 755 756 757
static subpicture_t *ParseText( decoder_t *p_dec, block_t *p_block )
{
    decoder_sys_t *p_sys = p_dec->p_sys;
    subpicture_t *p_spu = NULL;
    char *psz_subtitle = NULL;

758 759 760
    if( p_block->i_flags & BLOCK_FLAG_CORRUPTED )
        return NULL;

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778
    /* We cannot display a subpicture with no date */
    if( p_block->i_pts <= VLC_TS_INVALID )
    {
        msg_Warn( p_dec, "subtitle without a date" );
        return NULL;
    }

    /* Check validity of packet data */
    /* An "empty" line containing only \0 can be used to force
       and ephemer picture from the screen */

    if( p_block->i_buffer < 1 )
    {
        msg_Warn( p_dec, "no subtitle data" );
        return NULL;
    }

    psz_subtitle = malloc( p_block->i_buffer );
779
    if( unlikely( psz_subtitle == NULL ) )
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
780 781 782 783 784 785 786 787 788 789 790 791 792 793
        return NULL;
    memcpy( psz_subtitle, p_block->p_buffer, p_block->i_buffer );

    /* Create the subpicture unit */
    p_spu = decoder_NewSubpictureText( p_dec );
    if( !p_spu )
    {
        free( psz_subtitle );
        return NULL;
    }
    p_spu->i_start    = p_block->i_pts;
    p_spu->i_stop     = p_block->i_pts + p_block->i_length;
    p_spu->b_ephemer  = (p_block->i_length == 0);
    p_spu->b_absolute = false;
Sushma Reddy's avatar
Sushma Reddy committed
794

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
795 796 797
    subpicture_updater_sys_t *p_spu_sys = p_spu->updater.p_sys;

    p_spu_sys->align = SUBPICTURE_ALIGN_BOTTOM | p_sys->i_align;
Sushma Reddy's avatar
Sushma Reddy committed
798
    p_spu_sys->p_segments = ParseTTMLSubtitles( p_dec, p_spu_sys, psz_subtitle );
799
    free( psz_subtitle );
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
800 801 802 803

    return p_spu;
}

Sushma Reddy's avatar
Sushma Reddy committed
804 805


Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830
/****************************************************************************
 * DecodeBlock: the whole thing
 ****************************************************************************/
static subpicture_t *DecodeBlock( decoder_t *p_dec, block_t **pp_block )
{
    if( !pp_block || *pp_block == NULL )
        return NULL;

    block_t* p_block = *pp_block;
    subpicture_t *p_spu = ParseText( p_dec, p_block );

    block_Release( p_block );
    *pp_block = NULL;

    return p_spu;
}

/*****************************************************************************
 * OpenDecoder: probe the decoder and return score
 *****************************************************************************/
static int OpenDecoder( vlc_object_t *p_this )
{
    decoder_t *p_dec = (decoder_t*)p_this;
    decoder_sys_t *p_sys;

831
    if( p_dec->fmt_in.i_codec != VLC_CODEC_TTML )
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
832 833 834 835 836 837
        return VLC_EGENERIC;

    /* Allocate the memory needed to store the decoder's structure */
    p_dec->p_sys = p_sys = calloc( 1, sizeof( *p_sys ) );
    if( unlikely( p_sys == NULL ) )
        return VLC_ENOMEM;
Sushma Reddy's avatar
Sushma Reddy committed
838

839
    if( p_dec->fmt_in.p_extra != NULL && p_dec->fmt_in.i_extra > 0 )
Sushma Reddy's avatar
Sushma Reddy committed
840
        ParseTTMLStyles( p_dec );
Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
841 842 843 844

    p_dec->pf_decode_sub = DecodeBlock;
    p_dec->fmt_out.i_cat = SPU_ES;
    p_sys->i_align = var_InheritInteger( p_dec, "ttml-align" );
Sushma Reddy's avatar
Sushma Reddy committed
845

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
846 847 848 849 850 851 852 853 854 855 856
    return VLC_SUCCESS;
}

/*****************************************************************************
 * CloseDecoder: clean up the decoder
 *****************************************************************************/
static void CloseDecoder( vlc_object_t *p_this )
{
    decoder_t *p_dec = (decoder_t *)p_this;
    decoder_sys_t *p_sys = p_dec->p_sys;

857
    for( size_t i = 0; i < p_sys->i_styles; ++i )
Sushma Reddy's avatar
Sushma Reddy committed
858 859 860 861 862 863 864
    {
        free( p_sys->pp_styles[i]->psz_styleid );
        text_style_Delete( p_sys->pp_styles[i]->font_style );
        free( p_sys->pp_styles[i] );
    }
    TAB_CLEAN( p_sys->i_styles, p_sys->pp_styles );

Hugo Beauzée-Luyssen's avatar
Hugo Beauzée-Luyssen committed
865 866
    free( p_sys );
}