diff --git a/.editorconfig b/.editorconfig index 8fa69a5797..0ca45473ef 100644 --- a/.editorconfig +++ b/.editorconfig @@ -6,12 +6,6 @@ indent_size = 4 trim_trailing_whitespace = true insert_final_newline = true -[src/MatroskaParser.{c,h}] -indent_style = tab -indent_size = 2 -tab_width = 8 -trim_trailing_whitespace = false - [src/libresrc/**default_*.json] indent_style = tab indent_size = 2 diff --git a/.gitattributes b/.gitattributes index b60332b966..195da2c00f 100644 --- a/.gitattributes +++ b/.gitattributes @@ -6,8 +6,10 @@ subprojects/csri/meson.build -linguist-vendored subprojects/iconv/meson.build -linguist-vendored subprojects/luabins/meson.build -linguist-vendored -src/MatroskaParser.* linguist-vendored - libaegisub/lua/modules/lpeg.* linguist-vendored automation/include/moonscript.lua linguist-vendored + +# Compared byte-for-byte by the Matroska tests +tests/matroska/fixtures/*.behavior eol=lf +tests/matroska/fixtures/*.mk[av] binary diff --git a/LICENCE b/LICENCE index 64032a465f..6e4e204ec1 100644 --- a/LICENCE +++ b/LICENCE @@ -36,9 +36,6 @@ follows: src/gl/ - MIT license. See src/gl/glext.h -src/MatroskaParser.(c|h) - - Licensed to BSDL with permission from the author. - universalchardet/ - MPL 1.1 diff --git a/meson.build b/meson.build index 29b7fd390c..0edf405cad 100644 --- a/meson.build +++ b/meson.build @@ -179,6 +179,9 @@ endif zlib_dep = dependency('zlib') deps += zlib_dep +# The bundled libmatroska needs libebml >= 1.4.4 +libebml = dependency('libebml', version: '>=1.4.4', fallback: ['libebml', 'libebml_dep']) +libmatroska = dependency('libmatroska', version: '>=1.7.1', fallback: ['libmatroska', 'libmatroska_dep']) use_bundled_wx = force_all_fallbacks or force_fallback_for.contains('wxWidgets') if not use_bundled_wx diff --git a/src/MatroskaParser.c b/src/MatroskaParser.c deleted file mode 100644 index 31d5482054..0000000000 --- a/src/MatroskaParser.c +++ /dev/null @@ -1,3350 +0,0 @@ -/* - * Copyright (c) 2004-2009 Mike Matsnev. All Rights Reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions - * are met: - * - * 1. Redistributions of source code must retain the above copyright - * notice immediately at the beginning of the file, without modification, - * this list of conditions, and the following disclaimer. - * 2. Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * 3. Absolutely no warranty of function or purpose is made by the author - * Mike Matsnev. - * - * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR - * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES - * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. - * IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, INDIRECT, - * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT - * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, - * DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY - * THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF - * THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. -// -// Aegisub Project http://www.aegisub.org/ - -/// @file MatroskaParser.c -/// @brief Haali's low-level Matroska-parsing library -/// @ingroup video_input -/// - - */ - -#include -#include -#include -#include -#include - -#ifdef _WIN32 -#define inline __inline - -#include -#endif - -#ifndef EVCBUG -#define EVCBUG -#endif - -#include "MatroskaParser.h" - -#ifdef HAVE_ALLOCA_H -#include -#elif defined(HAVE_MALLOC_H) -#include -#endif /* HAVE_ALLOCA_H */ - -#ifdef HAVE_UNDERLINE_ALLOCA -#define alloca _alloca -#endif - -#ifdef MATROSKA_COMPRESSION_SUPPORT -#include -#endif - -#define EBML_VERSION 1 -#define EBML_MAX_ID_LENGTH 4 -#define EBML_MAX_SIZE_LENGTH 8 -#define MATROSKA_VERSION 2 -#define MATROSKA_DOCTYPE "matroska" - -#define MAX_STRING_LEN 1023 -#define QSEGSIZE 512 -#define MAX_TRACKS 32 -#define MAX_READAHEAD (256*1024) - -#define MAXCLUSTER (64*1048576) -#define MAXFRAME (4*1048576) - -#ifdef WIN32 -#define LL(x) x##i64 -#define ULL(x) x##ui64 -#else -#define LL(x) x##ll -#define ULL(x) x##ull -#endif - -#define MAXU64 ULL(0xffffffffffffffff) -#define ONE ULL(1) - -// compatibility -static char *mystrdup(struct InputStream *is,const char *src) { - size_t len; - char *dst; - - if (src==NULL) - return NULL; - - len = strlen(src); - dst = is->memalloc(is,len+1); - if (dst==NULL) - return NULL; - - memcpy(dst,src,len+1); - - return dst; -} - -static void mystrlcpy(char *dst,const char *src,unsigned size) { - unsigned i; - - for (i=0;i+1 0) - *dest = '\0'; - return; - } - - while (*fmt && dest < de) - switch (state) { - case 0: - if (*fmt == '%') { - ++fmt; - state = 1; - width = zero = neg = ll = 0; - } else - *dest++ = *fmt++; - break; - case 1: - if (*fmt == '-') { - neg = 1; - ++fmt; - state = 2; - break; - } - if (*fmt == '0') - zero = 1; - state = 2; // fallthrough - case 2: - if (*fmt >= '0' && *fmt <= '9') { - width = width * 10 + *fmt++ - '0'; - break; - } - state = 3; // fallthrough - case 3: - if (*fmt == 'l') { - ++ll; - ++fmt; - break; - } - state = 4; // fallthrough - case 4: - switch (*fmt) { - case 's': - myvsnprintf_string(&dest,de,va_arg(ap,const char *)); - break; - case 'd': - switch (ll) { - case 0: - myvsnprintf_int(&dest,de,width,zero,neg,10,'a',va_arg(ap,int)); - break; - case 1: - myvsnprintf_int(&dest,de,width,zero,neg,10,'a',va_arg(ap,long)); - break; - case 2: - myvsnprintf_int(&dest,de,width,zero,neg,10,'a',va_arg(ap,int64_t)); - break; - } - break; - case 'u': - switch (ll) { - case 0: - myvsnprintf_uint(&dest,de,width,zero,neg,10,'a',va_arg(ap,unsigned int)); - break; - case 1: - myvsnprintf_uint(&dest,de,width,zero,neg,10,'a',va_arg(ap,unsigned long)); - break; - case 2: - myvsnprintf_uint(&dest,de,width,zero,neg,10,'a',va_arg(ap,uint64_t)); - break; - } - break; - case 'x': - switch (ll) { - case 0: - myvsnprintf_uint(&dest,de,width,zero,neg,16,'a',va_arg(ap,unsigned int)); - break; - case 1: - myvsnprintf_uint(&dest,de,width,zero,neg,16,'a',va_arg(ap,unsigned long)); - break; - case 2: - myvsnprintf_uint(&dest,de,width,zero,neg,16,'a',va_arg(ap,uint64_t)); - break; - } - break; - case 'X': - switch (ll) { - case 0: - myvsnprintf_uint(&dest,de,width,zero,neg,16,'A',va_arg(ap,unsigned int)); - break; - case 1: - myvsnprintf_uint(&dest,de,width,zero,neg,16,'A',va_arg(ap,unsigned long)); - break; - case 2: - myvsnprintf_uint(&dest,de,width,zero,neg,16,'A',va_arg(ap,uint64_t)); - break; - } - break; - default: - break; - } - ++fmt; - state = 0; - break; - default: - state = 0; - break; - } - *dest = '\0'; -} - -static void errorjmp(MatroskaFile *mf,const char *fmt, ...) { - va_list ap; - - mf->cache->memfree(mf->cache, mf->cpbuf); - mf->cpbuf = NULL; - - va_start(ap, fmt); - myvsnprintf(mf->errmsg,sizeof(mf->errmsg),fmt,ap); - va_end(ap); - - mf->flags |= MPF_ERROR; - - longjmp(mf->jb,1); -} - -/////////////////////////////////////////////////////////////////////////// -// arrays -static void *ArrayAlloc(MatroskaFile *mf,void **base, - unsigned *cur,unsigned *max,unsigned elem_size) -{ - if (*cur>=*max) { - void *np; - unsigned newsize = *max * 2; - if (newsize==0) - newsize = 1; - - np = mf->cache->memrealloc(mf->cache,*base,newsize*elem_size); - if (np==NULL) - errorjmp(mf,"Out of memory in ArrayAlloc"); - - *base = np; - *max = newsize; - } - - return (char*)*base + elem_size * (*cur)++; -} - -static void ArrayReleaseMemory(MatroskaFile *mf,void **base, - unsigned cur,unsigned *max,unsigned elem_size) -{ - if (cur<*max) { - void *np = mf->cache->memrealloc(mf->cache,*base,cur*elem_size); - *base = np; - *max = cur; - } -} - - -#define ASGET(f,s,name) ArrayAlloc((f),(void**)&(s)->name,&(s)->n##name,&(s)->n##name##Size,sizeof(*((s)->name))) -#define AGET(f,name) ArrayAlloc((f),(void**)&(f)->name,&(f)->n##name,&(f)->n##name##Size,sizeof(*((f)->name))) -#define ARELEASE(f,s,name) ArrayReleaseMemory((f),(void**)&(s)->name,(s)->n##name,&(s)->n##name##Size,sizeof(*((s)->name))) - -/////////////////////////////////////////////////////////////////////////// -// queues -static struct QueueEntry *QPut(struct Queue *q,struct QueueEntry *qe) { - if (q->tail) - q->tail->next = qe; - qe->next = NULL; - q->tail = qe; - if (q->head==NULL) - q->head = qe; - - return qe; -} - -static struct QueueEntry *QGet(struct Queue *q) { - struct QueueEntry *qe = q->head; - if (qe == NULL) - return NULL; - q->head = qe->next; - if (q->tail == qe) - q->tail = NULL; - return qe; -} - -static struct QueueEntry *QAlloc(MatroskaFile *mf) { - struct QueueEntry *qe,**qep; - if (mf->QFreeList == NULL) { - unsigned i; - - qep = AGET(mf,QBlocks); - - *qep = mf->cache->memalloc(mf->cache,QSEGSIZE * sizeof(*qe)); - if (*qep == NULL) - errorjmp(mf,"Ouf of memory"); - - qe = *qep; - - for (i=0;iQFreeList = qe; - } - - qe = mf->QFreeList; - mf->QFreeList = qe->next; - - return qe; -} - -static inline void QFree(MatroskaFile *mf,struct QueueEntry *qe) { - qe->next = mf->QFreeList; - mf->QFreeList = qe; -} - -// fill the buffer at current position -static void fillbuf(MatroskaFile *mf) { - int rd; - - // advance buffer pointers - mf->bufbase += mf->buflen; - mf->buflen = mf->bufpos = 0; - - // get the relevant page - rd = mf->cache->read(mf->cache, mf->bufbase, mf->inbuf, IBSZ); - if (rd<0) - errorjmp(mf,"I/O Error: %s",mf->cache->geterror(mf->cache)); - - mf->buflen = rd; -} - -// fill the buffer and return next char -static int nextbuf(MatroskaFile *mf) { - fillbuf(mf); - - if (mf->bufpos < mf->buflen) - return (unsigned char)(mf->inbuf[mf->bufpos++]); - - return EOF; -} - -static inline int readch(MatroskaFile *mf) { - return mf->bufpos < mf->buflen ? (unsigned char)(mf->inbuf[mf->bufpos++]) : nextbuf(mf); -} - -static inline uint64_t filepos(MatroskaFile *mf) { - return mf->bufbase + mf->bufpos; -} - -static void readbytes(MatroskaFile *mf,void *buffer,uint64_t len) { - char *cp = buffer; - - if (mf->buflen < mf->bufpos) - errorjmp(mf,"Unreachable: buffer position larger than buffer length : %d > %d",mf->buflen,mf->bufpos); - - uint64_t nb = mf->buflen - mf->bufpos; - - if (nb > len) - nb = len; - - memcpy(cp, mf->inbuf + mf->bufpos, nb); - mf->bufpos += nb; - len -= nb; - cp += nb; - - if (len>0) { - mf->bufbase += mf->buflen; - mf->bufpos = mf->buflen = 0; - - nb = mf->cache->read(mf->cache, mf->bufbase, cp, len); - if (nb<0) - errorjmp(mf,"I/O Error: %s",mf->cache->geterror(mf->cache)); - if (nb != len) - errorjmp(mf,"Short read: got %d bytes of %d",nb,len); - mf->bufbase += len; - } -} - -static void skipbytes(MatroskaFile *mf,uint64_t len) { - if (mf->buflen < mf->bufpos) - errorjmp(mf,"Unreachable: buffer position larger than buffer length : %d > %d",mf->buflen,mf->bufpos); - - uint64_t nb = mf->buflen - mf->bufpos; - - if (nb > len) - nb = len; - - mf->bufpos += nb; - len -= nb; - - if (len>0) { - mf->bufbase += mf->buflen; - mf->bufpos = mf->buflen = 0; - - mf->bufbase += len; - } -} - -static void seek(MatroskaFile *mf,uint64_t pos) { - // see if pos is inside buffer - if (pos>=mf->bufbase && posbufbase+mf->buflen) - mf->bufpos = (unsigned)(pos - mf->bufbase); - else { - // invalidate buffer and set pointer - mf->bufbase = pos; - mf->buflen = mf->bufpos = 0; - } -} - -/////////////////////////////////////////////////////////////////////////// -// floating point -static inline MKFLOAT mkfi(int i) { -#ifdef MATROSKA_INTEGER_ONLY - MKFLOAT f; - f.v = (int64_t)i << 32; - return f; -#else - return i; -#endif -} - -static inline int64_t mul3(MKFLOAT scale,int64_t tc) { -#ifdef MATROSKA_INTEGER_ONLY - // x1 x0 - // y1 y0 - // -------------- - // x0*y0 - // x1*y0 - // x0*y1 - // x1*y1 - // -------------- - // .. r1 r0 .. - // - // r = ((x0*y0) >> 32) + (x1*y0) + (x0*y1) + ((x1*y1) << 32) - unsigned x0,x1,y0,y1; - uint64_t p; - char sign = 0; - - if (scale.v < 0) - sign = !sign, scale.v = -scale.v; - if (tc < 0) - sign = !sign, tc = -tc; - - x0 = (unsigned)scale.v; - x1 = (unsigned)((uint64_t)scale.v >> 32); - y0 = (unsigned)tc; - y1 = (unsigned)((uint64_t)tc >> 32); - - p = (uint64_t)x0*y0 >> 32; - p += (uint64_t)x0*y1; - p += (uint64_t)x1*y0; - p += (uint64_t)(x1*y1) << 32; - - return p; -#else - return (int64_t)(scale * tc); -#endif -} - -/////////////////////////////////////////////////////////////////////////// -// EBML support -static int readID(MatroskaFile *mf) { - int c1,c2,c3,c4; - - c1 = readch(mf); - if (c1 == EOF) - return EOF; - - if (c1 & 0x80) - return c1; - - if ((c1 & 0xf0) == 0) - errorjmp(mf,"Invalid first byte of EBML ID: %02X",c1); - - c2 = readch(mf); - if (c2 == EOF) -fail: - errorjmp(mf,"Got EOF while reading EBML ID"); - - if ((c1 & 0xc0) == 0x40) - return (c1<<8) | c2; - - c3 = readch(mf); - if (c3 == EOF) - goto fail; - - if ((c1 & 0xe0) == 0x20) - return (c1<<16) | (c2<<8) | c3; - - c4 = readch(mf); - if (c4 == EOF) - goto fail; - - if ((c1 & 0xf0) == 0x10) - return (c1<<24) | (c2<<16) | (c3<<8) | c4; - - return 0; // NOT REACHED -} - -static uint64_t readVLUIntImp(MatroskaFile *mf,int *mask) { - int c,d,m; - uint64_t v = 0; - - c = readch(mf); - if (c == EOF) - return 0; // XXX should errorjmp()? - - if (c == 0) - errorjmp(mf,"Invalid first byte of EBML integer: 0"); - - for (m=0;;++m) { - if (c & (0x80 >> m)) { - c &= 0x7f >> m; - if (mask) - *mask = m; - return v | ((uint64_t)c << m*8); - } - d = readch(mf); - if (d == EOF) - errorjmp(mf,"Got EOF while reading EBML unsigned integer"); - v = (v<<8) | d; - } - // NOT REACHED -} - -static inline uint64_t readVLUInt(MatroskaFile *mf) { - return readVLUIntImp(mf,NULL); -} - -static uint64_t readSize(MatroskaFile *mf) { - int m; - uint64_t v = readVLUIntImp(mf,&m); - - // see if it's unspecified - if (v == (MAXU64 >> (57-m*7))) - errorjmp(mf,"Unspecified element size is not supported here."); - - return v; -} - -static inline int64_t readVLSInt(MatroskaFile *mf) { - static int64_t bias[8] = { (ONE<<6)-1, (ONE<<13)-1, (ONE<<20)-1, (ONE<<27)-1, - (ONE<<34)-1, (ONE<<41)-1, (ONE<<48)-1, (ONE<<55)-1 }; - - int m; - int64_t v = readVLUIntImp(mf,&m); - - return v - bias[m]; -} - -static uint64_t readUInt(MatroskaFile *mf,unsigned int len) { - int c; - unsigned int m = len; - uint64_t v = 0; - - if (len==0) - return v; - if (len>8) - errorjmp(mf,"Unsupported integer size in readUInt: %u",len); - - do { - c = readch(mf); - if (c == EOF) - errorjmp(mf,"Got EOF while reading EBML unsigned integer"); - v = (v<<8) | c; - } while (--m); - - return v; -} - -static inline int64_t readSInt(MatroskaFile *mf,unsigned int len) { - int64_t v = readUInt(mf,(unsigned)len); - int s = 64 - (len<<3); - return (v << s) >> s; -} - -static MKFLOAT readFloat(MatroskaFile *mf,unsigned int len) { -#ifdef MATROSKA_INTEGER_ONLY - MKFLOAT f; - int shift; -#else - union { - unsigned int ui; - uint64_t ull; - float f; - double d; - } u; -#endif - - if (len!=4 && len!=8) - errorjmp(mf,"Invalid float size in readFloat: %u",len); - -#ifdef MATROSKA_INTEGER_ONLY - if (len == 4) { - unsigned ui = (unsigned)readUInt(mf,(unsigned)len); - f.v = (ui & 0x7fffff) | 0x800000; - if (ui & 0x80000000) - f.v = -f.v; - shift = (ui >> 23) & 0xff; - if (shift == 0) // assume 0 -zero: - shift = 0, f.v = 0; - else if (shift == 255) -inf: - if (ui & 0x80000000) - f.v = LL(0x8000000000000000); - else - f.v = LL(0x7fffffffffffffff); - else { - shift += -127 + 9; - if (shift > 39) - goto inf; -shift: - if (shift < 0) - f.v = f.v >> -shift; - else if (shift > 0) - f.v = f.v << shift; - } - } else if (len == 8) { - uint64_t ui = readUInt(mf,(unsigned)len); - f.v = (ui & LL(0xfffffffffffff)) | LL(0x10000000000000); - if (ui & 0x80000000) - f.v = -f.v; - shift = (int)((ui >> 52) & 0x7ff); - if (shift == 0) // assume 0 - goto zero; - else if (shift == 2047) - goto inf; - else { - shift += -1023 - 20; - if (shift > 10) - goto inf; - goto shift; - } - } - - return f; -#else - if (len==4) { - u.ui = (unsigned int)readUInt(mf,(unsigned)len); - return u.f; - } - - if (len==8) { - u.ull = readUInt(mf,(unsigned)len); - return u.d; - } - - return 0; -#endif -} - -static void readString(MatroskaFile *mf,uint64_t len,char *buffer,int buflen) { - unsigned int nread; - - if (buflen<1) - errorjmp(mf,"Invalid buffer size in readString: %d",buflen); - - nread = buflen - 1; - - if (nread > len) - nread = len; - - readbytes(mf,buffer,nread); - len -= nread; - - if (len>0) - skipbytes(mf,len); - - buffer[nread] = '\0'; -} - -static void readLangCC(MatroskaFile *mf, uint64_t len, char lcc[4]) { - uint64_t todo = len > 3 ? 3 : len; - - lcc[0] = lcc[1] = lcc[2] = lcc[3] = 0; - readbytes(mf, lcc, todo); - skipbytes(mf, len - todo); -} - -/////////////////////////////////////////////////////////////////////////// -// file parser -#define FOREACH(f,tl) \ - { \ - uint64_t tmplen = (tl); \ - { \ - uint64_t start = filepos(f); \ - uint64_t cur,len; \ - int id; \ - for (;;) { \ - cur = filepos(mf); \ - if (cur == start + tmplen) \ - break; \ - id = readID(f); \ - if (id==EOF) \ - errorjmp(mf,"Unexpected EOF while reading EBML container"); \ - len = readSize(mf); \ - switch (id) { - -#define ENDFOR1(f) \ - default: \ - skipbytes(f,len); \ - break; \ - } -#define ENDFOR2() \ - } \ - } \ - } - -#define ENDFOR(f) ENDFOR1(f) ENDFOR2() - -#define myalloca(f,c) alloca(c) -#define STRGETF(f,v,len,func) \ - { \ - char *TmpVal; \ - unsigned TmpLen = (len)>MAX_STRING_LEN ? MAX_STRING_LEN : (unsigned)(len); \ - TmpVal = func(f->cache,TmpLen+1); \ - if (TmpVal == NULL) \ - errorjmp(mf,"Out of memory"); \ - readString(f,len,TmpVal,TmpLen+1); \ - (v) = TmpVal; \ - } - -#define STRGETA(f,v,len) STRGETF(f,v,len,myalloca) -#define STRGETM(f,v,len) STRGETF(f,v,len,f->cache->memalloc) - -/* Removed because it's not used. -static int IsWritingApp(MatroskaFile *mf,const char *str) { - const char *cp = mf->Seg.WritingApp; - if (!cp) - return 0; - - while (*str && *str++==*cp++) ; - - return !*str; -} -*/ -static void parseEBML(MatroskaFile *mf,uint64_t toplen) { - uint64_t v; - char buf[32]; - - FOREACH(mf,toplen) - case 0x4286: // Version - v = readUInt(mf,(unsigned)len); - break; - case 0x42f7: // ReadVersion - v = readUInt(mf,(unsigned)len); - if (v > EBML_VERSION) - errorjmp(mf,"File requires version %d EBML parser",(int)v); - break; - case 0x42f2: // MaxIDLength - v = readUInt(mf,(unsigned)len); - if (v > EBML_MAX_ID_LENGTH) - errorjmp(mf,"File has identifiers longer than %d",(int)v); - break; - case 0x42f3: // MaxSizeLength - v = readUInt(mf,(unsigned)len); - if (v > EBML_MAX_SIZE_LENGTH) - errorjmp(mf,"File has integers longer than %d",(int)v); - break; - case 0x4282: // DocType - readString(mf,len,buf,sizeof(buf)); - if (strcmp(buf,MATROSKA_DOCTYPE)) - errorjmp(mf,"Unsupported DocType: %s",buf); - break; - case 0x4287: // DocTypeVersion - v = readUInt(mf,(unsigned)len); - break; - case 0x4285: // DocTypeReadVersion - v = readUInt(mf,(unsigned)len); - if (v > MATROSKA_VERSION) - errorjmp(mf,"File requires version %d Matroska parser",(int)v); - break; - ENDFOR(mf); -} - -static void parseSeekEntry(MatroskaFile *mf,uint64_t toplen) { - int seekid = 0; - uint64_t pos = (uint64_t)-1; - - FOREACH(mf,toplen) - case 0x53ab: // SeekID - if (len>EBML_MAX_ID_LENGTH) - errorjmp(mf,"Invalid ID size in parseSeekEntry: %d\n",(int)len); - seekid = (int)readUInt(mf,(unsigned)len); - break; - case 0x53ac: // SeekPos - pos = readUInt(mf,(unsigned)len); - break; - ENDFOR(mf); - - if (pos == (uint64_t)-1) - errorjmp(mf,"Invalid element position in parseSeekEntry"); - - pos += mf->pSegment; - switch (seekid) { - case 0x114d9b74: // next SeekHead - if (mf->pSeekHead) - errorjmp(mf,"SeekHead contains more than one SeekHead pointer"); - mf->pSeekHead = pos; - break; - case 0x1549a966: // SegmentInfo - mf->pSegmentInfo = pos; - break; - case 0x1f43b675: // Cluster - if (!mf->pCluster) - mf->pCluster = pos; - break; - case 0x1654ae6b: // Tracks - mf->pTracks = pos; - break; - case 0x1c53bb6b: // Cues - mf->pCues = pos; - break; - case 0x1941a469: // Attachments - mf->pAttachments = pos; - break; - case 0x1043a770: // Chapters - mf->pChapters = pos; - break; - case 0x1254c367: // tags - mf->pTags = pos; - break; - } -} - -static void parseSeekHead(MatroskaFile *mf,uint64_t toplen) { - FOREACH(mf,toplen) - case 0x4dbb: - parseSeekEntry(mf,len); - break; - ENDFOR(mf); -} - -static void parseSegmentInfo(MatroskaFile *mf,uint64_t toplen) { - MKFLOAT duration = mkfi(0); - - if (mf->seen.SegmentInfo) { - skipbytes(mf,toplen); - return; - } - - mf->seen.SegmentInfo = 1; - mf->Seg.TimecodeScale = 1000000; // Default value - - FOREACH(mf,toplen) - case 0x73a4: // SegmentUID - if (len!=sizeof(mf->Seg.UID)) - errorjmp(mf,"SegmentUID size is not %d bytes",mf->Seg.UID); - readbytes(mf,mf->Seg.UID,sizeof(mf->Seg.UID)); - break; - case 0x7384: // SegmentFilename - STRGETM(mf,mf->Seg.Filename,len); - break; - case 0x3cb923: // PrevUID - if (len!=sizeof(mf->Seg.PrevUID)) - errorjmp(mf,"PrevUID size is not %d bytes",mf->Seg.PrevUID); - readbytes(mf,mf->Seg.PrevUID,sizeof(mf->Seg.PrevUID)); - break; - case 0x3c83ab: // PrevFilename - STRGETM(mf,mf->Seg.PrevFilename,len); - break; - case 0x3eb923: // NextUID - if (len!=sizeof(mf->Seg.NextUID)) - errorjmp(mf,"NextUID size is not %d bytes",mf->Seg.NextUID); - readbytes(mf,mf->Seg.NextUID,sizeof(mf->Seg.NextUID)); - break; - case 0x3e83bb: // NextFilename - STRGETM(mf,mf->Seg.NextFilename,len); - break; - case 0x2ad7b1: // TimecodeScale - mf->Seg.TimecodeScale = readUInt(mf,(unsigned)len); - if (mf->Seg.TimecodeScale == 0) - errorjmp(mf,"Segment timecode scale is zero"); - break; - case 0x4489: // Duration - duration = readFloat(mf,(unsigned)len); - break; - case 0x4461: // DateUTC - mf->Seg.DateUTC = readUInt(mf,(unsigned)len); - mf->Seg.DateUTCValid = 1; - break; - case 0x7ba9: // Title - STRGETM(mf,mf->Seg.Title,len); - break; - case 0x4d80: // MuxingApp - STRGETM(mf,mf->Seg.MuxingApp,len); - break; - case 0x5741: // WritingApp - STRGETM(mf,mf->Seg.WritingApp,len); - break; - ENDFOR(mf); - - mf->Seg.Duration = mul3(duration,mf->Seg.TimecodeScale); -} - -static void parseFirstCluster(MatroskaFile *mf,uint64_t toplen) { - uint64_t end = filepos(mf) + toplen; - - mf->seen.Cluster = 1; - mf->firstTimecode = 0; - - FOREACH(mf,toplen) - case 0xe7: // Timecode - mf->firstTimecode += readUInt(mf,(unsigned)len); - break; - case 0xa3: // BlockEx - readVLUInt(mf); // track number - mf->firstTimecode += readSInt(mf, 2); - - skipbytes(mf,end - filepos(mf)); - return; - case 0xa0: // BlockGroup - FOREACH(mf,len) - case 0xa1: // Block - readVLUInt(mf); // track number - mf->firstTimecode += readSInt(mf,2); - - skipbytes(mf,end - filepos(mf)); - return; - ENDFOR(mf); - break; - ENDFOR(mf); -} - -static void parseVideoInfo(MatroskaFile *mf,uint64_t toplen,struct TrackInfo *ti) { - uint64_t v; - char dW = 0, dH = 0; - - FOREACH(mf,toplen) - case 0x9a: // FlagInterlaced - ti->AV.Video.Interlaced = readUInt(mf,(unsigned)len)!=0; - break; - case 0x53b8: // StereoMode - v = readUInt(mf,(unsigned)len); - if (v>3) - errorjmp(mf,"Invalid stereo mode"); - ti->AV.Video.StereoMode = (unsigned char)v; - break; - case 0xb0: // PixelWidth - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"PixelWidth is too large"); - ti->AV.Video.PixelWidth = (unsigned)v; - if (!dW) - ti->AV.Video.DisplayWidth = ti->AV.Video.PixelWidth; - break; - case 0xba: // PixelHeight - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"PixelHeight is too large"); - ti->AV.Video.PixelHeight = (unsigned)v; - if (!dH) - ti->AV.Video.DisplayHeight = ti->AV.Video.PixelHeight; - break; - case 0x54b0: // DisplayWidth - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"DisplayWidth is too large"); - ti->AV.Video.DisplayWidth = (unsigned)v; - dW = 1; - break; - case 0x54ba: // DisplayHeight - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"DisplayHeight is too large"); - ti->AV.Video.DisplayHeight = (unsigned)v; - dH = 1; - break; - case 0x54b2: // DisplayUnit - v = readUInt(mf,(unsigned)len); - if (v>4) - errorjmp(mf,"Invalid DisplayUnit: %d",(int)v); - ti->AV.Video.DisplayUnit = (unsigned char)v; - break; - case 0x54b3: // AspectRatioType - v = readUInt(mf,(unsigned)len); - if (v>2) - errorjmp(mf,"Invalid AspectRatioType: %d",(int)v); - ti->AV.Video.AspectRatioType = (unsigned char)v; - break; - case 0x54aa: // PixelCropBottom - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"PixelCropBottom is too large"); - ti->AV.Video.CropB = (unsigned)v; - break; - case 0x54bb: // PixelCropTop - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"PixelCropTop is too large"); - ti->AV.Video.CropT = (unsigned)v; - break; - case 0x54cc: // PixelCropLeft - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"PixelCropLeft is too large"); - ti->AV.Video.CropL = (unsigned)v; - break; - case 0x54dd: // PixelCropRight - v = readUInt(mf,(unsigned)len); - if (v>0xffffffff) - errorjmp(mf,"PixelCropRight is too large"); - ti->AV.Video.CropR = (unsigned)v; - break; - case 0x2eb524: // ColourSpace - ti->AV.Video.ColourSpace = (unsigned)readUInt(mf,4); - break; - case 0x2fb523: // GammaValue - ti->AV.Video.GammaValue = readFloat(mf,(unsigned)len); - break; - ENDFOR(mf); -} - -static void parseAudioInfo(MatroskaFile *mf,uint64_t toplen,struct TrackInfo *ti) { - uint64_t v; - - FOREACH(mf,toplen) - case 0xb5: // SamplingFrequency - ti->AV.Audio.SamplingFreq = readFloat(mf,(unsigned)len); - break; - case 0x78b5: // OutputSamplingFrequency - ti->AV.Audio.OutputSamplingFreq = readFloat(mf,(unsigned)len); - break; - case 0x9f: // Channels - v = readUInt(mf,(unsigned)len); - if (v<1 || v>255) - errorjmp(mf,"Invalid Channels value"); - ti->AV.Audio.Channels = (unsigned char)v; - break; - case 0x7d7b: // ChannelPositions - skipbytes(mf,len); - break; - case 0x6264: // BitDepth - v = readUInt(mf,(unsigned)len); -#if 0 - if ((v<1 || v>255) && !IsWritingApp(mf,"AVI-Mux GUI")) - errorjmp(mf,"Invalid BitDepth: %d",(int)v); -#endif - ti->AV.Audio.BitDepth = (unsigned char)v; - break; - ENDFOR(mf); - - if (ti->AV.Audio.Channels == 0) - ti->AV.Audio.Channels = 1; - if (mkv_TruncFloat(ti->AV.Audio.SamplingFreq) == 0) - ti->AV.Audio.SamplingFreq = mkfi(8000); - if (mkv_TruncFloat(ti->AV.Audio.OutputSamplingFreq)==0) - ti->AV.Audio.OutputSamplingFreq = ti->AV.Audio.SamplingFreq; -} - -static void CopyStr(char **src,char **dst) { - size_t l; - - if (!*src) - return; - - l = strlen(*src)+1; - memcpy(*dst,*src,l); - *src = *dst; - *dst += l; -} - -static void parseTrackEntry(MatroskaFile *mf,uint64_t toplen) { - struct TrackInfo t,*tp,**tpp; - uint64_t v; - char *cp = NULL, *cs = NULL; - size_t cplen = 0, cslen = 0, cpadd = 0; - unsigned CompScope, num_comp = 0; - - if (mf->nTracks >= MAX_TRACKS) - errorjmp(mf,"Too many tracks."); - - // clear track info - memset(&t,0,sizeof(t)); - - // fill default values - t.Enabled = 1; - t.Default = 1; - t.Lacing = 1; - t.TimecodeScale = mkfi(1); - t.DecodeAll = 1; - - FOREACH(mf,toplen) - case 0xd7: // TrackNumber - v = readUInt(mf,(unsigned)len); - if (v>255) - errorjmp(mf,"Track number is >255 (%d)",(int)v); - t.Number = (unsigned char)v; - break; - case 0x73c5: // TrackUID - t.UID = readUInt(mf,(unsigned)len); - break; - case 0x83: // TrackType - v = readUInt(mf,(unsigned)len); - if (v<1 || v>254) - errorjmp(mf,"Invalid track type: %d",(int)v); - t.Type = (unsigned char)v; - break; - case 0xb9: // Enabled - t.Enabled = readUInt(mf,(unsigned)len)!=0; - break; - case 0x88: // Default - t.Default = readUInt(mf,(unsigned)len)!=0; - break; - case 0x9c: // Lacing - t.Lacing = readUInt(mf,(unsigned)len)!=0; - break; - case 0x6de7: // MinCache - v = readUInt(mf,(unsigned)len); - if (v > 0xffffffff) - errorjmp(mf,"MinCache is too large"); - t.MinCache = (unsigned)v; - break; - case 0x6df8: // MaxCache - v = readUInt(mf,(unsigned)len); - if (v > 0xffffffff) - errorjmp(mf,"MaxCache is too large"); - t.MaxCache = (unsigned)v; - break; - case 0x23e383: // DefaultDuration - t.DefaultDuration = readUInt(mf,(unsigned)len); - break; - case 0x23314f: // TrackTimecodeScale - t.TimecodeScale = readFloat(mf,(unsigned)len); - break; - case 0x55ee: // MaxBlockAdditionID - t.MaxBlockAdditionID = (unsigned)readUInt(mf,(unsigned)len); - break; - case 0x536e: // Name - if (t.Name) - errorjmp(mf,"Duplicate Track Name"); - STRGETA(mf,t.Name,len); - break; - case 0x22b59c: // Language - readLangCC(mf, len, t.Language); - break; - case 0x86: // CodecID - if (t.CodecID) - errorjmp(mf,"Duplicate CodecID"); - STRGETA(mf,t.CodecID,len); - break; - case 0x63a2: // CodecPrivate - if (cp) - errorjmp(mf,"Duplicate CodecPrivate"); - cplen = (unsigned)len; - if (len > 262144) { // 256KB - cp = mf->cpbuf = mf->cache->memalloc(mf->cache, cplen); - if (!cp) - errorjmp(mf,"Out of memory"); - } - else - cp = alloca(cplen); - readbytes(mf,cp,cplen); - break; - case 0x258688: // CodecName - skipbytes(mf,len); - break; - case 0x3a9697: // CodecSettings - skipbytes(mf,len); - break; - case 0x3b4040: // CodecInfoURL - skipbytes(mf,len); - break; - case 0x26b240: // CodecDownloadURL - skipbytes(mf,len); - break; - case 0xaa: // CodecDecodeAll - t.DecodeAll = readUInt(mf,(unsigned)len)!=0; - break; - case 0x6fab: // TrackOverlay - v = readUInt(mf,(unsigned)len); - if (v>255) - errorjmp(mf,"Track number in TrackOverlay is too large: %d",(int)v); - t.TrackOverlay = (unsigned char)v; - break; - case 0xe0: // VideoInfo - parseVideoInfo(mf,len,&t); - break; - case 0xe1: // AudioInfo - parseAudioInfo(mf,len,&t); - break; - case 0x6d80: // ContentEncodings - FOREACH(mf,len) - case 0x6240: // ContentEncoding - // fill in defaults - t.CompEnabled = 1; - t.CompMethod = COMP_ZLIB; - CompScope = 1; - if (++num_comp > 1) - return; // only one compression layer supported - FOREACH(mf,len) - case 0x5031: // ContentEncodingOrder - readUInt(mf,(unsigned)len); - break; - case 0x5032: // ContentEncodingScope - CompScope = (unsigned)readUInt(mf,(unsigned)len); - break; - case 0x5033: // ContentEncodingType - if (readUInt(mf,(unsigned)len) != 0) - return; // encryption is not supported - break; - case 0x5034: // ContentCompression - FOREACH(mf,len) - case 0x4254: // ContentCompAlgo - v = readUInt(mf,(unsigned)len); - t.CompEnabled = 1; - switch (v) { - case 0: // Zlib - t.CompMethod = COMP_ZLIB; - break; - case 3: // prepend fixed data - t.CompMethod = COMP_PREPEND; - break; - default: - return; // unsupported compression, skip track - } - break; - case 0x4255: // ContentCompSettings - if (len > 256) - return; - cslen = (unsigned)len; - cs = alloca(cslen); - readbytes(mf, cs, cslen); - break; - ENDFOR(mf); - break; - // TODO Implement Encryption/Signatures - ENDFOR(mf); - break; - ENDFOR(mf); - break; - ENDFOR(mf); - - // validate track info - if (!t.CodecID) - errorjmp(mf,"Track has no Codec ID"); - - if (t.UID != 0) { - unsigned i; - for (i = 0; i < mf->nTracks; ++i) - if (mf->Tracks[i]->UID == t.UID) // duplicate track entry - return; - } - -#ifdef MATROSKA_COMPRESSION_SUPPORT - // handle compressed CodecPrivate - if (t.CompEnabled && t.CompMethod == COMP_ZLIB && (CompScope & 2) && cplen > 0) { - z_stream zs; - Bytef tmp[64], *ncp; - int code; - uLong ncplen; - - memset(&zs,0,sizeof(zs)); - if (inflateInit(&zs) != Z_OK) - errorjmp(mf, "inflateInit failed"); - - zs.next_in = (Bytef *)cp; - zs.avail_in = cplen; - - do { - zs.next_out = tmp; - zs.avail_out = sizeof(tmp); - - code = inflate(&zs, Z_NO_FLUSH); - } while (code == Z_OK); - - if (code != Z_STREAM_END) - errorjmp(mf, "invalid compressed data in CodecPrivate"); - - ncplen = zs.total_out; - ncp = alloca(ncplen); - - inflateReset(&zs); - - zs.next_in = (Bytef *)cp; - zs.avail_in = cplen; - zs.next_out = ncp; - zs.avail_out = ncplen; - - if (inflate(&zs, Z_FINISH) != Z_STREAM_END) - errorjmp(mf, "inflate failed"); - - inflateEnd(&zs); - - cp = (char *)ncp; - cplen = ncplen; - } -#endif - - if (t.CompEnabled && !(CompScope & 1)) { - t.CompEnabled = 0; - cslen = 0; - } - - // allocate new track - tpp = AGET(mf,Tracks); - - // copy strings - if (t.Name) - cpadd += strlen(t.Name)+1; - if (t.CodecID) - cpadd += strlen(t.CodecID)+1; - - tp = mf->cache->memalloc(mf->cache,sizeof(*tp) + cplen + cslen + cpadd); - if (tp == NULL) - errorjmp(mf,"Out of memory"); - - memcpy(tp,&t,sizeof(*tp)); - if (cplen) { - tp->CodecPrivate = tp+1; - tp->CodecPrivateSize = (unsigned)cplen; - memcpy(tp->CodecPrivate,cp,cplen); - } - if (cslen) { - tp->CompMethodPrivate = (char *)(tp+1) + cplen; - tp->CompMethodPrivateSize = (unsigned)cslen; - memcpy(tp->CompMethodPrivate, cs, cslen); - } - - cp = (char*)(tp+1) + cplen + cslen; - CopyStr(&tp->Name,&cp); - CopyStr(&tp->CodecID,&cp); - - // set default language - if (!tp->Language[0]) - memcpy(tp->Language, "eng", 4); - - *tpp = tp; -} - -static void parseTracks(MatroskaFile *mf,uint64_t toplen) { - mf->seen.Tracks = 1; - mf->cpbuf = NULL; - FOREACH(mf,toplen) - case 0xae: // TrackEntry - parseTrackEntry(mf,len); - mf->cache->memfree(mf->cache, mf->cpbuf); - mf->cpbuf = NULL; - break; - ENDFOR(mf); -} - -static void addCue(MatroskaFile *mf,uint64_t pos,uint64_t timecode) { - struct Cue *cc = AGET(mf,Cues); - cc->Time = timecode; - cc->Position = pos; - cc->Track = 0; - cc->Block = 0; -} - -static void fixupCues(MatroskaFile *mf) { - // adjust cues, shift cues if file does not start at 0 - unsigned i; - int64_t adjust = mf->firstTimecode * mf->Seg.TimecodeScale; - - for (i=0;inCues;++i) { - mf->Cues[i].Time *= mf->Seg.TimecodeScale; - mf->Cues[i].Time -= adjust; - } -} - -static void parseCues(MatroskaFile *mf,uint64_t toplen) { - jmp_buf jb; - uint64_t v; - struct Cue cc; - unsigned i,j,k; - - mf->seen.Cues = 1; - mf->nCues = 0; - cc.Block = 0; - - memcpy(&jb,&mf->jb,sizeof(jb)); - - if (setjmp(mf->jb)) { - memcpy(&mf->jb,&jb,sizeof(jb)); - mf->nCues = 0; - mf->seen.Cues = 0; - return; - } - - FOREACH(mf,toplen) - case 0xbb: // CuePoint - FOREACH(mf,len) - case 0xb3: // CueTime - cc.Time = readUInt(mf,(unsigned)len); - break; - case 0xb7: // CueTrackPositions - FOREACH(mf,len) - case 0xf7: // CueTrack - v = readUInt(mf,(unsigned)len); - if (v>255) - errorjmp(mf,"CueTrack points to an invalid track: %d",(int)v); - cc.Track = (unsigned char)v; - break; - case 0xf1: // CueClusterPosition - cc.Position = readUInt(mf,(unsigned)len); - break; - case 0x5378: // CueBlockNumber - cc.Block = readUInt(mf,(unsigned)len); - break; - case 0xea: // CodecState - readUInt(mf,(unsigned)len); - break; - case 0xdb: // CueReference - FOREACH(mf,len) - case 0x96: // CueRefTime - readUInt(mf,(unsigned)len); - break; - case 0x97: // CueRefCluster - readUInt(mf,(unsigned)len); - break; - case 0x535f: // CueRefNumber - readUInt(mf,(unsigned)len); - break; - case 0xeb: // CueRefCodecState - readUInt(mf,(unsigned)len); - break; - ENDFOR(mf); - break; - ENDFOR(mf); - break; - ENDFOR(mf); - - if (mf->nCues == 0 && mf->pCluster - mf->pSegment != cc.Position) - addCue(mf,mf->pCluster - mf->pSegment,mf->firstTimecode); - - memcpy(AGET(mf,Cues),&cc,sizeof(cc)); - break; - ENDFOR(mf); - - memcpy(&mf->jb,&jb,sizeof(jb)); - - ARELEASE(mf,mf,Cues); - - // bubble sort the cues and fuck the losers that write unordered cues - if (mf->nCues > 0) - for (i = mf->nCues - 1, k = 1; i > 0 && k > 0; --i) - for (j = k = 0; j < i; ++j) - if (mf->Cues[j].Time > mf->Cues[j+1].Time) { - struct Cue tmp = mf->Cues[j+1]; - mf->Cues[j+1] = mf->Cues[j]; - mf->Cues[j] = tmp; - ++k; - } -} - -static void parseAttachment(MatroskaFile *mf,uint64_t toplen) { - struct Attachment a,*pa; - - memset(&a,0,sizeof(a)); - FOREACH(mf,toplen) - case 0x467e: // Description - STRGETA(mf,a.Description,len); - break; - case 0x466e: // Name - STRGETA(mf,a.Name,len); - break; - case 0x4660: // MimeType - STRGETA(mf,a.MimeType,len); - break; - case 0x46ae: // UID - a.UID = readUInt(mf,(unsigned)len); - break; - case 0x465c: // Data - a.Position = filepos(mf); - a.Length = len; - skipbytes(mf,len); - break; - ENDFOR(mf); - - if (!a.Position) - return; - - pa = AGET(mf,Attachments); - memcpy(pa,&a,sizeof(a)); - - if (a.Description) - pa->Description = mystrdup(mf->cache,a.Description); - if (a.Name) - pa->Name = mystrdup(mf->cache,a.Name); - if (a.MimeType) - pa->MimeType = mystrdup(mf->cache,a.MimeType); -} - -static void parseAttachments(MatroskaFile *mf,uint64_t toplen) { - mf->seen.Attachments = 1; - - FOREACH(mf,toplen) - case 0x61a7: // AttachedFile - parseAttachment(mf,len); - break; - ENDFOR(mf); -} - -static void parseChapter(MatroskaFile *mf,uint64_t toplen,struct Chapter *parent) { - struct ChapterDisplay *disp; - struct ChapterProcess *proc; - struct ChapterCommand *cmd; - struct Chapter *ch = ASGET(mf,parent,Children); - - memset(ch,0,sizeof(*ch)); - - ch->Enabled = 1; - - FOREACH(mf,toplen) - case 0x73c4: // ChapterUID - ch->UID = readUInt(mf,(unsigned)len); - break; - case 0x6e67: // ChapterSegmentUID - if (len != sizeof(ch->SegmentUID)) - skipbytes(mf, len); - else - readbytes(mf, ch->SegmentUID, sizeof(ch->SegmentUID)); - break; - case 0x91: // ChapterTimeStart - ch->Start = readUInt(mf,(unsigned)len); - break; - case 0x92: // ChapterTimeEnd - ch->End = readUInt(mf,(unsigned)len); - break; - case 0x98: // ChapterFlagHidden - ch->Hidden = readUInt(mf,(unsigned)len)!=0; - break; - case 0x4598: // ChapterFlagEnabled - ch->Enabled = readUInt(mf,(unsigned)len)!=0; - break; - case 0x8f: // ChapterTrack - FOREACH(mf,len) - case 0x89: // ChapterTrackNumber - *(uint64_t*)(ASGET(mf,ch,Tracks)) = readUInt(mf,(unsigned)len); - break; - ENDFOR(mf); - break; - case 0x80: // ChapterDisplay - disp = NULL; - - FOREACH(mf,len) - case 0x85: // ChapterString - if (disp==NULL) { - disp = ASGET(mf,ch,Display); - memset(disp, 0, sizeof(*disp)); - } - if (disp->String) - skipbytes(mf,len); // Ignore duplicate string - else - STRGETM(mf,disp->String,len); - break; - case 0x437c: // ChapterLanguage - if (disp==NULL) { - disp = ASGET(mf,ch,Display); - memset(disp, 0, sizeof(*disp)); - } - readLangCC(mf, len, disp->Language); - break; - case 0x437e: // ChapterCountry - if (disp==NULL) { - disp = ASGET(mf,ch,Display); - memset(disp, 0, sizeof(*disp)); - } - readLangCC(mf, len, disp->Country); - break; - ENDFOR(mf); - - if (disp && !disp->String) - --ch->nDisplay; - break; - case 0x6944: // ChapProcess - proc = NULL; - - FOREACH(mf,len) - case 0x6955: // ChapProcessCodecID - if (proc == NULL) { - proc = ASGET(mf, ch, Process); - memset(proc, 0, sizeof(*proc)); - } - proc->CodecID = (unsigned)readUInt(mf,(unsigned)len); - break; - case 0x450d: // ChapProcessPrivate - if (proc == NULL) { - proc = ASGET(mf, ch, Process); - memset(proc, 0, sizeof(*proc)); - } - if (proc->CodecPrivate) - skipbytes(mf, len); - else { - proc->CodecPrivateLength = (unsigned)len; - STRGETM(mf,proc->CodecPrivate,len); - } - break; - case 0x6911: // ChapProcessCommand - if (proc == NULL) { - proc = ASGET(mf, ch, Process); - memset(proc, 0, sizeof(*proc)); - } - - cmd = NULL; - - FOREACH(mf,len) - case 0x6922: // ChapterCommandTime - if (cmd == NULL) { - cmd = ASGET(mf,proc,Commands); - memset(cmd, 0, sizeof(*cmd)); - } - cmd->Time = (unsigned)readUInt(mf,(unsigned)len); - break; - case 0x6933: // ChapterCommandString - if (cmd == NULL) { - cmd = ASGET(mf,proc,Commands); - memset(cmd, 0, sizeof(*cmd)); - } - if (cmd->Command) - skipbytes(mf,len); - else { - cmd->CommandLength = (unsigned)len; - STRGETM(mf,cmd->Command,len); - } - break; - ENDFOR(mf); - - if (cmd && !cmd->Command) - --proc->nCommands; - break; - ENDFOR(mf); - - if (proc && !proc->nCommands) - --ch->nProcess; - break; - case 0xb6: // Nested ChapterAtom - parseChapter(mf,len,ch); - break; - ENDFOR(mf); - - ARELEASE(mf,ch,Tracks); - ARELEASE(mf,ch,Display); - ARELEASE(mf,ch,Children); -} - -static void parseChapters(MatroskaFile *mf,uint64_t toplen) { - struct Chapter *ch; - - mf->seen.Chapters = 1; - - FOREACH(mf,toplen) - case 0x45b9: // EditionEntry - ch = AGET(mf,Chapters); - memset(ch, 0, sizeof(*ch)); - FOREACH(mf,len) - case 0x45bc: // EditionUID - ch->UID = readUInt(mf,(unsigned)len); - break; - case 0x45bd: // EditionFlagHidden - ch->Hidden = readUInt(mf,(unsigned)len)!=0; - break; - case 0x45db: // EditionFlagDefault - ch->Default = readUInt(mf,(unsigned)len)!=0; - break; - case 0x45dd: // EditionFlagOrdered - ch->Ordered = readUInt(mf,(unsigned)len)!=0; - break; - case 0xb6: // ChapterAtom - parseChapter(mf,len,ch); - break; - ENDFOR(mf); - break; - ENDFOR(mf); -} - -static void parseTags(MatroskaFile *mf,uint64_t toplen) { - struct Tag *tag; - struct Target *target; - struct SimpleTag *st; - - mf->seen.Tags = 1; - - FOREACH(mf,toplen) - case 0x7373: // Tag - tag = AGET(mf,Tags); - memset(tag,0,sizeof(*tag)); - - FOREACH(mf,len) - case 0x63c0: // Targets - FOREACH(mf,len) - case 0x63c5: // TrackUID - target = ASGET(mf,tag,Targets); - target->UID = readUInt(mf,(unsigned)len); - target->Type = TARGET_TRACK; - break; - case 0x63c4: // ChapterUID - target = ASGET(mf,tag,Targets); - target->UID = readUInt(mf,(unsigned)len); - target->Type = TARGET_CHAPTER; - break; - case 0x63c6: // AttachmentUID - target = ASGET(mf,tag,Targets); - target->UID = readUInt(mf,(unsigned)len); - target->Type = TARGET_ATTACHMENT; - break; - case 0x63c9: // EditionUID - target = ASGET(mf,tag,Targets); - target->UID = readUInt(mf,(unsigned)len); - target->Type = TARGET_EDITION; - break; - ENDFOR(mf); - break; - case 0x67c8: // SimpleTag - st = ASGET(mf,tag,SimpleTags); - memset(st,0,sizeof(*st)); - - FOREACH(mf,len) - case 0x45a3: // TagName - if (st->Name) - skipbytes(mf,len); - else - STRGETM(mf,st->Name,len); - break; - case 0x4487: // TagString - if (st->Value) - skipbytes(mf,len); - else - STRGETM(mf,st->Value,len); - break; - case 0x447a: // TagLanguage - readLangCC(mf, len, st->Language); - break; - case 0x4484: // TagDefault - st->Default = readUInt(mf,(unsigned)len)!=0; - break; - ENDFOR(mf); - - if (!st->Name || !st->Value) { - mf->cache->memfree(mf->cache,st->Name); - mf->cache->memfree(mf->cache,st->Value); - --tag->nSimpleTags; - } - break; - ENDFOR(mf); - break; - ENDFOR(mf); -} - -static void parseContainer(MatroskaFile *mf) { - uint64_t len; - int id = readID(mf); - if (id==EOF) - errorjmp(mf,"Unexpected EOF in parseContainer"); - - len = readSize(mf); - - switch (id) { - case 0x1549a966: // SegmentInfo - parseSegmentInfo(mf,len); - break; - case 0x1f43b675: // Cluster - parseFirstCluster(mf,len); - break; - case 0x1654ae6b: // Tracks - parseTracks(mf,len); - break; - case 0x1c53bb6b: // Cues - parseCues(mf,len); - break; - case 0x1941a469: // Attachments - parseAttachments(mf,len); - break; - case 0x1043a770: // Chapters - parseChapters(mf,len); - break; - case 0x1254c367: // Tags - parseTags(mf,len); - break; - } -} - -static void parseContainerPos(MatroskaFile *mf,uint64_t pos) { - seek(mf,pos); - parseContainer(mf); -} - -static void parsePointers(MatroskaFile *mf) { - jmp_buf jb; - - if (mf->pSegmentInfo && !mf->seen.SegmentInfo) - parseContainerPos(mf,mf->pSegmentInfo); - if (mf->pCluster && !mf->seen.Cluster) - parseContainerPos(mf,mf->pCluster); - if (mf->pTracks && !mf->seen.Tracks) - parseContainerPos(mf,mf->pTracks); - - memcpy(&jb,&mf->jb,sizeof(jb)); - - if (setjmp(mf->jb)) - mf->flags &= ~MPF_ERROR; // ignore errors - else { - if (mf->pCues && !mf->seen.Cues) - parseContainerPos(mf,mf->pCues); - if (mf->pAttachments && !mf->seen.Attachments) - parseContainerPos(mf,mf->pAttachments); - if (mf->pChapters && !mf->seen.Chapters) - parseContainerPos(mf,mf->pChapters); - if (mf->pTags && !mf->seen.Tags) - parseContainerPos(mf,mf->pTags); - } - - memcpy(&mf->jb,&jb,sizeof(jb)); -} - -static void parseSegment(MatroskaFile *mf,uint64_t toplen) { - uint64_t nextpos; - unsigned nSeekHeads = 0, dontstop = 0; - jmp_buf jb; - - memcpy(&jb,&mf->jb,sizeof(jb)); - - if (setjmp(mf->jb)) - mf->flags &= ~MPF_ERROR; - else { - // we want to read data until we find a seekhead or a trackinfo - FOREACH(mf,toplen) - case 0x114d9b74: // SeekHead - if (mf->flags & MKVF_AVOID_SEEKS) { - skipbytes(mf,len); - break; - } - - nextpos = filepos(mf) + len; - do { - mf->pSeekHead = 0; - parseSeekHead(mf,len); - ++nSeekHeads; - if (mf->pSeekHead) { // this is possibly a chained SeekHead - seek(mf,mf->pSeekHead); - id = readID(mf); - if (id==EOF) // chained SeekHead points to EOF? - break; - if (id != 0x114d9b74) // chained SeekHead doesnt point to a SeekHead? - break; - len = readSize(mf); - } - } while (mf->pSeekHead && nSeekHeads < 10); - seek(mf,nextpos); // resume reading segment - break; - case 0x1549a966: // SegmentInfo - mf->pSegmentInfo = cur; - parseSegmentInfo(mf,len); - break; - case 0x1f43b675: // Cluster - if (!mf->pCluster) - mf->pCluster = cur; - if (mf->seen.Cluster) - skipbytes(mf,len); - else - parseFirstCluster(mf,len); - break; - case 0x1654ae6b: // Tracks - mf->pTracks = cur; - parseTracks(mf,len); - break; - case 0x1c53bb6b: // Cues - mf->pCues = cur; - parseCues(mf,len); - break; - case 0x1941a469: // Attachments - mf->pAttachments = cur; - parseAttachments(mf,len); - break; - case 0x1043a770: // Chapters - mf->pChapters = cur; - parseChapters(mf,len); - break; - case 0x1254c367: // Tags - mf->pTags = cur; - parseTags(mf,len); - break; - ENDFOR1(mf); - // if we have pointers to all key elements - if (!dontstop && mf->pSegmentInfo && mf->pTracks && mf->pCluster) - break; - ENDFOR2(); - } - - memcpy(&mf->jb,&jb,sizeof(jb)); - - parsePointers(mf); -} - -static void parseBlockAdditions(MatroskaFile *mf, uint64_t toplen, uint64_t timecode, unsigned track) { - uint64_t add_id = 1, add_pos, add_len; - unsigned char have_add; - - FOREACH(mf, toplen) - case 0xa6: // BlockMore - have_add = 0; - FOREACH(mf, len) - case 0xee: // BlockAddId - add_id = readUInt(mf, (unsigned)len); - break; - case 0xa5: // BlockAddition - add_pos = filepos(mf); - add_len = len; - skipbytes(mf, len); - ++have_add; - break; - ENDFOR(mf); - if (have_add == 1 && id > 0 && id < 255) { - struct QueueEntry *qe = QAlloc(mf); - qe->Start = qe->End = timecode; - qe->Position = add_pos; - qe->Length = (unsigned)add_len; - qe->flags = FRAME_UNKNOWN_START | FRAME_UNKNOWN_END | - (((unsigned)add_id << FRAME_STREAM_SHIFT) & FRAME_STREAM_MASK); - - QPut(&mf->Queues[track],qe); - } - break; - ENDFOR(mf); -} - -static void parseBlockGroup(MatroskaFile *mf,uint64_t toplen,uint64_t timecode, int blockex) { - uint64_t v; - uint64_t duration = 0; - uint64_t dpos; - struct QueueEntry *qe,*qf = NULL; - unsigned char have_duration = 0, have_block = 0; - unsigned char gap = 0; - unsigned char lacing = 0; - unsigned char ref = 0; - unsigned char trackid; - unsigned tracknum = 0; - int c; - unsigned nframes = 0,i; - unsigned *sizes; - signed short block_timecode; - - if (blockex) - goto blockex; - - FOREACH(mf,toplen) - case 0xfb: // ReferenceBlock - readSInt(mf,(unsigned)len); - ref = 1; - break; -blockex: - cur = start = filepos(mf); - len = tmplen = toplen; - // fallthrough - case 0xa1: // Block - have_block = 1; - - dpos = filepos(mf); - - v = readVLUInt(mf); - if (v>255) - errorjmp(mf,"Invalid track number in Block: %d",(int)v); - trackid = (unsigned char)v; - - for (tracknum=0;tracknumnTracks;++tracknum) - if (mf->Tracks[tracknum]->Number == trackid) { - if (mf->trackMask & (1<Tracks[tracknum]->TimecodeScale, - (timecode - mf->firstTimecode + block_timecode) * mf->Seg.TimecodeScale); - - c = readch(mf); - if (c==EOF) - errorjmp(mf,"Unexpected EOF while reading Block flags"); - - if (blockex) - ref = (unsigned char)!(c & 0x80); - - gap = (unsigned char)(c & 0x1); - lacing = (unsigned char)((c >> 1) & 3); - - if (lacing) { - c = readch(mf); - if (c == EOF) - errorjmp(mf,"Unexpected EOF while reading lacing data"); - nframes = c+1; - } else - nframes = 1; - sizes = alloca(nframes*sizeof(*sizes)); - - switch (lacing) { - case 0: // No lacing - sizes[0] = (unsigned)(len - filepos(mf) + dpos); - break; - case 1: // Xiph lacing - sizes[nframes-1] = 0; - for (i=0;i1) - sizes[nframes-1] = (unsigned)(len - filepos(mf) + dpos) - sizes[0] - sizes[nframes-1]; - break; - case 2: // Fixed lacing - sizes[0] = (unsigned)(len - filepos(mf) + dpos)/nframes; - for (i=1;iStart = timecode; - qe->End = timecode; - qe->Position = v; - qe->Length = sizes[i]; - qe->flags = FRAME_UNKNOWN_END | FRAME_KF; - if (i == nframes-1 && gap) - qe->flags |= FRAME_GAP; - if (i > 0) - qe->flags |= FRAME_UNKNOWN_START; - - QPut(&mf->Queues[tracknum],qe); - - v += sizes[i]; - } - - // we want to still load these bytes into cache - for (v = filepos(mf) & ~0x3fff; v < len + dpos; v += 0x4000) - mf->cache->read(mf->cache,v,NULL,0); // touch page (FIXME this doesn't really do anything) - - skipbytes(mf,len - filepos(mf) + dpos); - - if (blockex) - goto out; - break; - case 0x9b: // BlockDuration - duration = readUInt(mf,(unsigned)len); - have_duration = 1; - break; - case 0x75a1: // BlockAdditions - if (nframes > 0) // have some frames - parseBlockAdditions(mf, len, timecode, tracknum); - else - skipbytes(mf, len); - break; - ENDFOR(mf); - -out: - if (!have_block) - errorjmp(mf,"Found a BlockGroup without Block"); - - if (nframes > 1) { - uint64_t defd = mf->Tracks[tracknum]->DefaultDuration; - v = qf->Start; - - if (have_duration) { - duration = mul3(mf->Tracks[tracknum]->TimecodeScale, - duration * mf->Seg.TimecodeScale); - - for (qe = qf; nframes > 1; --nframes, qe = qe->next) { - qe->Start = v; - v += defd; - duration -= defd; - qe->End = v; -#if 0 - qe->flags &= ~(FRAME_UNKNOWN_START|FRAME_UNKNOWN_END); -#endif - } - qe->Start = v; - qe->End = v + duration; - qe->flags &= ~FRAME_UNKNOWN_END; - } else if (mf->Tracks[tracknum]->DefaultDuration) { - for (qe = qf; nframes > 0; --nframes, qe = qe->next) { - qe->Start = v; - v += defd; - qe->End = v; - qe->flags &= ~(FRAME_UNKNOWN_START|FRAME_UNKNOWN_END); - } - } - } else if (nframes == 1) { - if (have_duration) { - qf->End = qf->Start + mul3(mf->Tracks[tracknum]->TimecodeScale, - duration * mf->Seg.TimecodeScale); - qf->flags &= ~FRAME_UNKNOWN_END; - } else if (mf->Tracks[tracknum]->DefaultDuration) { - qf->End = qf->Start + mf->Tracks[tracknum]->DefaultDuration; - qf->flags &= ~FRAME_UNKNOWN_END; - } - } - - if (ref) - while (qf) { - qf->flags &= ~FRAME_KF; - qf = qf->next; - } -} - -static void ClearQueue(MatroskaFile *mf,struct Queue *q) { - struct QueueEntry *qe,*qn; - - for (qe=q->head;qe;qe=qn) { - qn = qe->next; - qe->next = mf->QFreeList; - mf->QFreeList = qe; - } - - q->head = NULL; - q->tail = NULL; -} - -static void EmptyQueues(MatroskaFile *mf) { - unsigned i; - - for (i=0;inTracks;++i) - ClearQueue(mf,&mf->Queues[i]); -} - -static int readMoreBlocks(MatroskaFile *mf) { - uint64_t toplen, cstop; - int64_t cp; - int cid, ret = 0; - jmp_buf jb; - volatile unsigned retries = 0; - - if (mf->readPosition >= mf->pSegmentTop) - return EOF; - - memcpy(&jb,&mf->jb,sizeof(jb)); - - if (setjmp(mf->jb)) { // something evil happened here, try to resync - // always advance read position no matter what so - // we don't get caught in an endless loop - mf->readPosition = filepos(mf); - - ret = EOF; - - if (++retries > 3) // don't try too hard - goto ex; - - for (;;) { - if (filepos(mf) >= mf->pSegmentTop) - goto ex; - - cp = mf->cache->scan(mf->cache,filepos(mf),0x1f43b675); // cluster - - if (cp < 0 || (uint64_t)cp >= mf->pSegmentTop) - goto ex; - - seek(mf,cp); - - cid = readID(mf); - if (cid == EOF) - goto ex; - if (cid == 0x1f43b675) { - toplen = readSize(mf); - if (toplen < MAXCLUSTER) { - // reset error flags - mf->flags &= ~MPF_ERROR; - ret = RBRESYNC; - break; - } - } - } - - mf->readPosition = cp; - } - - cstop = mf->cache->getcachesize(mf->cache)>>1; - if (cstop > MAX_READAHEAD) - cstop = MAX_READAHEAD; - cstop += mf->readPosition; - - seek(mf,mf->readPosition); - - while (filepos(mf) < mf->pSegmentTop) { - cid = readID(mf); - if (cid == EOF) { - ret = EOF; - break; - } - toplen = readSize(mf); - - if (cid == 0x1f43b675) { // Cluster - unsigned char have_timecode = 0; - - FOREACH(mf,toplen) - case 0xe7: // Timecode - mf->tcCluster = readUInt(mf,(unsigned)len); - have_timecode = 1; - break; - case 0xa7: // Position - readUInt(mf,(unsigned)len); - break; - case 0xab: // PrevSize - readUInt(mf,(unsigned)len); - break; - case 0x5854: { // SilentTracks - // unsigned stmask = 0; - unsigned i, trk; - FOREACH(mf, len) - case 0x58d7: // SilentTrackNumber - trk = (unsigned)readUInt(mf, (unsigned)len); - for (i = 0; i < mf->nTracks; ++i) - if (mf->Tracks[i]->Number == trk) { - // stmask |= 1 << i; - break; - } - break; - ENDFOR(mf); - // TODO pass stmask to reading app - break; } - case 0xa0: // BlockGroup - if (!have_timecode) - errorjmp(mf,"Found BlockGroup before cluster TimeCode"); - parseBlockGroup(mf,len,mf->tcCluster, 0); - goto out; - case 0xa3: // BlockEx - if (!have_timecode) - errorjmp(mf,"Found BlockGroup before cluster TimeCode"); - parseBlockGroup(mf, len, mf->tcCluster, 1); - goto out; - ENDFOR(mf); -out:; - } else { - if (toplen > MAXFRAME) - errorjmp(mf,"Element in a cluster is too large around %llu, %X [%u]",filepos(mf),cid,(unsigned)toplen); - if (cid == 0xa0) // BlockGroup - parseBlockGroup(mf,toplen,mf->tcCluster, 0); - else if (cid == 0xa3) // BlockEx - parseBlockGroup(mf, toplen, mf->tcCluster, 1); - else - skipbytes(mf,toplen); - } - - if ((mf->readPosition = filepos(mf)) > cstop) - break; - } - - mf->readPosition = filepos(mf); - -ex: - memcpy(&mf->jb,&jb,sizeof(jb)); - - return ret; -} - -// this is almost the same as readMoreBlocks, except it ensures -// there are no partial frames queued, however empty queues are ok -static int fillQueues(MatroskaFile *mf,unsigned int mask) { - unsigned i,j; - int ret = 0; - - for (;;) { - j = 0; - - for (i=0;inTracks;++i) - if (mf->Queues[i].head && !(mask & (1<0) // have at least some frames - return ret; - - if ((ret = readMoreBlocks(mf)) < 0) { - j = 0; - for (i=0;inTracks;++i) - if (mf->Queues[i].head && !(mask & (1<pCluster; - uint64_t step = 10*1024*1024; - uint64_t size, tc, isize; - int64_t next_cluster; - int id, have_tc, bad; - struct Cue *cue; - - if (pos >= mf->pSegmentTop) - return; - - if (pos + step * 10 > mf->pSegmentTop) - step = (mf->pSegmentTop - pos) / 10; - if (step == 0) - step = 1; - - memcpy(&jb,&mf->jb,sizeof(jb)); - - // remove all cues - mf->nCues = 0; - - bad = 0; - - while (pos < mf->pSegmentTop) { - if (!mf->cache->progress(mf->cache,pos,mf->pSegmentTop)) - break; - - if (++bad > 50) { - pos += step; - bad = 0; - continue; - } - - // find next cluster header - next_cluster = mf->cache->scan(mf->cache,pos,0x1f43b675); // cluster - if (next_cluster < 0 || (uint64_t)next_cluster >= mf->pSegmentTop) - break; - - pos = next_cluster + 4; // prevent endless loops - - if (setjmp(mf->jb)) // something evil happened while reindexing - continue; - - seek(mf,next_cluster); - - id = readID(mf); - if (id == EOF) - break; - if (id != 0x1f43b675) // shouldn't happen - continue; - - size = readVLUInt(mf); - if (size >= MAXCLUSTER || size < 1024) - continue; - - have_tc = 0; - size += filepos(mf); - - while (filepos(mf) < (uint64_t)next_cluster + 1024) { - id = readID(mf); - if (id == EOF) - break; - - isize = readVLUInt(mf); - - if (id == 0xe7) { // cluster timecode - tc = readUInt(mf,(unsigned)isize); - have_tc = 1; - break; - } - - skipbytes(mf,isize); - } - - if (!have_tc) - continue; - - seek(mf,size); - id = readID(mf); - - if (id == EOF) - break; - - if (id != 0x1f43b675) // cluster - continue; - - // good cluster, remember it - cue = AGET(mf,Cues); - cue->Time = tc; - cue->Position = next_cluster - mf->pSegment; - cue->Block = 0; - cue->Track = 0; - - // advance to the next point - pos = next_cluster + step; - if (pos < size) - pos = size; - - bad = 0; - } - - fixupCues(mf); - - if (mf->nCues == 0) { - cue = AGET(mf,Cues); - cue->Time = mf->firstTimecode; - cue->Position = mf->pCluster - mf->pSegment; - cue->Block = 0; - cue->Track = 0; - } - - mf->cache->progress(mf->cache,0,0); - - memcpy(&mf->jb,&jb,sizeof(jb)); -} - -static void fixupChapter(uint64_t adj, struct Chapter *ch) { - unsigned i; - - if (ch->Start != 0) - ch->Start -= adj; - if (ch->End != 0) - ch->End -= adj; - - for (i=0;inChildren;++i) - fixupChapter(adj,&ch->Children[i]); -} - -static int64_t findLastTimecode(MatroskaFile *mf) { - uint64_t nd = 0; - unsigned n,vtrack; - - if (mf->nTracks == 0) - return -1; - - for (n=vtrack=0;nnTracks;++n) - if (mf->Tracks[n]->Type == TT_VIDEO) { - vtrack = n; - goto ok; - } - - return -1; -ok: - - EmptyQueues(mf); - - if (mf->nCues == 0) { - mf->readPosition = mf->pCluster + 13000000 > mf->pSegmentTop ? mf->pCluster : mf->pSegmentTop - 13000000; - mf->tcCluster = 0; - } else { - mf->readPosition = mf->Cues[mf->nCues - 1].Position + mf->pSegment; - mf->tcCluster = mf->Cues[mf->nCues - 1].Time / mf->Seg.TimecodeScale; - } - mf->trackMask = ~(1 << vtrack); - - do - while (mf->Queues[vtrack].head) - { - uint64_t tc = mf->Queues[vtrack].head->flags & FRAME_UNKNOWN_END ? - mf->Queues[vtrack].head->Start : mf->Queues[vtrack].head->End; - if (nd < tc) - nd = tc; - QFree(mf,QGet(&mf->Queues[vtrack])); - } - while (fillQueues(mf,0) != EOF); - - mf->trackMask = 0; - - EmptyQueues(mf); - - // there may have been an error, but at this point we will ignore it - if (mf->flags & MPF_ERROR) { - mf->flags &= ~MPF_ERROR; - if (nd == 0) - return -1; - } - - return nd; -} - -static void parseFile(MatroskaFile *mf) { - uint64_t len = filepos(mf), adjust; - unsigned i; - int id = readID(mf); - int m; - - if (id==EOF) - errorjmp(mf,"Unexpected EOF at start of file"); - - // files with multiple concatenated segments can have only - // one EBML prolog - if (len > 0 && id == 0x18538067) - goto segment; - - if (id!=0x1a45dfa3) - errorjmp(mf,"First element in file is not EBML"); - - parseEBML(mf,readSize(mf)); - - // next we need to find the first segment - for (;;) { - id = readID(mf); - if (id==EOF) - errorjmp(mf,"No segments found in the file"); -segment: - len = readVLUIntImp(mf,&m); - // see if it's unspecified - if (len == (MAXU64 >> (57-m*7))) - len = MAXU64; - if (id == 0x18538067) // Segment - break; - skipbytes(mf,len); - } - - // found it - mf->pSegment = filepos(mf); - if (len == MAXU64) { - mf->pSegmentTop = MAXU64; - if (mf->cache->getfilesize) { - int64_t seglen = mf->cache->getfilesize(mf->cache); - if (seglen > 0) - mf->pSegmentTop = seglen; - } - } else - mf->pSegmentTop = mf->pSegment + len; - parseSegment(mf,len); - - // check if we got all data - if (!mf->seen.SegmentInfo) - errorjmp(mf,"Couldn't find SegmentInfo"); - if (!mf->seen.Cluster) - mf->pCluster = mf->pSegmentTop; - - adjust = mf->firstTimecode * mf->Seg.TimecodeScale; - - for (i=0;inChapters;++i) - fixupChapter(adjust, &mf->Chapters[i]); - - fixupCues(mf); - - // release extra memory - ARELEASE(mf,mf,Tracks); - - // initialize reader - mf->Queues = mf->cache->memalloc(mf->cache,mf->nTracks * sizeof(*mf->Queues)); - if (mf->Queues == NULL) - errorjmp(mf, "Ouf of memory"); - memset(mf->Queues, 0, mf->nTracks * sizeof(*mf->Queues)); - - // try to detect real duration - if (!(mf->flags & MKVF_AVOID_SEEKS)) { - int64_t nd = findLastTimecode(mf); - if (nd > 0) - mf->Seg.Duration = nd; - } - - // move to first frame - mf->readPosition = mf->pCluster; - mf->tcCluster = mf->firstTimecode; -} - -static void DeleteChapter(MatroskaFile *mf,struct Chapter *ch) { - unsigned i,j; - - for (i=0;inDisplay;++i) - mf->cache->memfree(mf->cache,ch->Display[i].String); - mf->cache->memfree(mf->cache,ch->Display); - mf->cache->memfree(mf->cache,ch->Tracks); - - for (i=0;inProcess;++i) { - for (j=0;jProcess[i].nCommands;++j) - mf->cache->memfree(mf->cache,ch->Process[i].Commands[j].Command); - mf->cache->memfree(mf->cache,ch->Process[i].Commands); - mf->cache->memfree(mf->cache,ch->Process[i].CodecPrivate); - } - mf->cache->memfree(mf->cache,ch->Process); - - for (i=0;inChildren;++i) - DeleteChapter(mf,&ch->Children[i]); - mf->cache->memfree(mf->cache,ch->Children); -} - -/////////////////////////////////////////////////////////////////////////// -// public interface -MatroskaFile *mkv_OpenEx(InputStream *io, - uint64_t base, - unsigned flags, - char *err_msg,unsigned msgsize) -{ - MatroskaFile *mf = io->memalloc(io,sizeof(*mf)); - if (mf == NULL) { - mystrlcpy(err_msg,"Out of memory",msgsize); - return NULL; - } - - memset(mf,0,sizeof(*mf)); - - mf->cache = io; - mf->flags = flags; - io->progress(io,0,0); - - if (setjmp(mf->jb)==0) { - seek(mf,base); - parseFile(mf); - } else { // parser error - mystrlcpy(err_msg,mf->errmsg,msgsize); - mkv_Close(mf); - return NULL; - } - - return mf; -} - -MatroskaFile *mkv_Open(InputStream *io, - char *err_msg,unsigned msgsize) -{ - return mkv_OpenEx(io,0,0,err_msg,msgsize); -} - -void mkv_Close(MatroskaFile *mf) { - unsigned i,j; - - if (mf==NULL) - return; - - for (i=0;inTracks;++i) - mf->cache->memfree(mf->cache,mf->Tracks[i]); - mf->cache->memfree(mf->cache,mf->Tracks); - - for (i=0;inQBlocks;++i) - mf->cache->memfree(mf->cache,mf->QBlocks[i]); - mf->cache->memfree(mf->cache,mf->QBlocks); - - mf->cache->memfree(mf->cache,mf->Queues); - - mf->cache->memfree(mf->cache,mf->Seg.Title); - mf->cache->memfree(mf->cache,mf->Seg.MuxingApp); - mf->cache->memfree(mf->cache,mf->Seg.WritingApp); - mf->cache->memfree(mf->cache,mf->Seg.Filename); - mf->cache->memfree(mf->cache,mf->Seg.NextFilename); - mf->cache->memfree(mf->cache,mf->Seg.PrevFilename); - - mf->cache->memfree(mf->cache,mf->Cues); - - for (i=0;inAttachments;++i) { - mf->cache->memfree(mf->cache,mf->Attachments[i].Description); - mf->cache->memfree(mf->cache,mf->Attachments[i].Name); - mf->cache->memfree(mf->cache,mf->Attachments[i].MimeType); - } - mf->cache->memfree(mf->cache,mf->Attachments); - - for (i=0;inChapters;++i) - DeleteChapter(mf,&mf->Chapters[i]); - mf->cache->memfree(mf->cache,mf->Chapters); - - for (i=0;inTags;++i) { - for (j=0;jTags[i].nSimpleTags;++j) { - mf->cache->memfree(mf->cache,mf->Tags[i].SimpleTags[j].Name); - mf->cache->memfree(mf->cache,mf->Tags[i].SimpleTags[j].Value); - } - mf->cache->memfree(mf->cache,mf->Tags[i].Targets); - mf->cache->memfree(mf->cache,mf->Tags[i].SimpleTags); - } - mf->cache->memfree(mf->cache,mf->Tags); - - mf->cache->memfree(mf->cache,mf); -} - -const char *mkv_GetLastError(MatroskaFile *mf) { - return mf->errmsg[0] ? mf->errmsg : NULL; -} - -SegmentInfo *mkv_GetFileInfo(MatroskaFile *mf) { - return &mf->Seg; -} - -unsigned int mkv_GetNumTracks(MatroskaFile *mf) { - return mf->nTracks; -} - -TrackInfo *mkv_GetTrackInfo(MatroskaFile *mf,unsigned track) { - if (track>mf->nTracks) - return NULL; - - return mf->Tracks[track]; -} - -void mkv_GetAttachments(MatroskaFile *mf,Attachment **at,unsigned *count) { - *at = mf->Attachments; - *count = mf->nAttachments; -} - -void mkv_GetChapters(MatroskaFile *mf,Chapter **ch,unsigned *count) { - *ch = mf->Chapters; - *count = mf->nChapters; -} - -void mkv_GetTags(MatroskaFile *mf,Tag **tag,unsigned *count) { - *tag = mf->Tags; - *count = mf->nTags; -} - -uint64_t mkv_GetSegmentTop(MatroskaFile *mf) { - return mf->pSegmentTop; -} - -#define IS_DELTA(f) (!((f)->flags & FRAME_KF) || ((f)->flags & FRAME_UNKNOWN_START)) - -void mkv_Seek(MatroskaFile *mf,uint64_t timecode,unsigned flags) { - int i,j,m,ret; - unsigned n,z,mask; - uint64_t m_kftime[MAX_TRACKS]; - unsigned char m_seendf[MAX_TRACKS]; - - if (mf->flags & MKVF_AVOID_SEEKS) - return; - - if (timecode == 0) { - EmptyQueues(mf); - mf->readPosition = mf->pCluster; - mf->tcCluster = mf->firstTimecode; - mf->flags &= ~MPF_ERROR; - - return; - } - - if (mf->nCues==0) - reindex(mf); - - if (mf->nCues==0) - return; - - mf->flags &= ~MPF_ERROR; - - i = 0; - j = mf->nCues - 1; - - for (;;) { - if (i>j) { - j = j>=0 ? j : 0; - - if (setjmp(mf->jb)!=0) - return; - - mkv_SetTrackMask(mf,mf->trackMask); - - if (flags & (MKVF_SEEK_TO_PREV_KEYFRAME | MKVF_SEEK_TO_PREV_KEYFRAME_STRICT)) { - // we do this in two stages - // a. find the last keyframes before the require position - // b. seek to them - - // pass 1 - for (;;) { - for (n=0;nnTracks;++n) { - m_kftime[n] = MAXU64; - m_seendf[n] = 0; - } - - EmptyQueues(mf); - - mf->readPosition = mf->Cues[j].Position + mf->pSegment; - mf->tcCluster = mf->Cues[j].Time; - - for (;;) { - if ((ret = fillQueues(mf,0)) < 0 || ret == RBRESYNC) - return; - - // drain queues until we get to the required timecode - for (n=0;nnTracks;++n) { - if (mf->Queues[n].head && (mf->Queues[n].head->StartQueues[n].head)) - m_seendf[n] = 1; - else - m_kftime[n] = mf->Queues[n].head->Start; - } - - while (mf->Queues[n].head && mf->Queues[n].head->StartQueues[n].head)) - m_seendf[n] = 1; - else - m_kftime[n] = mf->Queues[n].head->Start; - QFree(mf,QGet(&mf->Queues[n])); - } - - // We've drained the queue, so the frame at head is the next one past the requered point. - // In strict mode we are done, but when seeking is not strict we use the head frame - // if it's not an audio track (we accept preroll within a frame for audio), and the head frame - // is a keyframe - if (!(flags & MKVF_SEEK_TO_PREV_KEYFRAME_STRICT)) - if (mf->Queues[n].head && (mf->Tracks[n]->Type != TT_AUDIO || mf->Queues[n].head->Start<=timecode)) - if (!IS_DELTA(mf->Queues[n].head)) - m_kftime[n] = mf->Queues[n].head->Start; - } - - for (n=0;nnTracks;++n) - if (mf->Queues[n].head && mf->Queues[n].head->Start>=timecode) - goto found; - } -found: - - for (n=0;nnTracks;++n) - if (!(mf->trackMask & (1<0) - { - // we need to restart the search from prev cue - --j; - goto again; - } - - break; -again:; - } - } else - for (n=0;nnTracks;++n) - m_kftime[n] = timecode; - - // now seek to this timecode - EmptyQueues(mf); - - mf->readPosition = mf->Cues[j].Position + mf->pSegment; - mf->tcCluster = mf->Cues[j].Time; - - for (mask=0;;) { - if ((ret = fillQueues(mf,mask)) < 0 || ret == RBRESYNC) - return; - - // drain queues until we get to the required timecode - for (n=0;nnTracks;++n) { - struct QueueEntry *qe; - for (qe = mf->Queues[n].head;qe && qe->StartQueues[n].head) - QFree(mf,QGet(&mf->Queues[n])); - } - - for (n=z=0;nnTracks;++n) - if (m_kftime[n]==MAXU64 || (mf->Queues[n].head && mf->Queues[n].head->Start>=m_kftime[n])) { - ++z; - mask |= 1<nTracks) - return; - } - } - - m = (i+j)>>1; - - if (timecode < mf->Cues[m].Time) - j = m-1; - else - i = m+1; - } -} - -void mkv_SkipToKeyframe(MatroskaFile *mf) { - unsigned n,wait; - uint64_t ht; - - if (setjmp(mf->jb)!=0) - return; - - // remove delta frames from queues - do { - wait = 0; - - if (fillQueues(mf,0)<0) - return; - - for (n=0;nnTracks;++n) - if (mf->Queues[n].head && !(mf->Queues[n].head->flags & FRAME_KF)) { - ++wait; - QFree(mf,QGet(&mf->Queues[n])); - } - } while (wait); - - // find highest queued time - for (n=0,ht=0;nnTracks;++n) - if (mf->Queues[n].head && htQueues[n].head->Start) - ht = mf->Queues[n].head->Start; - - // ensure the time difference is less than 100ms - do { - wait = 0; - - if (fillQueues(mf,0)<0) - return; - - for (n=0;nnTracks;++n) - while (mf->Queues[n].head && mf->Queues[n].head->next && - (mf->Queues[n].head->next->flags & FRAME_KF) && - ht - mf->Queues[n].head->Start > 100000000) - { - ++wait; - QFree(mf,QGet(&mf->Queues[n])); - } - - } while (wait); -} - -uint64_t mkv_GetLowestQTimecode(MatroskaFile *mf) { - unsigned n,seen; - uint64_t t; - - // find the lowest queued timecode - for (n=seen=0,t=0;nnTracks;++n) - if (mf->Queues[n].head && (!seen || t > mf->Queues[n].head->Start)) - t = mf->Queues[n].head->Start, seen=1; - - return seen ? t : (uint64_t)LL(-1); -} - -int mkv_TruncFloat(MKFLOAT f) { -#ifdef MATROSKA_INTEGER_ONLY - return (int)(f.v >> 32); -#else - return (int)f; -#endif -} - -#define FTRACK 0xffffffff - -void mkv_SetTrackMask(MatroskaFile *mf,unsigned int mask) { - unsigned int i; - - if (mf->flags & MPF_ERROR) - return; - - mf->trackMask = mask; - - for (i=0;inTracks;++i) - if (mask & (1<Queues[i]); -} - -int mkv_ReadFrame(MatroskaFile *mf, - unsigned int mask,unsigned int *track, - uint64_t *StartTime,uint64_t *EndTime, - uint64_t *FilePos,unsigned int *FrameSize, - unsigned int *FrameFlags) -{ - unsigned int i,j; - struct QueueEntry *qe; - - if (setjmp(mf->jb)!=0) - return -1; - - do { - // extract required frame, use block with the lowest timecode - for (j=FTRACK,i=0;inTracks;++i) - if (!(mask & (1<Queues[i].head) { - j = i; - ++i; - break; - } - - for (;inTracks;++i) - if (!(mask & (1<Queues[i].head && - mf->Queues[j].head->Start > mf->Queues[i].head->Start) - j = i; - - if (j != FTRACK) { - qe = QGet(&mf->Queues[j]); - - *track = j; - *StartTime = qe->Start; - *EndTime = qe->End; - *FilePos = qe->Position; - *FrameSize = qe->Length; - *FrameFlags = qe->flags; - - QFree(mf,qe); - - return 0; - } - - if (mf->flags & MPF_ERROR) - return -1; - - } while (fillQueues(mf,mask)>=0); - - return EOF; -} - -#ifdef MATROSKA_COMPRESSION_SUPPORT -/************************************************************************* - * Compressed streams support - ************************************************************************/ -struct CompressedStream { - MatroskaFile *mf; - z_stream zs; - - /* current compressed frame */ - uint64_t frame_pos; - unsigned frame_size; - char frame_buffer[2048]; - - /* decoded data buffer */ - char decoded_buffer[2048]; - unsigned decoded_ptr; - unsigned decoded_size; - - /* error handling */ - char errmsg[128]; -}; - -CompressedStream *cs_Create(/* in */ MatroskaFile *mf, - /* in */ unsigned tracknum, - /* out */ char *errormsg, - /* in */ unsigned msgsize) -{ - CompressedStream *cs; - TrackInfo *ti; - int code; - - ti = mkv_GetTrackInfo(mf, tracknum); - if (ti == NULL) { - mystrlcpy(errormsg, "No such track.", msgsize); - return NULL; - } - - if (!ti->CompEnabled) { - mystrlcpy(errormsg, "Track is not compressed.", msgsize); - return NULL; - } - - if (ti->CompMethod != COMP_ZLIB) { - mystrlcpy(errormsg, "Unsupported compression method.", msgsize); - return NULL; - } - - cs = mf->cache->memalloc(mf->cache,sizeof(*cs)); - if (cs == NULL) { - mystrlcpy(errormsg, "Ouf of memory.", msgsize); - return NULL; - } - - memset(&cs->zs,0,sizeof(cs->zs)); - code = inflateInit(&cs->zs); - if (code != Z_OK) { - mystrlcpy(errormsg, "ZLib error.", msgsize); - mf->cache->memfree(mf->cache,cs); - return NULL; - } - - cs->frame_size = 0; - cs->decoded_ptr = cs->decoded_size = 0; - cs->mf = mf; - - return cs; -} - -void cs_Destroy(/* in */ CompressedStream *cs) { - if (cs == NULL) - return; - - inflateEnd(&cs->zs); - cs->mf->cache->memfree(cs->mf->cache,cs); -} - -/* advance to the next frame in matroska stream, you need to pass values returned - * by mkv_ReadFrame */ -void cs_NextFrame(/* in */ CompressedStream *cs, - /* in */ uint64_t pos, - /* in */ unsigned size) -{ - cs->zs.avail_in = 0; - inflateReset(&cs->zs); - cs->frame_pos = pos; - cs->frame_size = size; - cs->decoded_ptr = cs->decoded_size = 0; -} - -/* read and decode more data from current frame, return number of bytes decoded, - * 0 on end of frame, or -1 on error */ -int cs_ReadData(CompressedStream *cs,char *buffer,unsigned bufsize) -{ - char *cp = buffer; - unsigned rd = 0; - unsigned todo; - int code; - - do { - /* try to copy data from decoded buffer */ - if (cs->decoded_ptr < cs->decoded_size) { - todo = cs->decoded_size - cs->decoded_ptr;; - if (todo > bufsize - rd) - todo = bufsize - rd; - - memcpy(cp, cs->decoded_buffer + cs->decoded_ptr, todo); - - rd += todo; - cp += todo; - cs->decoded_ptr += todo; - } else { - /* setup output buffer */ - cs->zs.next_out = (Bytef *)cs->decoded_buffer; - cs->zs.avail_out = sizeof(cs->decoded_buffer); - - /* try to read more data */ - if (cs->zs.avail_in == 0 && cs->frame_size > 0) { - todo = cs->frame_size; - if (todo > sizeof(cs->frame_buffer)) - todo = sizeof(cs->frame_buffer); - - if (cs->mf->cache->read(cs->mf->cache, cs->frame_pos, cs->frame_buffer, todo) != (int)todo) { - mystrlcpy(cs->errmsg, "File read failed", sizeof(cs->errmsg)); - return -1; - } - - cs->zs.next_in = (Bytef *)cs->frame_buffer; - cs->zs.avail_in = todo; - - cs->frame_pos += todo; - cs->frame_size -= todo; - } - - /* try to decode more data */ - code = inflate(&cs->zs,Z_NO_FLUSH); - if (code != Z_OK && code != Z_STREAM_END) { - mystrlcpy(cs->errmsg, "ZLib error.", sizeof(cs->errmsg)); - return -1; - } - - /* handle decoded data */ - if (cs->zs.avail_out == sizeof(cs->decoded_buffer)) /* EOF */ - break; - - cs->decoded_ptr = 0; - cs->decoded_size = sizeof(cs->decoded_buffer) - cs->zs.avail_out; - } - } while (rd < bufsize); - - return rd; -} - -/* return error message for the last error */ -const char *cs_GetLastError(CompressedStream *cs) -{ - if (!cs->errmsg[0]) - return NULL; - return cs->errmsg; -} -#endif - diff --git a/src/MatroskaParser.h b/src/MatroskaParser.h deleted file mode 100644 index 74d5ba6c64..0000000000 --- a/src/MatroskaParser.h +++ /dev/null @@ -1,386 +0,0 @@ -/* - * Copyright (c) 2004-2009 Mike Matsnev. All Rights Reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions - * are met: - * - * 1. Redistributions of source code must retain the above copyright - * notice immediately at the beginning of the file, without modification, - * this list of conditions, and the following disclaimer. - * 2. Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * 3. Absolutely no warranty of function or purpose is made by the author - * Mike Matsnev. - * - * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR - * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES - * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. - * IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, INDIRECT, - * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT - * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, - * DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY - * THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF - * THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - */ - -#include - -#ifndef MATROSKA_PARSER_H -#define MATROSKA_PARSER_H - -/* Random notes: - * - * The parser does not process frame data in any way and does not read it into - * the queue. The app should read it via mkv_ReadData if it is interested. - * - * The code here is 64-bit clean and was tested on FreeBSD/sparc 64-bit big endian - * system - */ - -#ifdef MPDLLBUILD -#define X __declspec(dllexport) -#else -#ifdef MPDLL -#define X __declspec(dllimport) -#pragma comment(lib,"MatroskaParser") -#else -#define X -#endif -#endif - -#define MATROSKA_COMPRESSION_SUPPORT - -#ifdef __cplusplus -extern "C" { -#endif - -/* MKFLOATing point */ -#ifdef MATROSKA_INTEGER_ONLY -typedef struct { - int64_t v; -} MKFLOAT; -#else -typedef double MKFLOAT; -#endif - -/* generic I/O */ -struct InputStream { - /* read bytes from stream */ - int (*read)(struct InputStream *cc,uint64_t pos,void *buffer,int count); - /* scan for a four byte signature, bytes must be nonzero */ - int64_t (*scan)(struct InputStream *cc,uint64_t start,unsigned signature); - /* get cache size, this is used to cap readahead */ - unsigned (*getcachesize)(struct InputStream *cc); - /* fetch last error message */ - const char *(*geterror)(struct InputStream *cc); - /* memory allocation */ - void *(*memalloc)(struct InputStream *cc,size_t size); - void *(*memrealloc)(struct InputStream *cc,void *mem,size_t newsize); - void (*memfree)(struct InputStream *cc,void *mem); - /* zero return causes parser to abort open */ - int (*progress)(struct InputStream *cc,uint64_t cur,uint64_t max); - /* get file size, optional, can be NULL or return -1 if filesize is unknown */ - int64_t (*getfilesize)(struct InputStream *cc); -}; - -typedef struct InputStream InputStream; - -/* matroska file */ -struct MatroskaFile; /* opaque */ - -typedef struct MatroskaFile MatroskaFile; - -#define COMP_ZLIB 0 -#define COMP_BZIP 1 -#define COMP_LZO1X 2 -#define COMP_PREPEND 3 - -#define TT_VIDEO 1 -#define TT_AUDIO 2 -#define TT_SUB 17 - -struct TrackInfo { - unsigned char Number; - unsigned char Type; - unsigned char TrackOverlay; - uint64_t UID; - uint64_t MinCache; - uint64_t MaxCache; - uint64_t DefaultDuration; - MKFLOAT TimecodeScale; - void *CodecPrivate; - unsigned CodecPrivateSize; - unsigned CompMethod; - void *CompMethodPrivate; - unsigned CompMethodPrivateSize; - unsigned MaxBlockAdditionID; - - unsigned int Enabled:1; - unsigned int Default:1; - unsigned int Lacing:1; - unsigned int DecodeAll:1; - unsigned int CompEnabled:1; - - union { - struct { - unsigned char StereoMode; - unsigned char DisplayUnit; - unsigned char AspectRatioType; - unsigned int PixelWidth; - unsigned int PixelHeight; - unsigned int DisplayWidth; - unsigned int DisplayHeight; - unsigned int CropL, CropT, CropR, CropB; - unsigned int ColourSpace; - MKFLOAT GammaValue; - - unsigned int Interlaced:1; - } Video; - struct { - MKFLOAT SamplingFreq; - MKFLOAT OutputSamplingFreq; - unsigned char Channels; - unsigned char BitDepth; - } Audio; - } AV; - - /* various strings */ - char *Name; - char Language[4]; - char *CodecID; -}; - -typedef struct TrackInfo TrackInfo; - -struct SegmentInfo { - char UID[16]; - char PrevUID[16]; - char NextUID[16]; - char *Filename; - char *PrevFilename; - char *NextFilename; - char *Title; - char *MuxingApp; - char *WritingApp; - uint64_t TimecodeScale; - uint64_t Duration; - int64_t DateUTC; - char DateUTCValid; -}; - -typedef struct SegmentInfo SegmentInfo; - -struct Attachment { - uint64_t Position; - uint64_t Length; - uint64_t UID; - char *Name; - char *Description; - char *MimeType; -}; - -typedef struct Attachment Attachment; - -struct ChapterDisplay { - char *String; - char Language[4]; - char Country[4]; -}; - -struct ChapterCommand { - unsigned Time; - unsigned CommandLength; - void *Command; -}; - -struct ChapterProcess { - unsigned CodecID; - unsigned CodecPrivateLength; - void *CodecPrivate; - unsigned nCommands,nCommandsSize; - struct ChapterCommand *Commands; -}; - -struct Chapter { - uint64_t UID; - uint64_t Start; - uint64_t End; - - unsigned nTracks,nTracksSize; - uint64_t *Tracks; - unsigned nDisplay,nDisplaySize; - struct ChapterDisplay *Display; - unsigned nChildren,nChildrenSize; - struct Chapter *Children; - unsigned nProcess,nProcessSize; - struct ChapterProcess *Process; - - char SegmentUID[16]; - - unsigned int Hidden:1; - unsigned int Enabled:1; - - // Editions - unsigned int Default:1; - unsigned int Ordered:1; -}; - -typedef struct Chapter Chapter; - -#define TARGET_TRACK 0 -#define TARGET_CHAPTER 1 -#define TARGET_ATTACHMENT 2 -#define TARGET_EDITION 3 -struct Target { - uint64_t UID; - unsigned Type; -}; - -struct SimpleTag { - char *Name; - char *Value; - char Language[4]; - unsigned Default:1; -}; - -struct Tag { - unsigned nTargets,nTargetsSize; - struct Target *Targets; - - unsigned nSimpleTags,nSimpleTagsSize; - struct SimpleTag *SimpleTags; -}; - -typedef struct Tag Tag; - -/* Open a matroska file - * io pointer is recorded inside MatroskaFile - */ -X MatroskaFile *mkv_Open(/* in */ InputStream *io, - /* out */ char *err_msg, - /* in */ unsigned msgsize); - -#define MKVF_AVOID_SEEKS 1 /* use sequential reading only */ - -X MatroskaFile *mkv_OpenEx(/* in */ InputStream *io, - /* in */ uint64_t base, - /* in */ unsigned flags, - /* out */ char *err_msg, - /* in */ unsigned msgsize); - -/* Close and deallocate mf - * NULL pointer is ok and is simply ignored - */ -X void mkv_Close(/* in */ MatroskaFile *mf); - -/* Fetch the error message of the last failed operation */ -X const char *mkv_GetLastError(/* in */ MatroskaFile *mf); - -/* Get file information */ -X SegmentInfo *mkv_GetFileInfo(/* in */ MatroskaFile *mf); - -/* Get track information */ -X unsigned int mkv_GetNumTracks(/* in */ MatroskaFile *mf); -X TrackInfo *mkv_GetTrackInfo(/* in */ MatroskaFile *mf,/* in */ unsigned track); - -/* chapters, tags and attachments */ -X void mkv_GetAttachments(/* in */ MatroskaFile *mf, - /* out */ Attachment **at, - /* out */ unsigned *count); -X void mkv_GetChapters(/* in */ MatroskaFile *mf, - /* out */ Chapter **ch, - /* out */ unsigned *count); -X void mkv_GetTags(/* in */ MatroskaFile *mf, - /* out */ Tag **tag, - /* out */ unsigned *count); - -X uint64_t mkv_GetSegmentTop(MatroskaFile *mf); - -/* Seek to specified timecode, - * if timecode is past end of file, - * all tracks are set to return EOF - * on next read - */ -#define MKVF_SEEK_TO_PREV_KEYFRAME 1 -#define MKVF_SEEK_TO_PREV_KEYFRAME_STRICT 2 - -X void mkv_Seek(/* in */ MatroskaFile *mf, - /* in */ uint64_t timecode /* in ns */, - /* in */ unsigned flags); - -X void mkv_SkipToKeyframe(MatroskaFile *mf); - -X uint64_t mkv_GetLowestQTimecode(MatroskaFile *mf); - -X int mkv_TruncFloat(MKFLOAT f); - -/************************************************************************* - * reading data, pull model - */ - -/* frame flags */ -#define FRAME_UNKNOWN_START 0x00000001 -#define FRAME_UNKNOWN_END 0x00000002 -#define FRAME_KF 0x00000004 -#define FRAME_GAP 0x00800000 -#define FRAME_STREAM_MASK 0xff000000 -#define FRAME_STREAM_SHIFT 24 - -/* This sets the masking flags for the parser, - * masked tracks [with 1s in their bit positions] - * will be ignored when reading file data. - * This call discards all parsed and queued frames - */ -X void mkv_SetTrackMask(/* in */ MatroskaFile *mf,/* in */ unsigned int mask); - -/* Read one frame from the queue. - * mask specifies what tracks to ignore. - * Returns -1 if there are no more frames in the specified - * set of tracks, 0 on success - */ -X int mkv_ReadFrame(/* in */ MatroskaFile *mf, - /* in */ unsigned int mask, - /* out */ unsigned int *track, - /* out */ uint64_t *StartTime /* in ns */, - /* out */ uint64_t *EndTime /* in ns */, - /* out */ uint64_t *FilePos /* in bytes from start of file */, - /* out */ unsigned int *FrameSize /* in bytes */, - /* out */ unsigned int *FrameFlags); - -#ifdef MATROSKA_COMPRESSION_SUPPORT -/* Compressed streams support */ -struct CompressedStream; - -typedef struct CompressedStream CompressedStream; - -X CompressedStream *cs_Create(/* in */ MatroskaFile *mf, - /* in */ unsigned tracknum, - /* out */ char *errormsg, - /* in */ unsigned msgsize); -X void cs_Destroy(/* in */ CompressedStream *cs); - -/* advance to the next frame in matroska stream, you need to pass values returned - * by mkv_ReadFrame */ -X void cs_NextFrame(/* in */ CompressedStream *cs, - /* in */ uint64_t pos, - /* in */ unsigned size); - -/* read and decode more data from current frame, return number of bytes decoded, - * 0 on end of frame, or -1 on error */ -X int cs_ReadData(CompressedStream *cs,char *buffer,unsigned bufsize); - -/* return error message for the last error */ -X const char *cs_GetLastError(CompressedStream *cs); -#endif - -#ifdef __cplusplus -} -#endif - -#undef X - -#endif diff --git a/src/include/agi_pre.h b/src/include/agi_pre.h index f611c05436..da531b82d0 100644 --- a/src/include/agi_pre.h +++ b/src/include/agi_pre.h @@ -35,9 +35,6 @@ /// insert it in every source file (under C/C++ -> Advanced -> Force Includes), /// then set stdwx.cpp to generate the precompiled header /// -/// @note Make sure that you disable use of precompiled headers on md5.c and -/// MatroskaParser.c, as well as any possible future .c files. - #ifdef __cplusplus // Block msvc from complaining about not using msvc-specific versions for diff --git a/src/matroska.cpp b/src/matroska.cpp new file mode 100644 index 0000000000..d2b7e6c45b --- /dev/null +++ b/src/matroska.cpp @@ -0,0 +1,1185 @@ +// Copyright (c) 2026, Aegisub contributors +// +// Permission to use, copy, modify, and distribute this software for any +// purpose with or without fee is hereby granted, provided that the above +// copyright notice and this permission notice appear in all copies. +// +// THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES +// WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF +// MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR +// ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES +// WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN +// ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF +// OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. +// +// Aegisub Project http://www.aegisub.org/ + +/// @file matroska.cpp +/// @brief Subtitle and attachment demuxer for Matroska files +/// +/// The segment's top-level layout, Info and Tracks are read with libebml and +/// libmatroska. Clusters and attachments are walked with a small EBML reader +/// instead so that only the bytes actually needed are read: cluster contents +/// are parsed lazily one cluster at a time, and attachment payloads are only +/// located until something asks for them. + +#include "matroska.h" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace agi::matroska { +using namespace libebml; +using namespace libmatroska; + +namespace { +// Element IDs handled by the raw EBML reader +constexpr uint64_t id_cluster_timestamp = 0xE7; +constexpr uint64_t id_simple_block = 0xA3; +constexpr uint64_t id_block_group = 0xA0; +constexpr uint64_t id_block = 0xA1; +constexpr uint64_t id_block_duration = 0x9B; +constexpr uint64_t id_attached_file = 0x61A7; +constexpr uint64_t id_file_name = 0x466E; +constexpr uint64_t id_file_description = 0x467E; +constexpr uint64_t id_file_mime_type = 0x4660; +constexpr uint64_t id_file_data = 0x465C; +constexpr uint64_t id_file_uid = 0x46AE; +constexpr uint64_t id_seek_head = 0x114D9B74; +constexpr uint64_t id_seek = 0x4DBB; +constexpr uint64_t id_seek_id = 0x53AB; +constexpr uint64_t id_seek_position = 0x53AC; +constexpr uint64_t id_info = 0x1549A966; +constexpr uint64_t id_tracks = 0x1654AE6B; +constexpr uint64_t id_attachments = 0x1941A469; +constexpr uint64_t id_cluster = 0x1F43B675; +constexpr uint64_t id_cues = 0x1C53BB6B; +constexpr uint64_t id_chapters = 0x1043A770; +constexpr uint64_t id_tags = 0x1254C367; + +bool is_top_level(uint64_t id) { + switch (id) { + case id_seek_head: case id_info: case id_tracks: case id_attachments: + case id_cluster: case id_cues: case id_chapters: case id_tags: + return true; + default: + return false; + } +} + +constexpr uint64_t track_type_video = 0x1; +constexpr uint64_t track_type_audio = 0x2; +constexpr uint64_t track_type_subtitle = 0x11; + +class FileReader final : public Reader { + agi::read_file_mapping file; + +public: + explicit FileReader(agi::fs::path const& filename) : file(filename) { } + + uint64_t Size() const override { return file.size(); } + + size_t Read(uint64_t position, void *buffer, size_t size) override { + if (position >= file.size()) + return 0; + size = std::min(size, file.size() - position); + memcpy(buffer, file.read(position, size), size); + return size; + } +}; + +/// Read from a Reader, reporting any failure as an IoError +size_t read_some(Reader& reader, uint64_t position, void *buffer, size_t size) { + size_t read; + try { + read = reader.Read(position, buffer, size); + } + catch (Error const&) { + throw; + } + catch (agi::Exception const& e) { + throw IoError(e.GetMessage()); + } + catch (std::exception const& e) { + throw IoError(e.what()); + } + catch (...) { + throw IoError("Matroska input read failed"); + } + if (read > size) + throw IoError("Matroska reader returned more bytes than requested"); + return read; +} + +/// Read until either the buffer is full or the input ends +size_t read_full(Reader& reader, uint64_t position, void *buffer, size_t size) { + size_t total = 0; + while (total < size) { + size_t read = read_some(reader, position + total, static_cast(buffer) + total, size - total); + if (!read) break; + total += read; + } + return total; +} + +/// Read-only libebml stream over a Reader. Errors are thrown straight through +/// libebml, which only ever catches exceptions to clean up and rethrow. +class ReaderCallback final : public IOCallback { + Reader& reader; + uint64_t position = 0; + +public: + explicit ReaderCallback(Reader& reader) : reader(reader) { } + + uint32 read(void *buffer, size_t size) override { + size = std::min(size, std::numeric_limits::max()); + size_t read = read_full(reader, position, buffer, size); + position += read; + return static_cast(read); + } + + void setFilePointer(int64 offset, seek_mode mode) override { + int64_t base = 0; + if (mode == seek_current) + base = static_cast(position); + else if (mode == seek_end) + base = static_cast(std::min(reader.Size(), INT64_MAX)); + if ((offset > 0 && base > INT64_MAX - offset) || base + offset < 0) + throw InvalidDataError("Invalid seek in Matroska input"); + position = static_cast(base + offset); + } + + size_t write(void const *, size_t) override { + throw IoError("Matroska input is read-only"); + } + + uint64 getFilePointer() override { return position; } + void close() override { } +}; + +template +T *child(EbmlMaster& master) { + return static_cast(master.FindFirstElt(EBML_INFO(T), false)); +} + +template +uint64_t uint_value(EbmlMaster& master, uint64_t fallback = 0) { + if (auto value = child(master)) + return static_cast(*value); + return fallback; +} + +template +std::string string_value(EbmlMaster& master, std::string fallback = "") { + if (auto value = child(master)) + return value->GetValue(); + return fallback; +} + +template +std::string unicode_value(EbmlMaster& master) { + if (auto value = child(master)) + return value->GetValue().GetUTF8(); + return ""; +} + +uint64_t checked_add(uint64_t lhs, uint64_t rhs, char const *message) { + if (rhs > UINT64_MAX - lhs) + throw InvalidDataError(message); + return lhs + rhs; +} + +int64_t checked_add_signed(int64_t lhs, int64_t rhs, char const *message) { + if ((rhs > 0 && lhs > INT64_MAX - rhs) || (rhs < 0 && lhs < INT64_MIN - rhs)) + throw InvalidDataError(message); + return lhs + rhs; +} + +uint64_t checked_scale(uint64_t value, double scale, char const *message) { + if (scale == 1.0) + return value; + double scaled = static_cast(value) * scale; + if (!(scaled >= 0) || scaled >= 18446744073709551616.0) + throw InvalidDataError(message); + return static_cast(scaled); +} + +uint64_t checked_multiply(uint64_t lhs, uint64_t rhs, char const *message) { + if (lhs && rhs > UINT64_MAX / lhs) + throw InvalidDataError(message); + return lhs * rhs; +} + +/// Get a block's timestamp in nanoseconds, which is +/// (cluster timestamp + block timestamp * TrackTimestampScale) * TimestampScale +/// as per RFC 9559 section 11.2. The result may be slightly negative. +int64_t block_timestamp(uint64_t cluster_time, int16_t relative, uint64_t timestamp_scale, double track_scale) { + constexpr char const *out_of_range = "Matroska timestamp is out of range"; + if (track_scale != 1.0) { + double nanoseconds = (static_cast(cluster_time) + relative * track_scale) * static_cast(timestamp_scale); + if (!(nanoseconds > -9.2e18 && nanoseconds < 9.2e18)) + throw InvalidDataError(out_of_range); + return static_cast(nanoseconds); + } + + if (cluster_time > static_cast(INT64_MAX)) + throw InvalidDataError(out_of_range); + int64_t ticks = checked_add_signed(static_cast(cluster_time), relative, out_of_range); + uint64_t magnitude = checked_multiply(ticks < 0 ? static_cast(-ticks) : static_cast(ticks), timestamp_scale, out_of_range); + if (magnitude > static_cast(INT64_MAX)) + throw InvalidDataError(out_of_range); + return ticks < 0 ? -static_cast(magnitude) : static_cast(magnitude); +} + +SubtitleCodec codec_from_id(std::string const& id) { + if (id == "S_TEXT/ASS") + return SubtitleCodec::ass; + if (id == "S_TEXT/SSA") + return SubtitleCodec::ssa; + if (id == "S_TEXT/UTF8") + return SubtitleCodec::srt; + return SubtitleCodec::unsupported; +} + +enum class Compression { none, zlib, header_strip, unsupported }; + +/// Get the compression applied to a track's frames. zlib_codec_private is set +/// if the track's CodecPrivate is zlib-compressed. +Compression parse_encodings(KaxContentEncodings& encodings, std::vector& stripped_header, bool& zlib_codec_private) { + KaxContentEncoding *encoding = nullptr; + for (auto element : encodings.GetElementList()) { + if (auto current = dynamic_cast(element)) { + // Chained encodings aren't used by anything that writes subtitles + if (encoding) return Compression::unsupported; + encoding = current; + } + } + if (!encoding) + return Compression::none; + // Type 1 is encryption + if (uint_value(*encoding) != 0) + return Compression::unsupported; + + auto compression = child(*encoding); + if (!compression) + return Compression::none; + + // Bit 1 is the frame contents and bit 2 is CodecPrivate + auto scope = uint_value(*encoding, 1); + auto algorithm = uint_value(*compression); + if (scope & 2) { + if (algorithm != 0) + return Compression::unsupported; + zlib_codec_private = true; + } + if (!(scope & 1)) + return Compression::none; + + switch (algorithm) { + case 0: + return Compression::zlib; + case 3: + if (auto settings = child(*compression)) + stripped_header.assign(settings->GetBuffer(), settings->GetBuffer() + settings->GetSize()); + return Compression::header_strip; + default: + return Compression::unsupported; + } +} + +std::vector inflate_packet(std::vector const& input, size_t limit) { + if (input.size() > std::numeric_limits::max()) + throw LimitError("Compressed Matroska subtitle packet is too large"); + + z_stream stream{}; + if (inflateInit(&stream) != Z_OK) + throw InvalidDataError("Failed to initialize zlib"); + struct InflateEnd { + z_stream *stream; + ~InflateEnd() { inflateEnd(stream); } + } inflate_end{&stream}; + + stream.next_in = const_cast(input.data()); + stream.avail_in = static_cast(input.size()); + + std::vector output; + uint8_t buffer[4096]; + int result; + do { + stream.next_out = buffer; + stream.avail_out = sizeof buffer; + result = inflate(&stream, Z_NO_FLUSH); + if (result != Z_OK && result != Z_STREAM_END) + throw InvalidDataError("Invalid zlib-compressed Matroska subtitle packet"); + size_t produced = sizeof buffer - stream.avail_out; + if (produced > limit - output.size()) + throw LimitError("Decompressed Matroska subtitle packet exceeds configured limit"); + output.insert(output.end(), buffer, buffer + produced); + } while (result != Z_STREAM_END); + return output; +} + +/// Demuxing state for every track in the file, not just subtitle tracks +struct TrackState { + uint64_t uid = 0; + uint64_t type = 0; + uint64_t default_duration = 0; + /// Deprecated multiplier for block-relative timestamps and block durations + double timestamp_scale = 1.0; + Compression compression = Compression::none; + std::vector stripped_header; +}; + +struct Frame { + size_t track = 0; + Timestamp start; + std::optional end; + uint64_t position = 0; + uint64_t size = 0; +}; + +/// The data portion of an element in the input +struct Location { + uint64_t position = 0; + uint64_t size = 0; + /// The element was cut off by the end of the file + bool truncated = false; +}; + +/// Position within a cluster, for reading it one block at a time +struct ClusterCursor { + Location cluster; + uint64_t position; + uint64_t cluster_time = 0; + + explicit ClusterCursor(Location const& cluster) : cluster(cluster), position(cluster.position) { } +}; + +struct Element { + uint64_t id = 0; + /// Position of the element's ID + uint64_t position = 0; + uint64_t data = 0; + uint64_t size = 0; + uint64_t end = 0; + bool unknown_size = false; + bool truncated = false; +}; +} // namespace + +std::unique_ptr OpenFile(agi::fs::path const& filename) { + try { + return std::make_unique(filename); + } + catch (agi::Exception const& e) { + throw IoError(e.GetMessage()); + } +} + +class Demuxer::Impl { + std::unique_ptr reader; + CancelCheck cancelled; + Limits limits; + + uint64_t timestamp_scale = 1000000; + std::optional duration; + bool duration_computed = false; + std::optional start_time; + bool start_time_computed = false; + + /// Data of the segment, clamped to the end of the input + uint64_t segment_start = 0; + uint64_t segment_end = 0; + /// Top-level elements already parsed, by position + std::set parsed_elements; + /// Positions of the metadata elements listed by SeekHeads + std::vector seek_targets; + bool seen_info = false; + bool seen_tracks = false; + std::vector all_tracks; + std::unordered_map track_by_number; + std::vector tracks; + std::vector attachments; + /// Location of each attachment's data, parallel to attachments + std::vector attachment_data; + + /// Clusters are indexed lazily, as finding them all means reading from + /// throughout the file + std::vector clusters; + uint64_t cluster_scan_position = 0; + bool clusters_complete = true; + + /// Index into all_tracks of the selected track + std::optional selected; + size_t next_cluster = 0; + /// The cluster currently being read, if any + std::optional cluster_cursor; + /// Frames of the selected track in the most recently read block + std::vector frames; + size_t next_frame = 0; + + void CheckCancelled() const { + bool cancel; + try { + cancel = cancelled && cancelled(); + } + catch (...) { + cancel = true; + } + if (cancel) + throw agi::UserCancelException("Matroska read cancelled"); + } + + void ReadExact(uint64_t position, void *buffer, size_t size) { + if (read_full(*reader, position, buffer, size) != size) + throw TruncatedError("Unexpected end of Matroska input"); + } + + uint8_t ReadByte(uint64_t position) { + uint8_t byte; + ReadExact(position, &byte, 1); + return byte; + } + + std::vector ReadBytes(Location const& location) { + if (location.size > std::numeric_limits::max()) + throw LimitError("Matroska element exceeds addressable memory"); + std::vector result(static_cast(location.size)); + ReadExact(location.position, result.data(), result.size()); + return result; + } + + // Raw EBML reading + + /// Read a variable-length integer, returning its value and encoded length + std::pair ReadVint(uint64_t position, uint64_t end) { + if (position >= end) + throw InvalidDataError("Truncated Matroska variable-length integer"); + uint8_t first = ReadByte(position); + unsigned length = 1; + uint8_t marker = 0x80; + while (length <= 8 && !(first & marker)) { + marker >>= 1; + ++length; + } + if (length > 8) + throw InvalidDataError("Invalid Matroska variable-length integer"); + if (length > end - position) + throw InvalidDataError("Truncated Matroska variable-length integer"); + uint64_t value = first & (marker - 1); + for (unsigned i = 1; i < length; ++i) + value = (value << 8) | ReadByte(position + i); + return {value, length}; + } + + /// Read an element's header. If allow_truncated is set, an element which + /// extends past parent_end is cut off there rather than rejected. + Element ReadElement(uint64_t position, uint64_t parent_end, bool allow_truncated = false) { + uint8_t first = ReadByte(position); + unsigned id_length = 1; + for (uint8_t mask = 0x80; id_length <= 4 && !(first & mask); mask >>= 1) + ++id_length; + if (id_length > 4) + throw InvalidDataError("Invalid Matroska element ID"); + uint64_t id = first; + for (unsigned i = 1; i < id_length; ++i) + id = (id << 8) | ReadByte(position + i); + + auto [size, size_length] = ReadVint(position + id_length, parent_end); + uint64_t data = position + id_length + size_length; + // All ones is an unknown size, which extends to the end of the parent + bool unknown_size = size == (uint64_t{1} << (7 * size_length)) - 1; + if (unknown_size) + size = parent_end - data; + if (data > parent_end) + throw InvalidDataError("Matroska element exceeds its parent"); + bool truncated = size > parent_end - data; + if (truncated) { + if (!allow_truncated) + throw InvalidDataError("Matroska element exceeds its parent"); + size = parent_end - data; + } + return {id, position, data, size, data + size, unknown_size, truncated}; + } + + uint64_t ReadUInt(Element const& element) { + if (element.size > 8) + throw InvalidDataError("Invalid Matroska integer"); + uint64_t value = 0; + for (uint64_t i = 0; i < element.size; ++i) + value = (value << 8) | ReadByte(element.data + i); + return value; + } + + std::string ReadString(Element const& element) { + if (element.size > limits.metadata_size) + throw LimitError("Matroska string exceeds configured limit"); + std::string value(static_cast(element.size), '\0'); + ReadExact(element.data, value.data(), value.size()); + // Strings may be padded with trailing nulls + value.resize(strnlen(value.data(), value.size())); + return value; + } + + // Metadata + + void ParseInfo(KaxInfo& info) { + timestamp_scale = uint_value(info, 1000000); + if (!timestamp_scale) + throw InvalidDataError("Invalid Matroska timestamp scale"); + if (auto value = child(info)) { + double nanoseconds = static_cast(*value) * timestamp_scale; + if (nanoseconds > 0 && nanoseconds < static_cast(INT64_MAX)) + duration = Timestamp{static_cast(nanoseconds)}; + } + } + + void ParseTrack(KaxTrackEntry& entry) { + TrackState state; + state.uid = uint_value(entry); + // Tracks may be repeated so that a reader joining a live stream can + // see them, so skip entries which were already seen + if (state.uid && std::any_of(all_tracks.begin(), all_tracks.end(), [&](TrackState const& track) { + return track.uid == state.uid; + })) + return; + + state.type = uint_value(entry); + state.default_duration = uint_value(entry); + if (auto scale = child(entry)) { + double value = static_cast(*scale); + if (value > 0 && std::isfinite(value)) + state.timestamp_scale = value; + } + bool zlib_codec_private = false; + if (auto encodings = child(entry)) + state.compression = parse_encodings(*encodings, state.stripped_header, zlib_codec_private); + + if (!track_by_number.emplace(uint_value(entry), all_tracks.size()).second) + throw InvalidDataError("Duplicate Matroska track number"); + + if (state.type == track_type_subtitle) { + SubtitleTrack track; + track.id.value = static_cast(all_tracks.size()); + track.uid = state.uid; + track.codec_id = string_value(entry); + track.codec = state.compression == Compression::unsupported ? SubtitleCodec::unsupported : codec_from_id(track.codec_id); + track.name = unicode_value(entry); + track.language = string_value(entry, "eng"); + track.enabled = uint_value(entry, 1); + track.is_default = uint_value(entry, 1); + if (auto codec_private = child(entry)) + track.codec_private.assign(codec_private->GetBuffer(), codec_private->GetBuffer() + codec_private->GetSize()); + if (zlib_codec_private && !track.codec_private.empty()) { + try { + track.codec_private = inflate_packet(track.codec_private, limits.metadata_size); + } + catch (Error const&) { + track.codec = SubtitleCodec::unsupported; + } + } + tracks.push_back(std::move(track)); + } + all_tracks.push_back(std::move(state)); + } + + void ParseAttachments(Location const& list) { + uint64_t end = checked_add(list.position, list.size, "Matroska attachments are out of range"); + for (uint64_t position = list.position; position < end;) { + auto file = ReadElement(position, end); + position = file.end; + if (file.id != id_attached_file) + continue; + + Attachment attachment; + std::optional data; + for (uint64_t child_position = file.data; child_position < file.end;) { + auto element = ReadElement(child_position, file.end); + child_position = element.end; + switch (element.id) { + case id_file_name: attachment.name = ReadString(element); break; + case id_file_description: attachment.description = ReadString(element); break; + case id_file_mime_type: attachment.mime_type = ReadString(element); break; + case id_file_uid: attachment.uid = ReadUInt(element); break; + case id_file_data: data = Location{element.data, element.size}; break; + } + } + if (!data) + continue; + + attachment.id.value = attachments.size(); + attachment.size = data->size; + attachments.push_back(std::move(attachment)); + attachment_data.push_back(*data); + } + } + + /// Read the top-level element at position, finding where it ends if its size is unknown + Element TopLevelElement(uint64_t position) { + // The last cluster of a partially downloaded file is usually cut off, + // and the blocks in it which are complete can still be read + auto element = ReadElement(position, segment_end, true); + if (element.truncated && element.id != id_cluster) + throw TruncatedError("Unexpected end of Matroska input"); + if (element.unknown_size) { + // Only clusters are written with unknown sizes in practice. They + // end where the next top-level element starts. + for (uint64_t child_position = element.data; child_position < segment_end;) { + auto child = ReadElement(child_position, segment_end); + if (is_top_level(child.id)) { + element.end = child_position; + break; + } + child_position = child.end; + } + element.size = element.end - element.data; + } + return element; + } + + /// Read a master element with libebml + template + std::unique_ptr ReadMaster(EbmlStream& stream, Element const& element) { + if (element.size > limits.metadata_size) + throw LimitError("Matroska metadata exceeds configured limit"); + stream.I_O().setFilePointer(static_cast(element.position)); + std::unique_ptr master(stream.FindNextID(EBML_INFO(T), UINT64_MAX)); + if (!dynamic_cast(master.get())) + throw InvalidDataError("Unexpected Matroska element"); + int upper = 0; + EbmlElement *found = nullptr; + static_cast(*master).Read(stream, EBML_CONTEXT(master.get()), upper, found, true); + delete found; + return master; + } + + void ParseSeekHead(Element const& seek_head) { + if (seek_head.size > limits.metadata_size) + throw LimitError("Matroska metadata exceeds configured limit"); + for (uint64_t position = seek_head.data; position < seek_head.end;) { + auto seek = ReadElement(position, seek_head.end); + position = seek.end; + if (seek.id != id_seek) + continue; + + std::optional id, offset; + for (uint64_t child_position = seek.data; child_position < seek.end;) { + auto child = ReadElement(child_position, seek.end); + child_position = child.end; + if (child.id == id_seek_id) + id = ReadUInt(child); + else if (child.id == id_seek_position) + offset = ReadUInt(child); + } + if (!id || !offset || *offset >= segment_end - segment_start) + continue; + if (*id == id_info || *id == id_tracks || *id == id_attachments || *id == id_seek_head) + seek_targets.push_back(segment_start + *offset); + } + } + + void ParseMetadata(EbmlStream& stream, Element const& element) { + if (!is_top_level(element.id) || element.id == id_cluster) + return; + if (!parsed_elements.insert(element.position).second) + return; + + if (element.id == id_info) { + auto info = ReadMaster(stream, element); + ParseInfo(static_cast(*info)); + seen_info = true; + } + else if (element.id == id_tracks) { + auto track_list = ReadMaster(stream, element); + for (auto child : static_cast(*track_list).GetElementList()) { + if (auto entry = dynamic_cast(child)) + ParseTrack(*entry); + } + seen_tracks = true; + } + else if (element.id == id_attachments) + ParseAttachments({element.data, element.size}); + else if (element.id == id_seek_head) + ParseSeekHead(element); + } + + /// Index the next cluster, returning false at the end of the segment. If + /// a stream is given, metadata found along the way is parsed too. + bool ScanNextCluster(EbmlStream *stream = nullptr) { + while (!clusters_complete && cluster_scan_position < segment_end) { + CheckCancelled(); + Element element; + try { + element = TopLevelElement(cluster_scan_position); + } + // Treat damage after the last readable cluster as the end of the + // file, as truncated files are common + catch (InvalidDataError const&) { + break; + } + catch (TruncatedError const&) { + break; + } + cluster_scan_position = element.end; + if (element.id == id_cluster) { + clusters.push_back({element.data, element.size, element.truncated}); + return true; + } + if (stream) + ParseMetadata(*stream, element); + } + clusters_complete = true; + return false; + } + + void ParseSegment() { + ReaderCallback io(*reader); + EbmlStream stream(io); + + // FindNextID returns whatever element comes next, as an EbmlDummy if + // it isn't the requested one + std::unique_ptr head(stream.FindNextID(EBML_INFO(EbmlHead), UINT64_MAX)); + if (!dynamic_cast(head.get())) + throw InvalidDataError("EBML header not found"); + head->SkipData(stream, EBML_CONTEXT(head.get())); + + std::unique_ptr segment; + for (;;) { + segment.reset(stream.FindNextID(EBML_INFO(KaxSegment), UINT64_MAX)); + if (!segment) + throw InvalidDataError("Matroska segment not found"); + if (dynamic_cast(segment.get())) + break; + // Skip anything else at the top level, such as Void + if (!segment->IsFiniteSize()) + throw InvalidDataError("Matroska segment not found"); + segment->SkipData(stream, EBML_CONTEXT(segment.get())); + } + + segment_start = segment->GetElementPosition() + segment->HeadSize(); + segment_end = reader->Size(); + if (segment->IsFiniteSize() && segment->GetSize() < segment_end - std::min(segment_start, segment_end)) + segment_end = segment_start + segment->GetSize(); + + // Metadata is normally before the first cluster + uint64_t position = segment_start; + while (position < segment_end) { + CheckCancelled(); + auto element = TopLevelElement(position); + if (element.id == id_cluster) { + cluster_scan_position = position; + clusters_complete = false; + break; + } + ParseMetadata(stream, element); + position = element.end; + } + + // Anything after the clusters should be listed in a SeekHead. Nested + // SeekHeads append to seek_targets as it is iterated. + for (size_t i = 0; i < seek_targets.size(); ++i) { + CheckCancelled(); + try { + ParseMetadata(stream, TopLevelElement(seek_targets[i])); + } + catch (InvalidDataError const&) { } + catch (TruncatedError const&) { } + } + + // Otherwise the only way to find them is to walk the entire file + if (!seen_info || !seen_tracks) { + while (ScanNextCluster(&stream)) + ; + } + } + + bool IsAudioOrVideo(size_t track) const { + return all_tracks[track].type == track_type_video || all_tracks[track].type == track_type_audio; + } + + /// Find the earliest audio or video timestamp, which is what FFmpeg (and + /// so LAV) reports as the start time and what mpv uses for Matroska in + /// practice. Like FFmpeg, subtitle tracks don't count. + void ComputeStartTime() { + bool has_audio_or_video = false; + for (size_t track = 0; track < all_tracks.size(); ++track) + has_audio_or_video = has_audio_or_video || IsAudioOrVideo(track); + if (!has_audio_or_video) + return; + for (size_t i = 0; i < clusters.size() || ScanNextCluster(); ++i) { + CheckCancelled(); + try { + ForEachFrame(clusters[i], [&](Frame const& frame) { + if (IsAudioOrVideo(frame.track) && (!start_time || frame.start.nanoseconds < start_time->nanoseconds)) + start_time = frame.start; + }); + } + catch (InvalidDataError const&) { } + catch (TruncatedError const&) { } + if (start_time) + return; + } + } + + /// Use the end of the last frame as the duration when the header lacks one + void ComputeDuration() { + if (duration || tracks.empty()) + return; + while (ScanNextCluster()) + ; + for (auto cluster = clusters.rbegin(); cluster != clusters.rend(); ++cluster) { + CheckCancelled(); + int64_t last_end = 0; + try { + ForEachFrame(*cluster, [&](Frame const& frame) { + if (frame.end) + last_end = std::max(last_end, frame.end->nanoseconds); + }); + } + catch (InvalidDataError const&) { } + catch (TruncatedError const&) { } + if (last_end > 0) { + duration = Timestamp{last_end}; + return; + } + } + } + + // Clusters + + /// Split a block into its frames, appending those for only_track (or all tracks) to frames + void ParseBlock(Element const& block, uint64_t cluster_time, std::optional block_duration, + std::optional only_track, std::vector& out) { + auto [track_number, track_bytes] = ReadVint(block.data, block.end); + auto found = track_by_number.find(track_number); + if (found == track_by_number.end()) + return; + size_t track = found->second; + if (only_track && *only_track != track) + return; + + if (block.size < track_bytes + 3) + throw InvalidDataError("Truncated Matroska block header"); + uint64_t cursor = block.data + track_bytes; + auto relative = static_cast((ReadByte(cursor) << 8) | ReadByte(cursor + 1)); + uint8_t flags = ReadByte(cursor + 2); + cursor += 3; + + unsigned lacing = (flags >> 1) & 3; + if (lacing && cursor >= block.end) + throw InvalidDataError("Truncated Matroska lacing header"); + unsigned count = lacing ? ReadByte(cursor++) + 1 : 1; + std::vector sizes(count); + uint64_t end = block.end; + if (lacing == 1) { // Xiph + uint64_t sum = 0; + for (unsigned i = 0; i + 1 < count; ++i) { + uint64_t value = 0; + uint8_t byte; + do { + if (cursor >= end) + throw InvalidDataError("Truncated Matroska Xiph lacing"); + byte = ReadByte(cursor++); + value += byte; + } while (byte == 255); + sizes[i] = value; + sum += value; + } + if (cursor > end || sum > end - cursor) + throw InvalidDataError("Invalid Matroska Xiph lacing"); + sizes.back() = end - cursor - sum; + } + else if (lacing == 2) { // Fixed-size + if ((end - cursor) % count) + throw InvalidDataError("Invalid Matroska fixed-size lacing"); + std::fill(sizes.begin(), sizes.end(), (end - cursor) / count); + } + else if (lacing == 3) { // EBML + auto [first, first_length] = ReadVint(cursor, end); + cursor += first_length; + sizes[0] = first; + uint64_t sum = first; + for (unsigned i = 1; i + 1 < count; ++i) { + auto [encoded, length] = ReadVint(cursor, end); + cursor += length; + int64_t bias = (int64_t{1} << (7 * length - 1)) - 1; + int64_t value = static_cast(sizes[i - 1]) + static_cast(encoded) - bias; + if (value < 0 || static_cast(value) > end - cursor) + throw InvalidDataError("Invalid Matroska EBML lacing"); + sizes[i] = static_cast(value); + sum += sizes[i]; + } + if (cursor > end || sum > end - cursor) + throw InvalidDataError("Invalid Matroska EBML lacing"); + sizes.back() = end - cursor - sum; + } + else + sizes[0] = end - cursor; + + constexpr char const *out_of_range = "Matroska timestamp is out of range"; + auto const& track_state = all_tracks[track]; + int64_t block_start = block_timestamp(cluster_time, relative, timestamp_scale, track_state.timestamp_scale); + uint64_t frame_duration = block_duration + ? checked_scale(checked_multiply(*block_duration, timestamp_scale, "Matroska block duration is out of range"), + track_state.timestamp_scale, "Matroska block duration is out of range") / count + : track_state.default_duration; + + for (unsigned i = 0; i < count; ++i) { + Frame frame; + frame.track = track; + frame.position = cursor; + frame.size = sizes[i]; + uint64_t lace_offset = checked_multiply(i, frame_duration, out_of_range); + if (lace_offset > static_cast(INT64_MAX) || frame_duration > static_cast(INT64_MAX)) + throw InvalidDataError(out_of_range); + int64_t start = checked_add_signed(block_start, static_cast(lace_offset), out_of_range); + frame.start = Timestamp{start}; + if (block_duration || frame_duration) + frame.end = Timestamp{checked_add_signed(start, static_cast(frame_duration), out_of_range)}; + out.push_back(frame); + cursor = checked_add(cursor, sizes[i], "Matroska frame exceeds its block"); + } + } + + /// Read the cluster's elements up to and including its next block, + /// appending that block's frames for only_track (or all tracks) to out. + /// Returns false at the end of the cluster. Going a block at a time keeps + /// memory use bounded, as clusters have no size limit. + bool ReadNextBlock(ClusterCursor& cursor, std::optional only_track, std::vector& out) { + uint64_t end = cursor.cluster.position + cursor.cluster.size; + while (cursor.position < end) { + CheckCancelled(); + Element element; + try { + element = ReadElement(cursor.position, end); + } + catch (InvalidDataError const&) { + // Everything after the first incomplete element is missing + if (!cursor.cluster.truncated) + throw; + cursor.position = end; + return false; + } + catch (TruncatedError const&) { + if (!cursor.cluster.truncated) + throw; + cursor.position = end; + return false; + } + cursor.position = element.end; + + if (element.id == id_cluster_timestamp) + cursor.cluster_time = ReadUInt(element); + else if (element.id == id_simple_block) { + ParseBlock(element, cursor.cluster_time, std::nullopt, only_track, out); + return true; + } + else if (element.id == id_block_group) { + std::optional block; + std::optional block_duration; + for (uint64_t child_position = element.data; child_position < element.end;) { + auto child = ReadElement(child_position, element.end); + child_position = child.end; + if (child.id == id_block) + block = child; + else if (child.id == id_block_duration) + block_duration = ReadUInt(child); + } + if (block) { + ParseBlock(*block, cursor.cluster_time, block_duration, only_track, out); + return true; + } + } + } + return false; + } + + /// Call fn with each frame of every track in a cluster + template + void ForEachFrame(Location const& cluster, Fn&& fn) { + ClusterCursor cursor(cluster); + std::vector block_frames; + while (ReadNextBlock(cursor, std::nullopt, block_frames)) { + for (auto const& frame : block_frames) + fn(frame); + block_frames.clear(); + } + } + + std::vector Decode(TrackState const& track, std::vector data) const { + switch (track.compression) { + case Compression::none: + return data; + case Compression::header_strip: + if (track.stripped_header.size() > limits.decompressed_size || data.size() > limits.decompressed_size - track.stripped_header.size()) + throw LimitError("Decompressed Matroska subtitle packet exceeds configured limit"); + data.insert(data.begin(), track.stripped_header.begin(), track.stripped_header.end()); + return data; + case Compression::zlib: + return inflate_packet(data, limits.decompressed_size); + case Compression::unsupported: + break; + } + throw UnsupportedError("Unsupported Matroska content encoding"); + } + +public: + Impl(std::unique_ptr input, CancelCheck cancel, Limits configured_limits) + : reader(std::move(input)) + , cancelled(std::move(cancel)) + , limits(configured_limits) + { + if (!reader) + throw InvalidDataError("Cannot open Matroska from a null reader"); + CheckCancelled(); + try { + ParseSegment(); + } + catch (std::exception const& e) { + // Our own errors are agi::Exceptions, so this is only libebml rejecting the file + throw InvalidDataError(e.what()); + } + } + + std::vector const& Tracks() const { return tracks; } + std::vector const& AttachmentList() const { return attachments; } + std::optional Duration() { + if (!duration_computed) { + ComputeDuration(); + duration_computed = true; + } + return duration; + } + + std::optional StartTime() { + if (!start_time_computed) { + ComputeStartTime(); + start_time_computed = true; + } + return start_time; + } + + void Select(TrackId id) { + auto track = std::find_if(tracks.begin(), tracks.end(), [&](SubtitleTrack const& track) { + return track.id == id; + }); + if (track == tracks.end()) { + if (id.value >= all_tracks.size()) + throw InvalidDataError("Invalid Matroska subtitle track"); + throw InvalidDataError("Selected Matroska track is not a subtitle track"); + } + if (track->codec == SubtitleCodec::unsupported) + throw UnsupportedError("Selected Matroska subtitle codec is unsupported"); + + CheckCancelled(); + selected = id.value; + next_cluster = 0; + cluster_cursor.reset(); + frames.clear(); + next_frame = 0; + } + + std::optional NextPacket() { + if (!selected) + throw InvalidDataError("No Matroska subtitle track has been selected"); + CheckCancelled(); + + while (next_frame == frames.size()) { + frames.clear(); + next_frame = 0; + if (!cluster_cursor) { + if (next_cluster == clusters.size() && !ScanNextCluster()) + return std::nullopt; + cluster_cursor.emplace(clusters[next_cluster++]); + } + bool more; + try { + more = ReadNextBlock(*cluster_cursor, selected, frames); + } + catch (...) { + // Skip the rest of a cluster which fails to parse, so that + // retrying moves on to the next one + cluster_cursor.reset(); + frames.clear(); + throw; + } + if (!more) + cluster_cursor.reset(); + } + + auto const& frame = frames[next_frame++]; + if (frame.size > limits.packet_size) + throw LimitError("Matroska subtitle packet exceeds configured limit"); + auto data = Decode(all_tracks[*selected], ReadBytes({frame.position, frame.size})); + return SubtitlePacket{{static_cast(*selected)}, frame.start, frame.end, std::move(data)}; + } + + std::vector AttachmentBytes(AttachmentId id) { + if (id.value >= attachments.size()) + throw InvalidDataError("Unknown Matroska attachment"); + auto const& data = attachment_data[id.value]; + if (data.size > limits.attachment_size) + throw LimitError("Matroska attachment exceeds configured limit"); + CheckCancelled(); + return ReadBytes(data); + } +}; + +Demuxer::Demuxer(std::unique_ptr reader, CancelCheck cancelled, Limits limits) +: impl(std::make_unique(std::move(reader), std::move(cancelled), limits)) +{ +} + +Demuxer::~Demuxer() = default; +Demuxer::Demuxer(Demuxer&&) noexcept = default; +Demuxer& Demuxer::operator=(Demuxer&&) noexcept = default; + +std::vector const& Demuxer::SubtitleTracks() const { + return impl->Tracks(); +} + +std::vector const& Demuxer::Attachments() const { + return impl->AttachmentList(); +} + +std::optional Demuxer::Duration() { + return impl->Duration(); +} + +std::optional Demuxer::StartTime() { + return impl->StartTime(); +} + +void Demuxer::SelectTrack(TrackId track) { + impl->Select(track); +} + +std::optional Demuxer::ReadPacket() { + return impl->NextPacket(); +} + +std::vector Demuxer::ReadAttachment(AttachmentId id) { + return impl->AttachmentBytes(id); +} + +} // namespace agi::matroska diff --git a/src/matroska.h b/src/matroska.h new file mode 100644 index 0000000000..83b8b278cb --- /dev/null +++ b/src/matroska.h @@ -0,0 +1,153 @@ +// Copyright (c) 2026, Aegisub contributors +// +// Permission to use, copy, modify, and distribute this software for any +// purpose with or without fee is hereby granted, provided that the above +// copyright notice and this permission notice appear in all copies. +// +// THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES +// WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF +// MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR +// ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES +// WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN +// ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF +// OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. +// +// Aegisub Project http://www.aegisub.org/ + +/// @file matroska.h +/// @brief Subtitle and attachment demuxer for Matroska files + +#pragma once + +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace agi::matroska { + +DEFINE_EXCEPTION(Error, agi::Exception); +DEFINE_EXCEPTION(IoError, Error); +DEFINE_EXCEPTION(TruncatedError, Error); +DEFINE_EXCEPTION(InvalidDataError, Error); +DEFINE_EXCEPTION(UnsupportedError, Error); +DEFINE_EXCEPTION(LimitError, Error); + +struct Timestamp { + int64_t nanoseconds = 0; +}; + +struct TrackId { + /// Index of the track among all tracks in the file, including non-subtitle tracks + uint32_t value = 0; + friend bool operator==(TrackId lhs, TrackId rhs) { return lhs.value == rhs.value; } +}; + +struct AttachmentId { + /// Index of the attachment in Demuxer::Attachments() + uint64_t value = 0; + friend bool operator==(AttachmentId lhs, AttachmentId rhs) { return lhs.value == rhs.value; } +}; + +enum class SubtitleCodec { ass, ssa, srt, unsupported }; + +struct SubtitleTrack { + TrackId id; + uint64_t uid = 0; + /// unsupported if either the codec or the track's content encoding is unsupported + SubtitleCodec codec = SubtitleCodec::unsupported; + std::string codec_id; + std::string name; + std::string language; + std::vector codec_private; + bool enabled = true; + bool is_default = false; +}; + +struct Attachment { + AttachmentId id; + uint64_t uid = 0; + std::string name; + std::string description; + std::string mime_type; + uint64_t size = 0; +}; + +struct SubtitlePacket { + TrackId track; + /// Timestamps as stored in the file, without subtracting StartTime() + Timestamp start; + std::optional end; + std::vector data; +}; + +/// Polled during parsing; returning true or throwing aborts with agi::UserCancelException +using CancelCheck = std::function; + +struct Limits { + /// Maximum size of a single stored subtitle packet + size_t packet_size = 16 * 1024 * 1024; + /// Maximum size of a single subtitle packet after decompression + size_t decompressed_size = 16 * 1024 * 1024; + /// Maximum size of an attachment returned by ReadAttachment + size_t attachment_size = 256 * 1024 * 1024; + /// Maximum size of each top-level metadata element other than attachments + size_t metadata_size = 16 * 1024 * 1024; +}; + +/// Random-access input owned by Demuxer for its entire lifetime. +class Reader { +public: + virtual ~Reader() = default; + virtual uint64_t Size() const = 0; + /// Read up to size bytes at position. A short read is permitted. Returning zero at + /// Size() means EOF; returning zero before Size() is treated as truncation. + virtual size_t Read(uint64_t position, void *buffer, size_t size) = 0; +}; + +/// Open a file-backed reader using Aegisub's mapped-file input. +std::unique_ptr OpenFile(agi::fs::path const& filename); + +/// The demuxer owns its Reader and all parser state. Returned metadata and +/// packet bytes are values, so none of them borrow storage from the parser. +class Demuxer final { + class Impl; + std::unique_ptr impl; + +public: + explicit Demuxer(std::unique_ptr reader, CancelCheck cancelled = {}, Limits limits = {}); + ~Demuxer(); + + Demuxer(Demuxer&&) noexcept; + Demuxer& operator=(Demuxer&&) noexcept; + Demuxer(Demuxer const&) = delete; + Demuxer& operator=(Demuxer const&) = delete; + + std::vector const& SubtitleTracks() const; + std::vector const& Attachments() const; + /// Segment duration, or nullopt if neither the header nor the last cluster + /// gives one. If the header lacks one, the first call has to find the last + /// cluster, which reads from throughout the file and can be cancelled. + std::optional Duration(); + /// Earliest audio or video timestamp, which players treat as the start of + /// the file, or nullopt if the file has no audio or video. The first call + /// reads up to the first cluster with audio or video and can be cancelled. + std::optional StartTime(); + + /// Select one subtitle track and rewind packet iteration to its beginning. + void SelectTrack(TrackId track); + /// Return the next decoded packet, or nullopt at end of input. On error or + /// cancellation the failed packet (or cluster) is consumed; callers may + /// retry to read the next packet. + std::optional ReadPacket(); + /// Copy an attachment from the input. The returned vector owns its bytes. + std::vector ReadAttachment(AttachmentId id); +}; + +} // namespace agi::matroska diff --git a/src/meson.build b/src/meson.build index 876d551ffb..74154d0326 100644 --- a/src/meson.build +++ b/src/meson.build @@ -1,7 +1,23 @@ subdir('libresrc') +aegisub_src_inc = include_directories('.') + +matroska_lib = static_library('aegisub-matroska', + 'matroska.cpp', + include_directories: libaegisub_inc, + dependencies: [boost_dep, zlib_dep, libebml, libmatroska], + # matroska.cpp brings the libebml/libmatroska namespaces into scope, which + # would leak into other files in a unity build. + override_options: ['unity=off'], +) + +deps += declare_dependency( + link_with: matroska_lib, + include_directories: aegisub_src_inc, + dependencies: [zlib_dep, libebml, libmatroska], +) + aegisub_src = files( - 'MatroskaParser.c', 'aegisublocale.cpp', 'ass_attachment.cpp', 'ass_dialogue.cpp', diff --git a/src/mkv_wrap.cpp b/src/mkv_wrap.cpp index ba56834951..48982d5cea 100644 --- a/src/mkv_wrap.cpp +++ b/src/mkv_wrap.cpp @@ -38,150 +38,66 @@ #include "ass_parser.h" #include "compat.h" #include "dialog_progress.h" -#include "MatroskaParser.h" +#include "matroska.h" #include "options.h" #include "subtitle_format_srt.h" #include -#include #include -#include #include #include #include -#include #include -#include +#include +#include +#include +#include #include // Keep this last so wxUSE_CHOICEDLG is set. -struct MkvStdIO final : InputStream { - agi::read_file_mapping file; - std::string error; +namespace { +namespace mkv = agi::matroska; - static int Read(InputStream *st, uint64_t pos, void *buffer, int count) { - auto *self = static_cast(st); - if (pos >= self->file.size()) - return 0; +/// Limit on the total size of the subtitle data read from a file +constexpr size_t max_total_subtitle_bytes = 64 * 1024 * 1024; - auto remaining = self->file.size() - pos; - if (remaining < INT_MAX) - count = std::min(static_cast(remaining), count); - - if (count <= 0) - return 0; - - try { - auto data = self->file.read(pos, count); - if (buffer) - memcpy(buffer, data, count); - } - catch (agi::Exception const& e) { - self->error = e.GetMessage(); - return -1; - } - - return count; - } - - static int64_t Scan(InputStream *st, uint64_t start, unsigned signature) { - auto *self = static_cast(st); - try { - unsigned cmp = 0; - for (auto i : boost::irange(start, self->file.size())) { - int c = *self->file.read(i, 1); - cmp = ((cmp << 8) | c) & 0xffffffff; - if (cmp == signature) - return i - 4; - } - } - catch (agi::Exception const& e) { - self->error = e.GetMessage(); - } - - return -1; - } - - static int64_t Size(InputStream *st) { - return static_cast(st)->file.size(); - } - - MkvStdIO(agi::fs::path const& filename) : file(filename) { - read = &MkvStdIO::Read; - scan = &MkvStdIO::Scan; - getcachesize = [](InputStream *) -> unsigned int { return 16 * 1024 * 1024; }; - geterror = [](InputStream *st) -> const char * { return ((MkvStdIO *)st)->error.c_str(); }; - memalloc = [](InputStream *, size_t size) { return malloc(size); }; - memrealloc = [](InputStream *, void *mem, size_t size) { return realloc(mem, size); }; - memfree = [](InputStream *, void *mem) { free(mem); }; - progress = [](InputStream *, uint64_t, uint64_t) { return 1; }; - getfilesize = &MkvStdIO::Size; - } -}; +/// Get a time in milliseconds relative to origin, which may be negative +int64_t relative_ms(mkv::Timestamp time, int64_t origin) { + if ((origin < 0 && time.nanoseconds > INT64_MAX + origin) || (origin > 0 && time.nanoseconds < INT64_MIN + origin)) + throw MatroskaException("Matroska subtitle timestamp is out of range"); + return (time.nanoseconds - origin) / 1000000; +} -static constexpr ptrdiff_t max_decompressed_subtitle_frame_bytes = 16 * 1024 * 1024; -static constexpr ptrdiff_t max_total_subtitle_bytes = 64 * 1024 * 1024; +agi::Time to_ass_time(int64_t milliseconds) { + if (milliseconds > std::numeric_limits::max()) + throw MatroskaException("Matroska subtitle timestamp is out of range"); + return static_cast(milliseconds); +} -static bool read_subtitles(agi::ProgressSink *ps, MatroskaFile *file, MkvStdIO *input, bool srt, double totalTime, AssParser *parser, CompressedStream *cs) { +/// Read the selected track's events. origin is the time in the file which +/// becomes time zero. +void read_subtitles(agi::ProgressSink *ps, mkv::Demuxer& demuxer, bool srt, int64_t origin, int64_t total_time, AssParser *parser) { std::vector> subList; - ptrdiff_t totalSubtitleBytes = 0; - - // Load blocks - uint64_t startTime, endTime, filePos; - unsigned int rt, frameSize, frameFlags; - - std::vector uncompBuf(cs ? 256 : 0); - + size_t total_bytes = 0; SrtTagParser srtParser; - while (mkv_ReadFrame(file, 0, &rt, &startTime, &endTime, &filePos, &frameSize, &frameFlags) == 0) { - if (ps->IsCancelled()) return true; - if (frameSize == 0) continue; - - std::string_view readBuf; - - if (cs) { - cs_NextFrame(cs, filePos, frameSize); - ptrdiff_t bytesRead = 0; - - while (true) { - if (bytesRead >= std::ssize(uncompBuf)) { - if (uncompBuf.size() >= max_decompressed_subtitle_frame_bytes) { - ps->Log(agi::format("Decompressed subtitle frame exceeds the %d MiB limit", max_decompressed_subtitle_frame_bytes / 1024 / 1024)); - return false; - } - uncompBuf.resize(std::min(std::ssize(uncompBuf) * 2, max_decompressed_subtitle_frame_bytes)); - } - - int res = cs_ReadData(cs, uncompBuf.data() + bytesRead, static_cast(std::ssize(uncompBuf) - bytesRead)); - if (res < 0) { - const char *err = cs_GetLastError(cs); - if (!err) err = "Unknown error"; - ps->Log("Failed to decompress subtitles: " + std::string(err)); - return false; - } - - bytesRead += res; - if (res == 0) - break; - } - - readBuf = std::string_view(uncompBuf.data(), bytesRead); - } else { - readBuf = std::string_view(input->file.read(filePos, frameSize), frameSize); - } + while (auto packet = demuxer.ReadPacket()) { + if (ps->IsCancelled()) return; + if (packet->data.empty()) continue; - if (std::ssize(readBuf) > max_total_subtitle_bytes - totalSubtitleBytes) { - ps->Log(agi::format("Matroska subtitle data exceeds the %d MiB limit", max_total_subtitle_bytes / 1024 / 1024)); - return false; - } - totalSubtitleBytes += std::ssize(readBuf); + if (packet->data.size() > max_total_subtitle_bytes - total_bytes) + throw MatroskaException(agi::format("Matroska subtitle data exceeds the %d MiB limit", max_total_subtitle_bytes / 1024 / 1024)); + total_bytes += packet->data.size(); - // Get start and end times - int64_t timecodeScaleLow = 1000000; - agi::Time subStart = startTime / timecodeScaleLow; - agi::Time subEnd = endTime / timecodeScaleLow; + int64_t start = relative_ms(packet->start, origin); + int64_t end = packet->end ? relative_ms(*packet->end, origin) : start; + // Lines entirely before the start of the file can't be shown + if (end < 0 || (end == 0 && start < 0)) + continue; + agi::Time subStart = to_ass_time(std::max(start, 0)); + agi::Time subEnd = to_ass_time(end); + std::string_view readBuf(reinterpret_cast(packet->data.data()), packet->data.size()); // Process SSA/ASS if (!srt) { @@ -190,13 +106,18 @@ static bool read_subtitles(agi::ProgressSink *ps, MatroskaFile *file, MkvStdIO * auto second = readBuf.find(',', first + 1); if (second == readBuf.npos) continue; - subList.emplace_back( - boost::lexical_cast(readBuf.substr(0, first)), - agi::format("Dialogue: %d,%s,%s,%s" - , boost::lexical_cast(readBuf.substr(first + 1, second - (first + 1))) - , subStart.GetAssFormatted() - , subEnd.GetAssFormatted() - , readBuf.substr(second + 1))); + try { + subList.emplace_back( + boost::lexical_cast(readBuf.substr(0, first)), + agi::format("Dialogue: %d,%s,%s,%s" + , boost::lexical_cast(readBuf.substr(first + 1, second - (first + 1))) + , subStart.GetAssFormatted() + , subEnd.GetAssFormatted() + , readBuf.substr(second + 1))); + } + catch (boost::bad_lexical_cast const&) { + throw MatroskaException("Malformed ASS packet in Matroska subtitle track"); + } } // Process SRT else { @@ -211,41 +132,58 @@ static bool read_subtitles(agi::ProgressSink *ps, MatroskaFile *file, MkvStdIO * subList.emplace_back(subList.size(), std::move(line)); } - ps->SetProgress(startTime / timecodeScaleLow, totalTime); + if (total_time > 0) + ps->SetProgress(subStart, total_time); } // Insert into file sort(begin(subList), end(subList)); - for (auto order_value_pair : subList) + for (auto const& order_value_pair : subList) parser->AddLine(order_value_pair.second); - return true; +} + +/// Run a task in a progress dialog, rethrowing any error it fails with +/// rather than just logging it to the dialog +void run_with_progress(wxString const& message, agi::ProgressSink *&active_sink, std::function task) { + DialogProgress progress(nullptr, _("Parsing Matroska"), message); + std::exception_ptr failure; + progress.Run([&](agi::ProgressSink *ps) { + active_sink = ps; + try { + task(ps); + } + catch (...) { + failure = std::current_exception(); + } + // The sink is destroyed when Run returns + active_sink = nullptr; + }); + if (failure) + std::rethrow_exception(failure); +} } void MatroskaWrapper::GetSubtitles(agi::fs::path const& filename, AssFile *target) { - MkvStdIO input(filename); - char err[2048]; - agi::scoped_holder file(mkv_Open(&input, err, sizeof(err)), mkv_Close); - if (!file) throw MatroskaException(err); - - // Get info - unsigned tracks = mkv_GetNumTracks(file); - std::vector tracksFound; - std::vector tracksNames; + // The demuxer polls for cancellation from whichever progress dialog is + // currently running it + agi::ProgressSink *active_sink = nullptr; + auto cancelled = [&] { return active_sink && active_sink->IsCancelled(); }; + + std::optional demuxer; + run_with_progress(_("Reading Matroska track information."), active_sink, [&](agi::ProgressSink *) { + demuxer.emplace(mkv::OpenFile(filename), cancelled); + }); // Find tracks - for (auto track : boost::irange(0u, tracks)) { - auto trackInfo = mkv_GetTrackInfo(file, track); - if (trackInfo->Type != 0x11) continue; - - // Known subtitle format - std::string CodecID(trackInfo->CodecID); - if (CodecID == "S_TEXT/SSA" || CodecID == "S_TEXT/ASS" || CodecID == "S_TEXT/UTF8") { - tracksFound.push_back(track); - tracksNames.emplace_back(agi::format("%d (%s %s)", track, CodecID, trackInfo->Language)); - if (trackInfo->Name) { - tracksNames.back() += ": "; - tracksNames.back() += trackInfo->Name; - } + std::vector tracksFound; + std::vector tracksNames; + for (auto const& track : demuxer->SubtitleTracks()) { + if (track.codec == mkv::SubtitleCodec::unsupported) continue; + tracksFound.push_back(&track); + tracksNames.emplace_back(agi::format("%d (%s %s)", track.id.value, track.codec_id, track.language)); + if (!track.name.empty()) { + tracksNames.back() += ": "; + tracksNames.back() += track.name; } } @@ -253,7 +191,7 @@ void MatroskaWrapper::GetSubtitles(agi::fs::path const& filename, AssFile *targe if (tracksFound.empty()) throw MatroskaException("File has no recognised subtitle tracks."); - unsigned trackToRead; + mkv::SubtitleTrack const *trackToRead; // Only one track found if (tracksFound.size() == 1) trackToRead = tracksFound[0]; @@ -267,18 +205,18 @@ void MatroskaWrapper::GetSubtitles(agi::fs::path const& filename, AssFile *targe } // Picked track - mkv_SetTrackMask(file, ~(1 << trackToRead)); - auto trackInfo = mkv_GetTrackInfo(file, trackToRead); - std::string CodecID(trackInfo->CodecID); - bool srt = CodecID == "S_TEXT/UTF8"; - bool ssa = CodecID == "S_TEXT/SSA"; + demuxer->SelectTrack(trackToRead->id); + bool srt = trackToRead->codec == mkv::SubtitleCodec::srt; + bool ssa = trackToRead->codec == mkv::SubtitleCodec::ssa; - AssParser parser(target, !ssa); + // Parse into a temporary file so a failure partway through leaves the target untouched + AssFile imported; + AssParser parser(&imported, !ssa); // Read private data if it's ASS/SSA if (!srt) { // Read raw data - std::string priv((const char *)trackInfo->CodecPrivate, trackInfo->CodecPrivateSize); + std::string priv(trackToRead->codec_private.begin(), trackToRead->codec_private.end()); // Load into file boost::char_separator sep("\r\n"); @@ -287,49 +225,37 @@ void MatroskaWrapper::GetSubtitles(agi::fs::path const& filename, AssFile *targe } // Load default if it's SRT else - target->LoadDefault(false, OPT_GET("Subtitle Format/SRT/Default Style Catalog")->GetString()); + imported.LoadDefault(false, OPT_GET("Subtitle Format/SRT/Default Style Catalog")->GetString()); parser.AddLine("[Events]"); - agi::scoped_holder cs(nullptr, cs_Destroy); - if (trackInfo->CompEnabled) { - cs = cs_Create(file, trackToRead, err, sizeof(err)); - if (!cs) - throw MatroskaException(err); - } - - // Read timecode scale - auto segInfo = mkv_GetFileInfo(file); - int64_t timecodeScale = mkv_TruncFloat(trackInfo->TimecodeScale) * segInfo->TimecodeScale; - - // Progress bar - auto totalTime = double(segInfo->Duration) / timecodeScale; - DialogProgress progress(nullptr, _("Parsing Matroska"), _("Reading subtitles from Matroska file.")); - bool result; - progress.Run([&](agi::ProgressSink *ps) { result = read_subtitles(ps, file, &input, srt, totalTime, &parser, cs); }); - - if (!result) - throw MatroskaException("Failed to read subtitles"); + run_with_progress(_("Reading subtitles from Matroska file."), active_sink, [&](agi::ProgressSink *ps) { + // Players such as mpv and MPC-HC treat the earliest audio or video + // timestamp as the start of the file, so make that time zero. Files + // with only subtitles are never played by themselves, so keep their + // timestamps as-is. Aegisub's video timeline instead makes the first + // video frame time zero (TypesettingTools/Aegisub#21), so in files + // where video starts after audio, imported lines currently appear late + // by that delay, but are exported with their original timing. + // + // These may need to read through the file, so are done in the progress dialog. + auto file_start = demuxer->StartTime(); + int64_t origin = file_start ? file_start->nanoseconds : 0; + auto duration = demuxer->Duration(); + int64_t totalTime = duration ? duration->nanoseconds / 1000000 : 0; + read_subtitles(ps, *demuxer, srt, origin, totalTime, &parser); + }); + + target->swap(imported); } bool MatroskaWrapper::HasSubtitles(agi::fs::path const& filename) { - char err[2048]; try { - MkvStdIO input(filename); - agi::scoped_holder file(mkv_Open(&input, err, sizeof(err)), mkv_Close); - if (!file) return false; - - // Find tracks - auto tracks = mkv_GetNumTracks(file); - for (auto track : boost::irange(0u, tracks)) { - auto trackInfo = mkv_GetTrackInfo(file, track); - - if (trackInfo->Type == 0x11) { - std::string CodecID(trackInfo->CodecID); - if (CodecID == "S_TEXT/SSA" || CodecID == "S_TEXT/ASS" || CodecID == "S_TEXT/UTF8") - return true; - } - } + mkv::Demuxer demuxer(mkv::OpenFile(filename)); + auto const& tracks = demuxer.SubtitleTracks(); + return std::any_of(tracks.begin(), tracks.end(), [](mkv::SubtitleTrack const& track) { + return track.codec != mkv::SubtitleCodec::unsupported; + }); } catch (...) { // We don't care about why we couldn't read subtitles here diff --git a/subprojects/libebml.wrap b/subprojects/libebml.wrap new file mode 100644 index 0000000000..1f78d2cf44 --- /dev/null +++ b/subprojects/libebml.wrap @@ -0,0 +1,13 @@ +[wrap-file] +directory = libebml-release-1.4.5 +source_url = https://github.com/Matroska-Org/libebml/archive/refs/tags/release-1.4.5.tar.gz +source_filename = libebml-release-1.4.5.tar.gz +source_hash = 86c99573cd0957884f26547d1a8fa0c979e4d6d57484dfd387345846e6720f49 +patch_filename = libebml_1.4.5-3_patch.zip +patch_url = https://wrapdb.mesonbuild.com/v2/libebml_1.4.5-3/get_patch +patch_hash = d8a188b12e288c49fe7f738e135cae86b3c09451b4de9ad4ee2ac56478e7640c +source_fallback_url = https://github.com/mesonbuild/wrapdb/releases/download/libebml_1.4.5-3/libebml-release-1.4.5.tar.gz +wrapdb_version = 1.4.5-3 + +[provide] +libebml = libebml_dep diff --git a/subprojects/libmatroska.wrap b/subprojects/libmatroska.wrap new file mode 100644 index 0000000000..3ab3c27e53 --- /dev/null +++ b/subprojects/libmatroska.wrap @@ -0,0 +1,13 @@ +[wrap-file] +directory = libmatroska-release-1.7.1 +source_url = https://github.com/Matroska-Org/libmatroska/archive/refs/tags/release-1.7.1.tar.gz +source_filename = libmatroska-release-1.7.1.tar.gz +source_hash = 64763443947833e6c17f1f555f4bb0df6c9f91881810d9d5e0f0bad3622d308b +patch_filename = libmatroska_1.7.1-5_patch.zip +patch_url = https://wrapdb.mesonbuild.com/v2/libmatroska_1.7.1-5/get_patch +patch_hash = a4bf967bf7312f0129d4ce4b000a8f3fe254dce1e74eb42bd0d3dbdafb28e136 +source_fallback_url = https://github.com/mesonbuild/wrapdb/releases/download/libmatroska_1.7.1-5/libmatroska-release-1.7.1.tar.gz +wrapdb_version = 1.7.1-5 + +[provide] +libmatroska = libmatroska_dep diff --git a/tests/matroska/fixtures/attachment.txt b/tests/matroska/fixtures/attachment.txt new file mode 100644 index 0000000000..ef53deafcb --- /dev/null +++ b/tests/matroska/fixtures/attachment.txt @@ -0,0 +1 @@ +Aegisub deterministic attachment fixture. diff --git a/tests/matroska/fixtures/audio-only-opus.behavior b/tests/matroska/fixtures/audio-only-opus.behavior new file mode 100644 index 0000000000..f5235479c1 --- /dev/null +++ b/tests/matroska/fixtures/audio-only-opus.behavior @@ -0,0 +1,2 @@ +duration 256504416 +start 0 diff --git a/tests/matroska/fixtures/audio-only-opus.mka b/tests/matroska/fixtures/audio-only-opus.mka new file mode 100644 index 0000000000..f308e65847 Binary files /dev/null and b/tests/matroska/fixtures/audio-only-opus.mka differ diff --git a/tests/matroska/fixtures/compressed-zlib.behavior b/tests/matroska/fixtures/compressed-zlib.behavior new file mode 100644 index 0000000000..6ba2447a4e --- /dev/null +++ b/tests/matroska/fixtures/compressed-zlib.behavior @@ -0,0 +1,4 @@ +duration 2000000000 +start - +track 0 2756463499428226085 S_TEXT/UTF8 und 0 1 1 +packet 0 500000000 2500000000 24 da1dfbf2ee283ed2 diff --git a/tests/matroska/fixtures/compressed-zlib.mkv b/tests/matroska/fixtures/compressed-zlib.mkv new file mode 100644 index 0000000000..a70b13b2af Binary files /dev/null and b/tests/matroska/fixtures/compressed-zlib.mkv differ diff --git a/tests/matroska/fixtures/encodings.behavior b/tests/matroska/fixtures/encodings.behavior new file mode 100644 index 0000000000..2ae5c0d284 --- /dev/null +++ b/tests/matroska/fixtures/encodings.behavior @@ -0,0 +1,5 @@ +duration 1500000000 +start - +track 0 1 S_TEXT/UTF8 eng 0 1 1 +track 1 2 S_TEXT/UTF8 eng 0 1 1 +packet 0 0 1500000000 5 5a0d15131ec7a1 diff --git a/tests/matroska/fixtures/encodings.mkv b/tests/matroska/fixtures/encodings.mkv new file mode 100644 index 0000000000..0e76a0ea08 Binary files /dev/null and b/tests/matroska/fixtures/encodings.mkv differ diff --git a/tests/matroska/fixtures/generate.py b/tests/matroska/fixtures/generate.py new file mode 100755 index 0000000000..1700ad9b4c --- /dev/null +++ b/tests/matroska/fixtures/generate.py @@ -0,0 +1,110 @@ +#!/usr/bin/env python3 +"""Regenerate the small, deterministic project-owned parity fixture.""" +import pathlib,subprocess +r=pathlib.Path(__file__).parent +subprocess.run(['ffmpeg','-hide_banner','-loglevel','error','-y','-fflags','+bitexact','-f','lavfi','-i','color=size=16x16:rate=1:duration=3','-i',r/'subtitle.ass','-i',r/'utf8.srt','-map','0:v','-map','1','-map','2','-c:v','ffv1','-c:s:0','ass','-c:s:1','srt','-metadata:s:s:0','language=eng','-metadata:s:s:0','title=ASS track','-metadata:s:s:1','language=jpn','-attach',r/'attachment.txt','-metadata:s:t','mimetype=text/plain','-metadata:s:t','filename=attachment.txt','-map_metadata','-1',r/'subtitle-attachment.mkv'],check=True) +subprocess.run(['mkvmerge','-o',r/'compressed-zlib.mkv','--compression','0:zlib',r/'utf8.srt'],check=True) +opus=r/'audio-only.opus' +subprocess.run(['ffmpeg','-hide_banner','-loglevel','error','-y','-fflags','+bitexact','-f','lavfi','-i','sine=frequency=440:duration=0.25','-c:a','libopus',opus],check=True) +subprocess.run(['mkvmerge','-o',r/'audio-only-opus.mka',opus],check=True) +opus.unlink() +subprocess.run(['ffmpeg','-hide_banner','-loglevel','error','-y','-fflags','+bitexact','-f','lavfi','-i','color=size=16x16:rate=1:duration=1','-map_metadata','-1','-c:v','ffv1',r/'video-only.mkv'],check=True) + +def element(element_id, payload): + length=len(payload) + size_length=next(n for n in range(1,9) if length < (1 << (7*n))-1) + encoded=(length | (1 << (7*size_length))).to_bytes(size_length,'big') + return bytes.fromhex(element_id)+encoded+payload +def uint_element(element_id, value): + size=max(1,(value.bit_length()+7)//8) + return element(element_id,value.to_bytes(size,'big')) + +source=(r/'video-only.mkv').read_bytes() +header_size=4+(source[4]&0x7f)+1 +header=source[:header_size] +info=element('1549a966',uint_element('2ad7b1',1000000)) +block=element('a3',b'\x41\x01\x00\x00\x80hello') +cluster=element('1f43b675',uint_element('e7',0)+block) +entry=element('ae',uint_element('d7',257)+uint_element('73c5',257)+uint_element('83',17)+element('86',b'S_TEXT/UTF8')) +tracks=element('1654ae6b',entry) +seek_entry=element('4dbb',element('53ab',bytes.fromhex('1654ae6b'))+uint_element('53ac',0)) +seek_head=element('114d9b74',seek_entry) +tracks_position=len(seek_head)+len(info)+len(cluster) +seek_entry=element('4dbb',element('53ab',bytes.fromhex('1654ae6b'))+uint_element('53ac',tracks_position)) +seek_head=element('114d9b74',seek_entry) +segment_payload=seek_head+info+cluster+tracks +(r/'tracks-after-cluster.mkv').write_bytes(header+element('18538067',segment_payload)) + +# Content encodings: a header-stripped track read via a BlockGroup with a +# duration, and an encrypted track which should be reported as unsupported +# without preventing the rest of the file from being read. +def encoding(*children): + return element('6d80',element('6240',b''.join(children))) +stripped=element('ae',uint_element('d7',1)+uint_element('73c5',1)+uint_element('83',17)+element('86',b'S_TEXT/UTF8') + +encoding(uint_element('5033',0)+element('5034',uint_element('4254',3)+element('4255',b'hel')))) +encrypted=element('ae',uint_element('d7',2)+uint_element('73c5',2)+uint_element('83',17)+element('86',b'S_TEXT/UTF8') + +encoding(uint_element('5033',1)+element('5035',b''))) +group=element('a0',element('a1',b'\x81\x00\x00\x00lo')+uint_element('9b',1500)) +encrypted_block=element('a3',b'\x82\x00\x00\x80xx') +cluster=element('1f43b675',uint_element('e7',0)+group+encrypted_block) +segment_payload=info+element('1654ae6b',stripped+encrypted)+cluster +(r/'encodings.mkv').write_bytes(header+element('18538067',segment_payload)) + +# Tracks repeated as live muxers do, a zlib-compressed CodecPrivate with +# uncompressed frames (scope 2), and attachments without UIDs or data +import zlib +entry=element('ae',uint_element('d7',1)+uint_element('73c5',7)+uint_element('83',17)+element('86',b'S_TEXT/UTF8') + +element('63a2',zlib.compress(b'private data')) + +encoding(uint_element('5032',2)+uint_element('5033',0)+element('5034',uint_element('4254',0)))) +tracks=element('1654ae6b',entry) +def attached(name,data=None): + return element('61a7',element('466e',name)+element('4660',b'text/plain')+(element('465c',data) if data is not None else b'')) +attachments=element('1941a469',attached(b'a.txt',b'aaa')+attached(b'b.txt',b'bb')+attached(b'c.txt')) +cluster=element('1f43b675',uint_element('e7',0)+element('a3',b'\x81\x00\x00\x80hi')) +(r/'repeated-tracks.mkv').write_bytes(header+element('18538067',info+tracks+tracks+attachments+cluster)) + +# A block timestamp which overflows int64 once the block's relative timestamp +# is added to the cluster's +entry=element('ae',uint_element('d7',1)+uint_element('73c5',1)+uint_element('83',17)+element('86',b'S_TEXT/UTF8')) +cluster=element('1f43b675',uint_element('e7',(1<<63)-1)+element('a3',b'\x81\x00\x01\x80x')) +(r/'timestamp-overflow.mkv').write_bytes(header+element('18538067',info+element('1654ae6b',entry)+cluster)) + +# Timing: a Void before the Segment, a file which starts at 10s, and a +# subtitle track with a (deprecated) +# TrackTimecodeScale of 2, which applies to the block's relative timestamp and +# duration but not the cluster's timestamp +import struct +video=element('ae',uint_element('d7',1)+uint_element('73c5',1)+uint_element('83',1)+element('86',b'V_TEST')) +scaled=element('ae',uint_element('d7',2)+uint_element('73c5',2)+uint_element('83',17)+element('86',b'S_TEXT/UTF8') + +element('23314f',struct.pack('>d',2.0))) +group=element('a0',element('a1',b'\x82\x01\xf4\x00sub')+uint_element('9b',1000)) +cluster=element('1f43b675',uint_element('e7',10000)+element('a3',b'\x81\x00\x00\x80v')+group) +(r/'timing.mkv').write_bytes(header+element('ec',bytes(14))+element('18538067',info+element('1654ae6b',video+scaled)+cluster)) + +# A Void followed by CRC-prefixed Info and Tracks, as mkvmerge and ffmpeg +# write. Losing the Info would show as the second packet being at 4ms rather +# than 2ms. +crc=element('bf',bytes(4)) +entry=element('ae',uint_element('d7',1)+uint_element('73c5',1)+uint_element('83',17)+element('86',b'S_TEXT/UTF8')) +void_info=element('1549a966',crc+uint_element('2ad7b1',500000)) +clusters=element('1f43b675',uint_element('e7',0)+element('a3',b'\x81\x00\x00\x80a'))+element('1f43b675',uint_element('e7',4)+element('a3',b'\x81\x00\x00\x80b')) +(r/'void-crc.mkv').write_bytes(header+element('18538067',element('ec',bytes(6))+void_info+element('ec',bytes(6))+element('1654ae6b',crc+entry)+clusters)) + +# The start of the file is the earliest audio or video timestamp, ignoring a +# subtitle line before it +audio=element('ae',uint_element('d7',2)+uint_element('73c5',2)+uint_element('83',2)+element('86',b'A_TEST')) +subtitle=element('ae',uint_element('d7',3)+uint_element('73c5',3)+uint_element('83',17)+element('86',b'S_TEXT/UTF8')) +early=element('a0',element('a1',b'\x83\xfe\x0c\x00early')+uint_element('9b',1000)) +cluster=element('1f43b675',uint_element('e7',1000)+early+element('a3',b'\x82\x00\x00\x80a')+element('a3',b'\x81\x00\x53\x80v')) +(r/'start-time.mkv').write_bytes(header+element('18538067',info+element('1654ae6b',video+audio+subtitle)+cluster)) + +# A partially downloaded file, whose last cluster ends partway through a block +entry=element('ae',uint_element('d7',1)+uint_element('73c5',1)+uint_element('83',17)+element('86',b'S_TEXT/UTF8')) +tracks=element('1654ae6b',entry) +cluster=element('1f43b675',uint_element('e7',0)+element('a3',b'\x81\x00\x00\x80one')+element('a3',b'\x81\x00\x0a\x80two') + +element('a3',b'\x81\x00\x14\x80three-is-cut-off')) +(r/'truncated.mkv').write_bytes((header+element('18538067',info+tracks+cluster))[:-8]) + +# Info after the clusters with no SeekHead to find it +clusters=element('1f43b675',uint_element('e7',0)+element('a3',b'\x81\x00\x00\x80a'))+element('1f43b675',uint_element('e7',4)+element('a3',b'\x81\x00\x00\x80b')) +(r/'late-info.mkv').write_bytes(header+element('18538067',tracks+clusters+element('1549a966',uint_element('2ad7b1',500000)))) diff --git a/tests/matroska/fixtures/late-info.behavior b/tests/matroska/fixtures/late-info.behavior new file mode 100644 index 0000000000..6f1ada46db --- /dev/null +++ b/tests/matroska/fixtures/late-info.behavior @@ -0,0 +1,5 @@ +duration - +start - +track 0 1 S_TEXT/UTF8 eng 0 1 1 +packet 0 0 - 1 44bd8ad473cd9906 +packet 0 2000000 - 1 44bd89d473cd9753 diff --git a/tests/matroska/fixtures/late-info.mkv b/tests/matroska/fixtures/late-info.mkv new file mode 100644 index 0000000000..123e561898 Binary files /dev/null and b/tests/matroska/fixtures/late-info.mkv differ diff --git a/tests/matroska/fixtures/malformed.behavior b/tests/matroska/fixtures/malformed.behavior new file mode 100644 index 0000000000..5ceaadcf04 --- /dev/null +++ b/tests/matroska/fixtures/malformed.behavior @@ -0,0 +1 @@ +error EBML header not found diff --git a/tests/matroska/fixtures/malformed.mkv b/tests/matroska/fixtures/malformed.mkv new file mode 100644 index 0000000000..dabc35d2ef --- /dev/null +++ b/tests/matroska/fixtures/malformed.mkv @@ -0,0 +1 @@ +This is deliberately not an EBML document. diff --git a/tests/matroska/fixtures/repeated-tracks.behavior b/tests/matroska/fixtures/repeated-tracks.behavior new file mode 100644 index 0000000000..68fccb67db --- /dev/null +++ b/tests/matroska/fixtures/repeated-tracks.behavior @@ -0,0 +1,6 @@ +duration - +start - +track 0 7 S_TEXT/UTF8 eng 12 1 1 +attachment a.txt text/plain 3 e172ef510dc287ec +attachment b.txt text/plain 2 9ba86500c657e843 +packet 0 0 - 2 9bca6a00c674d728 diff --git a/tests/matroska/fixtures/repeated-tracks.mkv b/tests/matroska/fixtures/repeated-tracks.mkv new file mode 100644 index 0000000000..51cb26896d Binary files /dev/null and b/tests/matroska/fixtures/repeated-tracks.mkv differ diff --git a/tests/matroska/fixtures/start-time.behavior b/tests/matroska/fixtures/start-time.behavior new file mode 100644 index 0000000000..ef942f83c3 --- /dev/null +++ b/tests/matroska/fixtures/start-time.behavior @@ -0,0 +1,4 @@ +duration 1500000000 +start 1000000000 +track 2 3 S_TEXT/UTF8 eng 0 1 1 +packet 2 500000000 1500000000 5 ae5b9f31c09c6b2a diff --git a/tests/matroska/fixtures/start-time.mkv b/tests/matroska/fixtures/start-time.mkv new file mode 100644 index 0000000000..edeb852a29 Binary files /dev/null and b/tests/matroska/fixtures/start-time.mkv differ diff --git a/tests/matroska/fixtures/subtitle-attachment.behavior b/tests/matroska/fixtures/subtitle-attachment.behavior new file mode 100644 index 0000000000..8ba247ca3c --- /dev/null +++ b/tests/matroska/fixtures/subtitle-attachment.behavior @@ -0,0 +1,8 @@ +duration 3000000000 +start 0 +track 1 2330455654525085584 S_TEXT/ASS eng ASS track 480 1 1 +track 2 7642018816776784136 S_TEXT/UTF8 jpn 0 1 0 +attachment attachment.txt text/plain 42 6e22e83d2777896d +packet 1 250000000 1500000000 25 76874757fddd3870 +packet 1 1500000000 2750000000 24 8f6874da13b53ae +packet 2 500000000 2500000000 24 da1dfbf2ee283ed2 diff --git a/tests/matroska/fixtures/subtitle-attachment.mkv b/tests/matroska/fixtures/subtitle-attachment.mkv new file mode 100644 index 0000000000..1b542f33bc Binary files /dev/null and b/tests/matroska/fixtures/subtitle-attachment.mkv differ diff --git a/tests/matroska/fixtures/subtitle.ass b/tests/matroska/fixtures/subtitle.ass new file mode 100644 index 0000000000..cfd9b69d07 --- /dev/null +++ b/tests/matroska/fixtures/subtitle.ass @@ -0,0 +1,9 @@ +[Script Info] +ScriptType: v4.00+ +[V4+ Styles] +Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding +Style: Default,Arial,20,&H00FFFFFF,&H000000FF,&H00000000,&H00000000,0,0,0,0,100,100,0,0,1,2,0,2,10,10,10,1 +[Events] +Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text +Dialogue: 0,0:00:00.25,0:00:01.50,Default,,0,0,0,,alpha +Dialogue: 1,0:00:01.50,0:00:02.75,Default,,0,0,0,,beta diff --git a/tests/matroska/fixtures/timestamp-overflow.behavior b/tests/matroska/fixtures/timestamp-overflow.behavior new file mode 100644 index 0000000000..90ad1ce38b --- /dev/null +++ b/tests/matroska/fixtures/timestamp-overflow.behavior @@ -0,0 +1,4 @@ +duration - +start - +track 0 1 S_TEXT/UTF8 eng 0 1 1 +error Matroska timestamp is out of range diff --git a/tests/matroska/fixtures/timestamp-overflow.mkv b/tests/matroska/fixtures/timestamp-overflow.mkv new file mode 100644 index 0000000000..0f2cdfddf0 Binary files /dev/null and b/tests/matroska/fixtures/timestamp-overflow.mkv differ diff --git a/tests/matroska/fixtures/timing.behavior b/tests/matroska/fixtures/timing.behavior new file mode 100644 index 0000000000..5208e50dc4 --- /dev/null +++ b/tests/matroska/fixtures/timing.behavior @@ -0,0 +1,4 @@ +duration 13000000000 +start 10000000000 +track 1 2 S_TEXT/UTF8 eng 0 1 1 +packet 1 11000000000 13000000000 3 58db605150dd5fa7 diff --git a/tests/matroska/fixtures/timing.mkv b/tests/matroska/fixtures/timing.mkv new file mode 100644 index 0000000000..ef098fc6cb Binary files /dev/null and b/tests/matroska/fixtures/timing.mkv differ diff --git a/tests/matroska/fixtures/tracks-after-cluster.behavior b/tests/matroska/fixtures/tracks-after-cluster.behavior new file mode 100644 index 0000000000..59eb30829e --- /dev/null +++ b/tests/matroska/fixtures/tracks-after-cluster.behavior @@ -0,0 +1,4 @@ +duration - +start - +track 0 257 S_TEXT/UTF8 eng 0 1 1 +packet 0 0 - 5 5a0d15131ec7a1 diff --git a/tests/matroska/fixtures/tracks-after-cluster.mkv b/tests/matroska/fixtures/tracks-after-cluster.mkv new file mode 100644 index 0000000000..71c446ec7a Binary files /dev/null and b/tests/matroska/fixtures/tracks-after-cluster.mkv differ diff --git a/tests/matroska/fixtures/truncated.behavior b/tests/matroska/fixtures/truncated.behavior new file mode 100644 index 0000000000..9d0210f0f8 --- /dev/null +++ b/tests/matroska/fixtures/truncated.behavior @@ -0,0 +1,5 @@ +duration - +start - +track 0 1 S_TEXT/UTF8 eng 0 1 1 +packet 0 0 - 3 3822b9513ee0e701 +packet 0 10000000 - 3 963c525173d77c8b diff --git a/tests/matroska/fixtures/truncated.mkv b/tests/matroska/fixtures/truncated.mkv new file mode 100644 index 0000000000..c67c992095 Binary files /dev/null and b/tests/matroska/fixtures/truncated.mkv differ diff --git a/tests/matroska/fixtures/utf8.srt b/tests/matroska/fixtures/utf8.srt new file mode 100644 index 0000000000..7279a778ab --- /dev/null +++ b/tests/matroska/fixtures/utf8.srt @@ -0,0 +1,3 @@ +1 +00:00:00,500 --> 00:00:02,500 +Unicode: 日本語 café diff --git a/tests/matroska/fixtures/video-only.behavior b/tests/matroska/fixtures/video-only.behavior new file mode 100644 index 0000000000..620b1e3625 --- /dev/null +++ b/tests/matroska/fixtures/video-only.behavior @@ -0,0 +1,2 @@ +duration 1000000000 +start 0 diff --git a/tests/matroska/fixtures/video-only.mkv b/tests/matroska/fixtures/video-only.mkv new file mode 100644 index 0000000000..6dd0dae1e6 Binary files /dev/null and b/tests/matroska/fixtures/video-only.mkv differ diff --git a/tests/matroska/fixtures/void-crc.behavior b/tests/matroska/fixtures/void-crc.behavior new file mode 100644 index 0000000000..6f1ada46db --- /dev/null +++ b/tests/matroska/fixtures/void-crc.behavior @@ -0,0 +1,5 @@ +duration - +start - +track 0 1 S_TEXT/UTF8 eng 0 1 1 +packet 0 0 - 1 44bd8ad473cd9906 +packet 0 2000000 - 1 44bd89d473cd9753 diff --git a/tests/matroska/fixtures/void-crc.mkv b/tests/matroska/fixtures/void-crc.mkv new file mode 100644 index 0000000000..eef333b336 Binary files /dev/null and b/tests/matroska/fixtures/void-crc.mkv differ diff --git a/tests/meson.build b/tests/meson.build index 18939e7edb..da7c381a68 100644 --- a/tests/meson.build +++ b/tests/meson.build @@ -35,6 +35,7 @@ tests_src = [ 'tests/line_iterator.cpp', 'tests/line_wrap.cpp', 'tests/lua_lfs.cpp', + 'tests/matroska.cpp', 'tests/mru.cpp', 'tests/option.cpp', 'tests/path.cpp', diff --git a/tests/tests/matroska.cpp b/tests/tests/matroska.cpp new file mode 100644 index 0000000000..113e9e7c8e --- /dev/null +++ b/tests/tests/matroska.cpp @@ -0,0 +1,172 @@ +#include + +#include "matroska.h" +#include "util.h" + +#include +#include +#include +#include +#include + +namespace { +using namespace agi::matroska; + +agi::fs::path fixture_dir() { + return util::test_data_dir() / "matroska" / "fixtures"; +} + +agi::fs::path fixture(char const *name) { + return fixture_dir() / name; +} + +uint64_t fnv1a(std::vector const& data) { + uint64_t hash = 1469598103934665603ULL; + for (auto byte : data) { + hash ^= byte; + hash *= 1099511628211ULL; + } + return hash; +} + +std::string describe_time(std::optional const& time) { + return time ? std::to_string(time->nanoseconds) : "-"; +} + +// Dump everything the demuxer exposes for a file in a stable text format +std::string describe(agi::fs::path const& path) { + std::ostringstream out; + try { + Demuxer demuxer(OpenFile(path)); + out << "duration\t" << describe_time(demuxer.Duration()) << '\n'; + out << "start\t" << describe_time(demuxer.StartTime()) << '\n'; + for (auto const& track : demuxer.SubtitleTracks()) + out << "track\t" << track.id.value << '\t' << track.uid << '\t' << track.codec_id << '\t' + << track.language << '\t' << track.name << '\t' << track.codec_private.size() << '\t' + << track.enabled << '\t' << track.is_default << '\n'; + for (auto const& attachment : demuxer.Attachments()) + out << "attachment\t" << attachment.name << '\t' << attachment.mime_type << '\t' << attachment.size + << '\t' << std::hex << fnv1a(demuxer.ReadAttachment(attachment.id)) << std::dec << '\n'; + for (auto const& track : demuxer.SubtitleTracks()) { + if (track.codec == SubtitleCodec::unsupported) continue; + demuxer.SelectTrack(track.id); + while (auto packet = demuxer.ReadPacket()) + out << "packet\t" << packet->track.value << '\t' << packet->start.nanoseconds << '\t' + << describe_time(packet->end) << '\t' << packet->data.size() << '\t' + << std::hex << fnv1a(packet->data) << std::dec << '\n'; + } + } + catch (Error const& e) { + out << "error\t" << e.GetMessage() << '\n'; + } + return out.str(); +} + +class ReaderProxy final : public Reader { + std::unique_ptr source; +public: + bool fail = false; + bool short_reads = false; + explicit ReaderProxy(agi::fs::path const& path) : source(OpenFile(path)) {} + uint64_t Size() const override { return source->Size(); } + size_t Read(uint64_t position, void *buffer, size_t size) override { + if (fail) throw IoError("injected reader failure"); + if (short_reads && size > 1) size = 1; + return source->Read(position, buffer, size); + } +}; +} + +TEST(Matroska, MetadataAndAttachments) { + Demuxer demuxer(OpenFile(fixture("subtitle-attachment.mkv"))); + ASSERT_FALSE(demuxer.SubtitleTracks().empty()); + ASSERT_EQ(1u, demuxer.Attachments().size()); + auto bytes = demuxer.ReadAttachment(demuxer.Attachments()[0].id); + EXPECT_EQ(demuxer.Attachments()[0].size, bytes.size()); + EXPECT_THROW(demuxer.ReadAttachment(AttachmentId{UINT64_MAX}), InvalidDataError); +} + +TEST(Matroska, ShortReadsAreSupported) { + auto reader = std::make_unique(fixture("subtitle-attachment.mkv")); + reader->short_reads = true; + Demuxer demuxer(std::move(reader)); + EXPECT_FALSE(demuxer.SubtitleTracks().empty()); +} + +TEST(Matroska, ReaderFailureDuringOpenIsAnIoError) { + auto reader = std::make_unique(fixture("subtitle-attachment.mkv")); + reader->fail = true; + EXPECT_THROW(Demuxer(std::move(reader)), IoError); +} + +TEST(Matroska, MissingFileIsAnIoError) { + EXPECT_THROW(OpenFile(fixture("does-not-exist.mkv")), IoError); +} + +TEST(Matroska, CancellationCallbackExceptionsCancel) { + EXPECT_THROW(Demuxer(OpenFile(fixture("subtitle-attachment.mkv")), []() -> bool { throw 42; }), agi::UserCancelException); +} + +TEST(Matroska, PacketScanCanBeCancelled) { + bool cancelled = false; + Demuxer demuxer(OpenFile(fixture("subtitle-attachment.mkv")), [&] { return cancelled; }); + demuxer.SelectTrack(demuxer.SubtitleTracks()[0].id); + cancelled = true; + EXPECT_THROW((void)demuxer.ReadPacket(), agi::UserCancelException); +} + +TEST(Matroska, RejectsUnsupportedTrackAndInvalidIds) { + Demuxer demuxer(OpenFile(fixture("video-only.mkv"))); + EXPECT_TRUE(demuxer.SubtitleTracks().empty()); + EXPECT_THROW(demuxer.SelectTrack(TrackId{UINT32_MAX}), InvalidDataError); +} + +TEST(Matroska, PacketAndAttachmentLimits) { + Limits limits; limits.packet_size = 1; limits.attachment_size = 1; limits.decompressed_size = 1; + Demuxer demuxer(OpenFile(fixture("subtitle-attachment.mkv")), {}, limits); + ASSERT_FALSE(demuxer.SubtitleTracks().empty()); + demuxer.SelectTrack(demuxer.SubtitleTracks()[0].id); + EXPECT_THROW(demuxer.ReadPacket(), LimitError); + ASSERT_FALSE(demuxer.Attachments().empty()); + EXPECT_THROW(demuxer.ReadAttachment(demuxer.Attachments()[0].id), LimitError); +} + +TEST(Matroska, DecompressionLimitAndFailedPacketConsumption) { + Limits limits; limits.decompressed_size = 1; + Demuxer compressed(OpenFile(fixture("compressed-zlib.mkv")), {}, limits); + compressed.SelectTrack(compressed.SubtitleTracks()[0].id); + EXPECT_THROW(compressed.ReadPacket(), LimitError); + + auto reader = std::make_unique(fixture("subtitle-attachment.mkv")); + auto *control = reader.get(); + Demuxer demuxer(std::move(reader)); + demuxer.SelectTrack(demuxer.SubtitleTracks()[0].id); + control->fail = true; + EXPECT_THROW(demuxer.ReadPacket(), IoError); + control->fail = false; + // The failed packet was already consumed before its payload read began. + EXPECT_NO_THROW((void)demuxer.ReadPacket()); +} + +// Each fixture's .behavior file records the expected describe() output. When a +// fixture is added or changed, the failure message contains the new output. +TEST(Matroska, FixtureBehavior) { + std::vector fixtures; + for (auto const& entry : std::filesystem::directory_iterator(fixture_dir())) { + auto ext = entry.path().extension(); + if (ext == ".mkv" || ext == ".mka") + fixtures.emplace_back(entry.path()); + } + std::sort(fixtures.begin(), fixtures.end()); + ASSERT_FALSE(fixtures.empty()); + + for (auto const& path : fixtures) { + auto expected_path = path; + expected_path.replace_extension(".behavior"); + std::ifstream expected_file(expected_path, std::ios::binary); + std::stringstream expected; + expected << expected_file.rdbuf(); + auto actual = describe(path); + EXPECT_EQ(expected.str(), actual) << path.filename(); + } +}