Put all functions related to mkvtoolnix tools into a dedicated module.

This commit is contained in:
Frédéric Tronel
2026-08-29 15:30:29 +02:00
parent 3769244bdb
commit 280eb1facb
2 changed files with 479 additions and 455 deletions
+463
View File
@@ -0,0 +1,463 @@
# SPDX-License-Identifier: GPL-2.0-or-later
#
# Copyright (C) 2026 Frédéric Tronel
import logging
from io import TextIOWrapper
import re
from datetime import timedelta
from os import (
read,
SEEK_SET,
lseek,
write,
fstat,
set_inheritable,
ftruncate
)
import os
from subprocess import PIPE, Popen
from typing import IO
from typeguard import typechecked
import hexdump
from tqdm import tqdm
from tscut.matroska.ebml import change_ebml_element_size
# Found codec private data using mkvinfo
@typechecked
def get_codec_private_data_from_mkv(mkvinfo_path:str,
input_file: IO[bytes]) -> tuple[int, bytes]|tuple[None,None]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
found = False
env = {**os.environ, 'LANG': 'C'}
# Output example
# Codec's private data: size 48 (H.264 profile: High @L4.0) hexdump 01 64 00 28 ff e1 00 1b 67\
# 64 00 28 ac d9 40 78 04 4f dc d4 04 04 05 00 00 92 ef 00 1d ad a6 1f 16 2d 96 01 00 06 68 fb\
# a3 cb 22 c0 fd f8 f8 00 at 406 size 51 data size 48
with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE,
close_fds=False, env=env) as mkvinfo:
out, _ = mkvinfo.communicate()
lines = out.decode('utf8')
reg_exp = (r"^.*Codec's private data: size ([0-9]+) \(H.264.*\) hexdump "
r"(?P<hexdump>([0-9a-f]{2} )+)at (?P<position>[0-9]+) size (?P<size>[0-9]+).*$")
p = re.compile(reg_exp)
for line in lines.splitlines():
m = p.match(line)
if m is not None:
size = int(m.group('size'))
position = int(m.group('position'))
logger.debug("Found codec private data at position: %s, size: %d", position, size)
found = True
mkvinfo.wait()
break
if found:
lseek(infd, position, SEEK_SET)
data = read(infd, size)
return position, data
logger.error("Impossible to retrieve codec private data from MKV !")
return None, None
@typechecked
def parse_mkv_tree(mkvinfo_path:str, input_file: IO[bytes]) -> dict[str,tuple[int,int]]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
env = {**os.environ, 'LANG': 'C'}
elements = {}
with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE,
close_fds=False, env=env) as mkvinfo:
out, _ = mkvinfo.communicate()
lines = out.decode('utf8')
prefix = []
reg_exp = (r"(^(?P<root>\+)|(\|(?P<depth>[ ]*\+))).*at (?P<position>[0-9]+)"
r" size (?P<size>[0-9]+).*$")
p = re.compile(reg_exp)
prev_depth = -1
for line in lines.splitlines():
m = p.match(line)
if m is None:
logger.error("Impossible to match line: %s", line)
else:
position = int(m.group('position'))
size = int(m.group('size'))
root = m.group('root') is not None
if root:
depth = 0
else:
depth = len(m.group('depth'))
if depth > prev_depth:
for _ in range(depth-prev_depth):
prefix.append(1)
elif depth == prev_depth:
subid = prefix[-1]
subid+=1
prefix.pop()
prefix.append(subid)
else:
for _ in range(prev_depth-depth):
prefix.pop()
subid = prefix[-1]
subid+=1
prefix.pop()
prefix.append(subid)
prev_depth = depth
key=".".join(map(str, prefix))
elements[key] = (position, size)
mkvinfo.wait()
return elements
@typechecked
def change_codec_private_data(mkvinfo_path:str, input_file: IO[bytes], codec_data:bytes) -> None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
current_length = fstat(infd).st_size
logger.info('Current size of file: %d', current_length)
position, current_data = get_codec_private_data_from_mkv(mkvinfo_path, input_file)
if position is None or current_data is None:
return None
current_data_length = len(current_data)
future_length = current_length - current_data_length + len(codec_data)
logger.info('Expected size of file: %d', future_length)
logger.info('Current data at position %d: %s', position, hexdump.dump(current_data, sep=":"))
logger.info('Future data: %s', hexdump.dump(codec_data, sep=":"))
elements = parse_mkv_tree(mkvinfo_path, input_file)
found = False
for key, (pos,size) in elements.items():
if pos == position:
logger.info('Codec private data key: %s', key)
found = True
break
if not found:
logger.error('Impossible to retrieve the key of codec private data')
exit(-1)
if current_length < future_length:
lseek(infd, position+current_data_length, SEEK_SET)
tail = read(infd, current_length-(position+current_data_length))
# We extend the file at the end with zeroes
ftruncate(infd, future_length)
lseek(infd, position+len(codec_data), SEEK_SET)
write(infd, tail)
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
elif current_length == future_length:
# Almost nothing to do except overwriting old private codec data with new ones.
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
else:
lseek(infd, position+current_data_length, SEEK_SET)
tail = read(infd, current_length-(position+current_data_length))
lseek(infd, position+len(codec_data), SEEK_SET)
write(infd, tail)
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
# We reduce the length of file.
ftruncate(infd, future_length)
# We have to modify the tree elements up to the root that contains the codec private data.
keys = key.split('.')
logger.info(keys)
delta = future_length-current_length
# if there is no modification of the private codec data, no need to change anything.
if delta != 0:
for _ in range(len(keys)-1):
keys.pop()
key=".".join(map(str, keys))
pos, size = elements[key]
logger.info('Trying to fix element with key: %s at position: %d with actual size: %d.',
key, pos, size)
# Changing an element can increase its size (in very rare case).
# In that case, we update the new delta that will be larger (because the element has
# been resized).
delta+=change_ebml_element_size(input_file, pos, delta)
@typechecked
def extract_mkv_part(mkvmerge_path:str, input_file:IO[bytes], output_file:IO[bytes],
begin:timedelta, end:timedelta) -> None:
logger = logging.getLogger(__name__)
logger.info('Extract video between I-frames at %s and %s', begin,end)
infd = input_file.fileno()
outfd = output_file.fileno()
lseek(infd, 0, SEEK_SET)
lseek(outfd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
env = {**os.environ, 'LANG': 'C'}
warnings = []
command = [mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '--split', f'parts:{begin}-{end}',
f'/proc/self/fd/{infd:d}']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
elif line.startswith('Warning'):
warnings.append(line)
pb.update(100-pb.n)
pb.refresh()
pb.close()
status = mkvmerge.wait()
if status == 1:
logger.warning('Extraction returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Extraction returns errors')
# Merge a list of mkv files passed as input, and produce a new MKV as output
@typechecked
def merge_mkvs(mkvmerge_path:str, inputs: list[IO[bytes]], output_name:str,
concatenate: bool=True,
timestamps: dict[int, IO[str]] | None = None) -> IO[bytes]|None:
logger = logging.getLogger(__name__)
if timestamps is None:
timestamps = {}
fds = []
try:
out = open(output_name, 'wb+')
except OSError:
logger.error('Impossible to create file: %s', output_name)
return None
outfd = out.fileno()
lseek(outfd, 0, SEEK_SET)
fds.append(outfd)
set_inheritable(outfd, True)
# Timestamps of merged tracks are modified by the length of the preceding track.
# The default mode ('file') is using the largest timestamp of the whole file which may create
# desynchronize video and sound.
merge_params = [mkvmerge_path, '--append-mode', 'track']
first = True
partnum = 0
for mkv in inputs:
if mkv is not None:
fd = mkv.fileno()
fds.append(fd)
set_inheritable(fd, True)
# If we pass a timestamps file associated with the considered track, use it.
if partnum in timestamps:
tsfd = timestamps[partnum].fileno()
lseek(tsfd, 0, SEEK_SET)
fds.append(tsfd)
set_inheritable(tsfd, True)
merge_params.extend(['--timestamps', f'{partnum:d}:/proc/self/fd/{tsfd:d}'])
if first:
merge_params.append(f'/proc/self/fd/{fd:d}')
first = False
elif concatenate:
merge_params.append(f'+/proc/self/fd/{fd:d}')
else:
merge_params.append(f'/proc/self/fd/{fd:d}')
partnum+=1
merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}'])
# We merge all files.
warnings = []
env = {**os.environ, 'LANG': 'C'}
logger.debug('Executing: LANG=C %s', merge_params)
with Popen(merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Merging')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.n = int(m['progress'])
pb.update()
elif line.startswith('Warning'):
warnings.append(line)
status = mkvmerge.wait()
if status == 1:
logger.warning('Extraction returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Extraction returns errors')
for fd in fds:
set_inheritable(fd, False)
return out
@typechecked
def extract_track_from_mkv(mkvextract_path: str, input_file: IO[bytes], index,
output_file: IO[bytes], timestamps) -> None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
outfd = output_file.fileno()
lseek(outfd, 0, SEEK_SET)
set_inheritable(outfd, True)
tsfd = timestamps.fileno()
lseek(tsfd, 0, SEEK_SET)
set_inheritable(tsfd, True)
params = [ mkvextract_path, f'/proc/self/fd/{infd:d}', 'tracks',
f'{index:d}:/proc/self/fd/{outfd:d}', 'timestamps_v2',
f'{index:d}:/proc/self/fd/{tsfd:d}']
env = {**os.environ, 'LANG': 'C'}
logger.debug('Executing: LANG=C %s', params)
with Popen(params, stdout=PIPE, close_fds=False, env=env) as extract:
pb = tqdm(TextIOWrapper(extract.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction of track')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
extract.wait()
if extract.returncode != 0:
logger.error('Mkvextract returns an error code: %d', extract.returncode)
else:
logger.info('Track %d was succesfully extracted.', index)
@typechecked
def remove_video_tracks_from_mkv(mkvmerge_path:str, input_file: IO[bytes],
output_file: IO[bytes]) -> None:
logger = logging.getLogger(__name__)
outfd = output_file.fileno()
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
lseek(outfd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
params = [ mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '-D', f'/proc/self/fd/{infd:d}']
logger.debug('Executing: LANG=C %s', params)
env = {**os.environ, 'LANG': 'C'}
with Popen(params, stdout=PIPE, close_fds=False, env=env) as remove:
pb = tqdm(TextIOWrapper(remove.stdout, encoding="utf-8"), total=100, unit='%',
desc='Removal of video track:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
remove.wait()
if remove.returncode != 0:
logger.error('Mkvmerge returns an error code: %d', remove.returncode)
else:
logger.info('Video tracks were succesfully extracted.')
@typechecked
def remux_srt_subtitles(mkvmerge_path:str, input_file: IO[bytes], output_filename: str,
subtitles) -> None:
logger = logging.getLogger(__name__)
try:
out = open(output_filename, 'w', encoding='utf8')
except OSError:
logger.error('Impossible to create file: %s', output_filename)
return None
outfd = out.fileno()
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
mkv_merge_params = [mkvmerge_path, f'/proc/self/fd/{infd:d}']
for fd, lang in subtitles:
lseek(fd, 0, SEEK_SET)
set_inheritable(fd, True)
mkv_merge_params.extend(['--language', f'0:{lang}', f'/proc/self/fd/{fd:d}'])
mkv_merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}'])
warnings = []
env = {**os.environ, 'LANG': 'C'}
logger.info('Remux subtitles: %s', mkv_merge_params)
with Popen(mkv_merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Remux subtitles:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.n = int(m['progress'])
pb.update()
elif line.startswith('Warning'):
warnings.append(line)
status = mkvmerge.wait()
if status == 1:
logger.warning('Remux subtitles returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Remux subtitles returns errors')
return None
+16 -455
View File
@@ -2,36 +2,24 @@
'''A module to remove parts of video (.e.g advertisements) with single frame precision.'''
# Standard modules
import json
import logging
import os.path
import re
from datetime import timedelta
from enum import IntEnum, unique
from io import BytesIO, TextIOWrapper
from math import ceil, floor, log
from os import (
SEEK_SET,
close,
fstat,
ftruncate,
lseek,
memfd_create,
read,
set_inheritable,
write,
)
from subprocess import PIPE, Popen
from sys import exit
from typing import IO
# Third party libraries
import hexdump
from iso639 import Lang
from iso639.exceptions import InvalidLanguageValue
from tqdm import tqdm
from typeguard import typechecked
from tscut.tools.mkvtoolnix import change_codec_private_data
from tscut.h264.avc import (
get_avc_config_from_h264,
parse_codec_private
@@ -54,16 +42,22 @@ from tscut.tools.ffmpeg import (
ffmpeg_convert
)
from tscut.tools.timeframe import (
compare_time_interval,
parse_time_interval
compare_time_interval,
parse_time_interval
)
from tscut.tools.discovery import check_required_tools
from tscut.ocr.subtitles import (
do_ocr,
extract_srt,
get_tesseract_supported_lang
do_ocr,
extract_srt,get_tesseract_supported_lang
)
from tscut.tools.mkvtoolnix import (
extract_mkv_part,
extract_track_from_mkv,
get_codec_private_data_from_mkv,
merge_mkvs,
remove_video_tracks_from_mkv,
remux_srt_subtitles
)
# Useful SPS/PPS discussion.
# https://copyprogramming.com/howto/including-sps-and-pps-in-a-raw-h264-track
@@ -116,441 +110,8 @@ class SupportedFormat(IntEnum):
# ffmpeg -i <InputFile (before concatenation)> -c:v copy -an -sn -bsf:v trace_headers -t 0.01\
# -report -loglevel 0 -f null -
# Found codec private data using mkvinfo
@typechecked
def get_codec_private_data_from_mkv(mkvinfo_path:str,
input_file: IO[bytes]) -> tuple[int, bytes]|tuple[None,None]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
found = False
env = {**os.environ, 'LANG': 'C'}
# Output example
# Codec's private data: size 48 (H.264 profile: High @L4.0) hexdump 01 64 00 28 ff e1 00 1b 67\
# 64 00 28 ac d9 40 78 04 4f dc d4 04 04 05 00 00 92 ef 00 1d ad a6 1f 16 2d 96 01 00 06 68 fb\
# a3 cb 22 c0 fd f8 f8 00 at 406 size 51 data size 48
with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE,
close_fds=False, env=env) as mkvinfo:
out, _ = mkvinfo.communicate()
out = out.decode('utf8')
reg_exp = (r"^.*Codec's private data: size ([0-9]+) \(H.264.*\) hexdump "
r"(?P<hexdump>([0-9a-f]{2} )+)at (?P<position>[0-9]+) size (?P<size>[0-9]+).*$")
p = re.compile(reg_exp)
for line in out.splitlines():
m = p.match(line)
if m is not None:
size = int(m.group('size'))
position = int(m.group('position'))
logger.debug("Found codec private data at position: %s, size: %d", position, size)
found = True
mkvinfo.wait()
break
if found:
lseek(infd, position, SEEK_SET)
data = read(infd, size)
return position, data
return None, None
@typechecked
def parse_mkv_tree(mkvinfo_path:str, input_file: IO[bytes]) -> dict[str,tuple[int,int]]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
env = {**os.environ, 'LANG': 'C'}
elements = {}
with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE,
close_fds=False, env=env) as mkvinfo:
out, _ = mkvinfo.communicate()
out = out.decode('utf8')
prefix = []
reg_exp = (r"(^(?P<root>\+)|(\|(?P<depth>[ ]*\+))).*at (?P<position>[0-9]+)"
r" size (?P<size>[0-9]+).*$")
p = re.compile(reg_exp)
prev_depth = -1
for line in out.splitlines():
m = p.match(line)
if m is None:
logger.error("Impossible to match line: %s", line)
else:
position = int(m.group('position'))
size = int(m.group('size'))
root = m.group('root') is not None
if root:
depth = 0
else:
depth = len(m.group('depth'))
if depth > prev_depth:
for _ in range(depth-prev_depth):
prefix.append(1)
elif depth == prev_depth:
subid = prefix[-1]
subid+=1
prefix.pop()
prefix.append(subid)
else:
for _ in range(prev_depth-depth):
prefix.pop()
subid = prefix[-1]
subid+=1
prefix.pop()
prefix.append(subid)
prev_depth = depth
key=".".join(map(str, prefix))
elements[key] = (position, size)
mkvinfo.wait()
return elements
@typechecked
def change_codec_private_data(mkvinfo_path:str, input_file: IO[bytes], codec_data:bytes) -> None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
current_length = fstat(infd).st_size
logger.info('Current size of file: %d', current_length)
position, current_data = get_codec_private_data_from_mkv(mkvinfo_path, input_file)
current_data_length = len(current_data)
future_length = current_length - current_data_length + len(codec_data)
logger.info('Expected size of file: %d', future_length)
logger.info('Current data at position %d: %s', position, hexdump.dump(current_data, sep=":"))
logger.info('Future data: %s', hexdump.dump(codec_data, sep=":"))
elements = parse_mkv_tree(mkvinfo_path, input_file)
found = False
for key, (pos,size) in elements.items():
if pos == position:
logger.info('Codec private data key: %s', key)
found = True
break
if not found:
logger.error('Impossible to retrieve the key of codec private data')
exit(-1)
if current_length < future_length:
lseek(infd, position+current_data_length, SEEK_SET)
tail = read(infd, current_length-(position+current_data_length))
# We extend the file at the end with zeroes
ftruncate(infd, future_length)
lseek(infd, position+len(codec_data), SEEK_SET)
write(infd, tail)
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
elif current_length == future_length:
# Almost nothing to do except overwriting old private codec data with new ones.
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
else:
lseek(infd, position+current_data_length, SEEK_SET)
tail = read(infd, current_length-(position+current_data_length))
lseek(infd, position+len(codec_data), SEEK_SET)
write(infd, tail)
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
# We reduce the length of file.
ftruncate(infd, future_length)
# We have to modify the tree elements up to the root that contains the codec private data.
keys = key.split('.')
logger.info(keys)
delta = future_length-current_length
# if there is no modification of the private codec data, no need to change anything.
if delta != 0:
for _ in range(len(keys)-1):
keys.pop()
key=".".join(map(str, keys))
pos, size = elements[key]
logger.info('Trying to fix element with key: %s at position: %d with actual size: %d.',
key, pos, size)
# Changing an element can increase its size (in very rare case).
# In that case, we update the new delta that will be larger (because the element has
# been resized).
delta+=change_ebml_element_size(input_file, pos, delta)
@typechecked
def extract_mkv_part(mkvmerge_path:str, input_file:IO[bytes], output_file:IO[bytes],
begin:timedelta, end:timedelta) -> None:
logger = logging.getLogger(__name__)
logger.info('Extract video between I-frames at %s and %s', begin,end)
infd = input_file.fileno()
outfd = output_file.fileno()
lseek(infd, 0, SEEK_SET)
lseek(outfd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
env = {**os.environ, 'LANG': 'C'}
warnings = []
command = [mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '--split', f'parts:{begin}-{end}',
f'/proc/self/fd/{infd:d}']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
elif line.startswith('Warning'):
warnings.append(line)
pb.update(100-pb.n)
pb.refresh()
pb.close()
status = mkvmerge.wait()
if status == 1:
logger.warning('Extraction returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Extraction returns errors')
# Merge a list of mkv files passed as input, and produce a new MKV as output
@typechecked
def merge_mkvs(mkvmerge_path:str, inputs: list[IO[bytes]], output_name:str,
concatenate: bool=True, timestamps: dict[int, IO[str]] | None = None) -> IO[bytes]|None:
logger = logging.getLogger(__name__)
if timestamps is None:
timestamps = {}
fds = []
try:
out = open(output_name, 'wb+')
except OSError:
logger.error('Impossible to create file: %s', output_name)
return None
outfd = out.fileno()
lseek(outfd, 0, SEEK_SET)
fds.append(outfd)
set_inheritable(outfd, True)
# Timestamps of merged tracks are modified by the length of the preceding track.
# The default mode ('file') is using the largest timestamp of the whole file which may create
# desynchronize video and sound.
merge_params = [mkvmerge_path, '--append-mode', 'track']
first = True
partnum = 0
for mkv in inputs:
if mkv is not None:
fd = mkv.fileno()
fds.append(fd)
set_inheritable(fd, True)
# If we pass a timestamps file associated with the considered track, use it.
if partnum in timestamps:
tsfd = timestamps[partnum].fileno()
lseek(tsfd, 0, SEEK_SET)
fds.append(tsfd)
set_inheritable(tsfd, True)
merge_params.extend(['--timestamps', f'{partnum:d}:/proc/self/fd/{tsfd:d}'])
if first:
merge_params.append(f'/proc/self/fd/{fd:d}')
first = False
elif concatenate:
merge_params.append(f'+/proc/self/fd/{fd:d}')
else:
merge_params.append(f'/proc/self/fd/{fd:d}')
partnum+=1
merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}'])
# We merge all files.
warnings = []
env = {**os.environ, 'LANG': 'C'}
logger.debug('Executing: LANG=C %s', merge_params)
with Popen(merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Merging')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.n = int(m['progress'])
pb.update()
elif line.startswith('Warning'):
warnings.append(line)
status = mkvmerge.wait()
if status == 1:
logger.warning('Extraction returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Extraction returns errors')
for fd in fds:
set_inheritable(fd, False)
return out
@typechecked
def extract_track_from_mkv(mkvextract_path: str, input_file: IO[bytes], index,
output_file: IO[bytes], timestamps) -> None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
outfd = output_file.fileno()
lseek(outfd, 0, SEEK_SET)
set_inheritable(outfd, True)
tsfd = timestamps.fileno()
lseek(tsfd, 0, SEEK_SET)
set_inheritable(tsfd, True)
params = [ mkvextract_path, f'/proc/self/fd/{infd:d}', 'tracks',
f'{index:d}:/proc/self/fd/{outfd:d}', 'timestamps_v2',
f'{index:d}:/proc/self/fd/{tsfd:d}']
env = {**os.environ, 'LANG': 'C'}
logger.debug('Executing: LANG=C %s', params)
with Popen(params, stdout=PIPE, close_fds=False, env=env) as extract:
pb = tqdm(TextIOWrapper(extract.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction of track')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
extract.wait()
if extract.returncode != 0:
logger.error('Mkvextract returns an error code: %d', extract.returncode)
else:
logger.info('Track %d was succesfully extracted.', index)
@typechecked
def remove_video_tracks_from_mkv(mkvmerge_path:str, input_file: IO[bytes],
output_file: IO[bytes]) -> None:
logger = logging.getLogger(__name__)
outfd = output_file.fileno()
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
lseek(outfd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
params = [ mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '-D', f'/proc/self/fd/{infd:d}']
logger.debug('Executing: LANG=C %s', params)
env = {**os.environ, 'LANG': 'C'}
with Popen(params, stdout=PIPE, close_fds=False, env=env) as remove:
pb = tqdm(TextIOWrapper(remove.stdout, encoding="utf-8"), total=100, unit='%',
desc='Removal of video track:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
remove.wait()
if remove.returncode != 0:
logger.error('Mkvmerge returns an error code: %d', remove.returncode)
else:
logger.info('Video tracks were succesfully extracted.')
@typechecked
def remux_srt_subtitles(mkvmerge_path:str, input_file: IO[bytes], output_filename: str,
subtitles) -> None:
logger = logging.getLogger(__name__)
try:
out = open(output_filename, 'w', encoding='utf8')
except OSError:
logger.error('Impossible to create file: %s', output_filename)
return None
outfd = out.fileno()
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
mkv_merge_params = [mkvmerge_path, f'/proc/self/fd/{infd:d}']
for fd, lang in subtitles:
lseek(fd, 0, SEEK_SET)
set_inheritable(fd, True)
mkv_merge_params.extend(['--language', f'0:{lang}', f'/proc/self/fd/{fd:d}'])
mkv_merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}'])
warnings = []
env = {**os.environ, 'LANG': 'C'}
logger.info('Remux subtitles: %s', mkv_merge_params)
with Popen(mkv_merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Remux subtitles:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.n = int(m['progress'])
pb.update()
elif line.startswith('Warning'):
warnings.append(line)
status = mkvmerge.wait()
if status == 1:
logger.warning('Remux subtitles returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Remux subtitles returns errors')
return None
@typechecked
def concatenate_h264_parts(h264parts: list[IO[bytes]], output: IO[bytes]) -> None:
@@ -582,7 +143,7 @@ def concatenate_h264_parts(h264parts: list[IO[bytes]], output: IO[bytes]) -> Non
def concatenate_h264_ts_parts(h264_ts_parts: list[IO[bytes]], output: IO[bytes]) -> None:
logger = logging.getLogger(__name__)
header = '# timestamp format v2\n'
header = '# timestamp format v2\n'.encode('ascii')
output.write(header)
@@ -605,7 +166,7 @@ def concatenate_h264_ts_parts(h264_ts_parts: list[IO[bytes]], output: IO[bytes])
break
ts = offset + float(line)
last = max(last,ts)
output.write(f'{ts:f}\n')
output.write(f'{ts:f}\n'.encode('ascii'))
if first:
first = False