Files
removeads/src/tscut/tscut.py
T

1877 lines
70 KiB
Python
Executable File

#!/usr/bin/env python3
'''A module to remove parts of video (.e.g advertisements) with single frame precision.'''
# Standard modules
import json
import logging
import os.path
import re
from datetime import timedelta
from enum import IntEnum, unique
from io import BytesIO, TextIOWrapper
from math import ceil, floor, log
from os import (
SEEK_SET,
close,
fstat,
ftruncate,
lseek,
memfd_create,
read,
set_inheritable,
write,
)
from shutil import which
from subprocess import PIPE, Popen
from sys import exit
from typing import IO
# Third party libraries
import hexdump
from iso639 import Lang
from iso639.exceptions import InvalidLanguageValue
from tqdm import tqdm
from typeguard import typechecked
from tscut.h264.avc import (dump_codec_private_data,
get_avc_config_from_h264,
parse_codec_private)
# Useful SPS/PPS discussion.
# https://copyprogramming.com/howto/including-sps-and-pps-in-a-raw-h264-track
# https://gitlab.com/mbunkus/mkvtoolnix/-/issues/2390
# New strategy: a possible way of handling multiple SPS/PPS gracefully.
# Encode each head and trailer with FFMPEG using only I-frame (to be sure the NAL unit will never
# refer to another image).
# Encode using an different SPS-ID all of them (using sps-id parameter of libx264 library, e.g
# 1 instead of 0).
# For the video track produce only a raw H264 file and a file containing timestamps of the
# different frames.
# For the rest of the tracks (audio, subtitles) produce directly a MKV (this is already done).
# Concatenate all raw H264 in a giant one (like cat), and the same for timestamps of video frames
# (to keep sound and video synchronized).
# Then use mkvmerge to remux the H264 track and the rest of tracks.
# MKVmerge "concatenate" subcommand is able to concatenate different SPS/PPS data into a bigger
# Private Codec Data.
# However, this is proved to be not reliable. Sometimes it results in a AVC context containing
# a single SPS/PPS.
# So we have to rely on a manual parsing of the H264 AVC context of original movie
# and the ones produced for headers and trailers, and then merging them into a bigger AVC context.
# Then finally, change the Private Codec Data in the final MKV.
@typechecked
def check_required_tools() -> tuple[bool,dict[str,str]]:
"""
Checks if all required external tools are installed.
This function verifies the presence of required and optional external tools on the system.
It returns a tuple containing a boolean indicating whether all optional tools are installed,
along with a dictionary containing the paths to all tools.
Args:
None
Returns:
tuple[bool, dict[str, str]]:
- bool: True if all optional tools are installed, False otherwise
- dict[str, str]: dictionary containing the paths to all tools
"""
logger = logging.getLogger(__name__)
all_optional_tools = True
paths = {}
required = ['ffmpeg', 'ffprobe', 'mkvmerge', 'mkvinfo']
optional = ['mkvextract', 'vobsubocr','tesseract']
for tool in required:
path = which(tool)
if path is None:
logger.error('Required tool: %s is missing.',tool)
exit(-1)
else:
paths[tool] = path
for tool in optional:
path = which(tool)
if path is None:
logger.info('Optional tool: %s is missing.',tool)
all_optional_tools = False
else:
paths[tool] = path
return all_optional_tools, paths
@typechecked
def get_tesseract_supported_lang(tesseract_path:str) -> dict[Lang, str]|None:
"""
Retrieves the set of natural languages supported by the Tesseract OCR tool.
This function runs the Tesseract binary with the --list-langs option and parses the output
to extract the supported languages.
Args:
tesseract_path (str): The path to the Tesseract binary.
Returns:
dict[Lang, str] | None:
- A dictionary mapping Lang objects to their corresponding language codes
(e.g., "eng" for English)
- None if an error occurs while running the Tesseract binary
"""
logger = logging.getLogger(__name__)
res = {}
with Popen([tesseract_path, '--list-langs'], stdout=PIPE) as tesseract:
for line in tesseract.stdout:
line = line.decode('utf8')
p = re.compile('(?P<lang>[a-z]{3})\n')
m = re.match(p,line)
if m is not None:
try:
lang = m.group('lang')
key = Lang(lang)
res[key] = lang
except InvalidLanguageValue as e:
logger.warning('Invalid language: %s', e)
pass
tesseract.wait()
if tesseract.returncode != 0:
logger.error("Tesseract returns an error code: %d",tesseract.returncode)
return None
return res
@typechecked
def get_frame_rate(ffprobe_path:str, input_file: IO[bytes]) -> float|None:
"""
Retrieves the frame rate of a video file using the ffprobe tool.
This function runs the ffprobe binary with the specified input file and parses the output
to extract the frame rate.
It uses two methods to calculate the frame rate: one based on the timestamp of the frames
and another based on the duration of the frames.
If the two calculated frame rates are significantly different, the function returns an error
Args:
ffprobe_path (str): The path to the ffprobe binary.
input_file (IO[bytes]): The input video file.
Returns:
float | None:
- The frame rate of the video file as a floating-point number
- None if an error occurs while running the ffprobe binary or if the calculated
frame rates are inconsistent
"""
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
mean_duration = 0.
nb_frames1 = 0
nb_frames2 = 0
min_ts = None
max_ts = None
interlaced = False
params = [ffprobe_path, '-loglevel', 'quiet', '-select_streams', 'v', '-show_frames',
'-read_intervals', '00%+30', '-of', 'json', f'/proc/self/fd/{infd:d}']
env = {**os.environ, 'LANG': 'C'}
with Popen(params, stdout=PIPE, close_fds=False, env=env) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'frames' in out:
for frame in out['frames']:
if 'interlaced_frame' in frame:
if frame['interlaced_frame'] == 1:
interlaced = True
if 'pts_time' in frame:
ts = float(frame['pts_time'])
if min_ts is None:
min_ts = ts
if max_ts is None:
max_ts = ts
min_ts = min(min_ts, ts)
max_ts = max(max_ts, ts)
nb_frames1+=1
if 'duration_time' in frame:
mean_duration+=float(frame['duration_time'])
nb_frames2+=1
else:
return None
ffprobe.wait()
if ffprobe.returncode != 0:
logger.error("ffprobe returns an error code: %d", ffprobe.returncode)
return None
frame_rate1 = nb_frames1/(max_ts-min_ts)
frame_rate2 = nb_frames2 / mean_duration
if abs(frame_rate1 - frame_rate2) > 0.2:
if not interlaced:
logger.error('Video is not interlaced and the disperancy between frame rates is too \
big: %f / %f', frame_rate1, frame_rate2)
return None
if abs(frame_rate1*2 - frame_rate2) < 0.2:
return frame_rate2/2
logger.error('Video is interlaced and the disperancy between frame rates is too big:\
%f / %f', frame_rate1, frame_rate2)
return None
return frame_rate2
@typechecked
def get_subtitles_tracks(ffprobe_path:str, mkv_path: str) -> dict[str,str]|None:
logger = logging.getLogger(__name__)
tracks={}
with Popen([ffprobe_path, '-loglevel', 'quiet', '-select_streams', 's', '-show_entries',
'stream=index,codec_name:stream_tags=language', '-of', 'json', mkv_path],
stdout=PIPE) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'streams' in out:
for stream in out['streams']:
index = stream['index']
codec = stream['codec']
lang = stream['tags']['language']
if codec == 'dvd_subtitle':
if lang not in tracks:
tracks[lang] = [index]
else:
current_langs = tracks[lang]
current_langs.append(index)
tracks[lang] = current_langs
else:
return None
ffprobe.wait()
if ffprobe.returncode != 0:
logger.error("ffprobe returns an error code: %d", ffprobe.returncode)
return None
return tracks
@typechecked
def extract_srt(mkvextract:str, filename:str, subtitles:dict[str, list[int]],
langs:dict[Lang,str]) -> list[tuple[str,str,str,str]]|None:
logger = logging.getLogger(__name__)
params = [mkvextract, filename, 'tracks']
res = []
for lang in subtitles:
iso = Lang(lang)
if iso in langs:
ocrlang = langs[iso]
else:
logger.warning("Language not supported by Tesseract: %s", iso.name)
ocrlang ='osd'
if len(subtitles[lang]) == 1:
params.append(f'{subtitles[lang][0]:d}:{lang}')
res.append((f'{lang}.idx', f'{lang}.sub', lang, ocrlang))
else:
count = 1
for track in subtitles[lang]:
params.append(f'{track:d}:{lang}-{count:d}')
res.append((f'{lang}-{count:d}.idx', f'{lang}-{count:d}.sub', lang, ocrlang))
count = count+1
logger.debug('Executing %s', params)
env = {**os.environ, 'LANG': 'C'}
with Popen(params, stdout=PIPE, close_fds=False, env=env) as extract:
pb = tqdm(TextIOWrapper(extract.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
extract.wait()
# mkvextract returns 0, 1 or 2 as error code.
match extract.returncode:
case 0:
logger.info('Subtitle tracks were succesfully extracted.')
case 1:
logger.warning('Mkvextract returns warning')
case 2:
logger.error('Mkvextract returns an error code: %d', extract.returncode)
res = None
return res
@typechecked
def do_ocr(vobsubocr:str, idxs: list[tuple[str,str,str,str]], duration:timedelta,
temporaries:list[IO[bytes]], dump_mem_fd:bool=False):
logger = logging.getLogger(__name__)
res = []
for idx_name, _, lang, iso in idxs:
srtname = f'{os.path.splitext(idx_name)[0]}.srt'
# Tesseract seems to recognize the three dots ... as "su"
ldots = re.compile('^su\n$')
# Timestamps produced by vobsubocr: 01:52:19,861 --> 01:52:21,641
timestamps = re.compile((r'^[0-9]{2}:[0-9]{2}:[0-9]{2},[0-9]{3} \-\-> (?P<hours>[0-9]{2}):'
r'(?P<minutes>[0-9]{2}):(?P<seconds>[0-9]{2}),[0-9]{3}$'))
srtfd = memfd_create(srtname, flags=0)
with Popen([vobsubocr, '--lang', iso, idx_name], stdout=PIPE) as ocr:
pb = tqdm(TextIOWrapper(ocr.stdout, encoding="utf-8"), total=
int(duration/timedelta(seconds=1)), unit='s', desc='OCR')
for line in pb:
m = re.match(ldots,line)
if m is not None:
write(srtfd, '...'.encode(encoding='UTF-8'))
else:
write(srtfd, line.encode(encoding='UTF-8'))
m = re.match(timestamps, line)
if m is not None:
hours = int(m.group('hours'))
minutes = int(m.group('hours'))
seconds = int(m.group('seconds'))
ts = timedelta(hours=hours, minutes=minutes, seconds=seconds)
pb.n = int(ts/timedelta(seconds=1))
pb.update()
status = ocr.wait()
if status != 0:
logger.error('OCR failed with status code: %d', status)
if dump_mem_fd:
try:
with open(srtname,'w', encoding='utf8') as dump_srt:
lseek(srtfd, 0, SEEK_SET)
srt_length = fstat(srtfd).st_size
buf = read(srtfd, srt_length)
outfd = dump_srt.fileno()
pos = 0
while pos < srt_length:
pos+=write(outfd, buf[pos:])
temporaries.append(dump_srt)
except OSError:
logger.error('Impossible to create file: %s', srtname)
return None
srt_length = fstat(srtfd).st_size
if srt_length > 0:
res.append((srtfd, lang))
return res
@unique
class SupportedFormat(IntEnum):
TS = 1
MP4 = 2
MATROSKA = 3
def __str__(self):
match self:
case SupportedFormat.TS:
return 'mpegts'
case SupportedFormat.MP4:
return 'mov,mp4,m4a,3gp,3g2,mj2'
case SupportedFormat.MATROSKA:
return 'matroska,webm'
case _:
return 'Unsupported format'
# Extract SPS/PPS
# https://gitlab.com/mbunkus/mkvtoolnix/-/issues/2390
# ffmpeg -i <InputFile (before concatenation)> -c:v copy -an -sn -bsf:v trace_headers -t 0.01\
# -report -loglevel 0 -f null -
# Found codec private data using mkvinfo
@typechecked
def get_codec_private_data_from_mkv(mkvinfo_path:str,
input_file: IO[bytes]) -> tuple[int, bytes]|tuple[None,None]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
found = False
env = {**os.environ, 'LANG': 'C'}
# Output example
# Codec's private data: size 48 (H.264 profile: High @L4.0) hexdump 01 64 00 28 ff e1 00 1b 67\
# 64 00 28 ac d9 40 78 04 4f dc d4 04 04 05 00 00 92 ef 00 1d ad a6 1f 16 2d 96 01 00 06 68 fb\
# a3 cb 22 c0 fd f8 f8 00 at 406 size 51 data size 48
with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE,
close_fds=False, env=env) as mkvinfo:
out, _ = mkvinfo.communicate()
out = out.decode('utf8')
reg_exp = (r"^.*Codec's private data: size ([0-9]+) \(H.264.*\) hexdump "
r"(?P<hexdump>([0-9a-f]{2} )+)at (?P<position>[0-9]+) size (?P<size>[0-9]+).*$")
p = re.compile(reg_exp)
for line in out.splitlines():
m = p.match(line)
if m is not None:
size = int(m.group('size'))
position = int(m.group('position'))
logger.debug("Found codec private data at position: %s, size: %d", position, size)
found = True
mkvinfo.wait()
break
if found:
lseek(infd, position, SEEK_SET)
data = read(infd, size)
return position, data
return None, None
@typechecked
def parse_mkv_tree(mkvinfo_path:str, input_file: IO[bytes]) -> dict[str,tuple[int,int]]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
env = {**os.environ, 'LANG': 'C'}
elements = {}
with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE,
close_fds=False, env=env) as mkvinfo:
out, _ = mkvinfo.communicate()
out = out.decode('utf8')
prefix = []
reg_exp = (r"(^(?P<root>\+)|(\|(?P<depth>[ ]*\+))).*at (?P<position>[0-9]+)"
r" size (?P<size>[0-9]+).*$")
p = re.compile(reg_exp)
prev_depth = -1
for line in out.splitlines():
m = p.match(line)
if m is None:
logger.error("Impossible to match line: %s", line)
else:
position = int(m.group('position'))
size = int(m.group('size'))
root = m.group('root') is not None
if root:
depth = 0
else:
depth = len(m.group('depth'))
if depth > prev_depth:
for _ in range(depth-prev_depth):
prefix.append(1)
elif depth == prev_depth:
subid = prefix[-1]
subid+=1
prefix.pop()
prefix.append(subid)
else:
for _ in range(prev_depth-depth):
prefix.pop()
subid = prefix[-1]
subid+=1
prefix.pop()
prefix.append(subid)
prev_depth = depth
key=".".join(map(str, prefix))
elements[key] = (position, size)
mkvinfo.wait()
return elements
@typechecked
def change_codec_private_data(mkvinfo_path:str, input_file: IO[bytes], codec_data:bytes) -> None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
current_length = fstat(infd).st_size
logger.info('Current size of file: %d', current_length)
position, current_data = get_codec_private_data_from_mkv(mkvinfo_path, input_file)
current_data_length = len(current_data)
future_length = current_length - current_data_length + len(codec_data)
logger.info('Expected size of file: %d', future_length)
logger.info('Current data at position %d: %s', position, hexdump.dump(current_data, sep=":"))
logger.info('Future data: %s', hexdump.dump(codec_data, sep=":"))
elements = parse_mkv_tree(mkvinfo_path, input_file)
found = False
for key, (pos,size) in elements.items():
if pos == position:
logger.info('Codec private data key: %s', key)
found = True
break
if not found:
logger.error('Impossible to retrieve the key of codec private data')
exit(-1)
if current_length < future_length:
lseek(infd, position+current_data_length, SEEK_SET)
tail = read(infd, current_length-(position+current_data_length))
# We extend the file at the end with zeroes
ftruncate(infd, future_length)
lseek(infd, position+len(codec_data), SEEK_SET)
write(infd, tail)
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
elif current_length == future_length:
# Almost nothing to do except overwriting old private codec data with new ones.
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
else:
lseek(infd, position+current_data_length, SEEK_SET)
tail = read(infd, current_length-(position+current_data_length))
lseek(infd, position+len(codec_data), SEEK_SET)
write(infd, tail)
lseek(infd, position, SEEK_SET)
write(infd, codec_data)
# We reduce the length of file.
ftruncate(infd, future_length)
# We have to modify the tree elements up to the root that contains the codec private data.
keys = key.split('.')
logger.info(keys)
delta = future_length-current_length
# if there is no modification of the private codec data, no need to change anything.
if delta != 0:
for _ in range(len(keys)-1):
keys.pop()
key=".".join(map(str, keys))
pos, size = elements[key]
logger.info('Trying to fix element with key: %s at position: %d with actual size: %d.',
key, pos, size)
# Changing an element can increase its size (in very rare case).
# In that case, we update the new delta that will be larger (because the element has
# been resized).
delta+=change_ebml_element_size(input_file, pos, delta)
@typechecked
def get_format(ffprobe_path:str, input_file: IO[bytes]) -> dict|None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_format', '-of', 'json', '-i',
f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'format' in out:
return out['format']
else:
logger.error('Impossible to retrieve format of file')
return None
@typechecked
def get_movie_duration(ffprobe_path:str, input_file: IO[bytes]) -> timedelta|None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_format', '-of', 'json', '-i',
f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'format' in out and 'duration' in out['format']:
duration = floor(float(out['format']['duration']))
ts = timedelta(seconds=duration)
return ts
else:
logger.error('Impossible to retrieve duration of movie')
return None
# ffprobe -loglevel quiet -select_streams v:0 -show_entries stream=width,height -of json sample.ts
@typechecked
def get_video_dimensions(ffprobe_path:str, input_file: IO[bytes]) -> tuple[int,int]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
with Popen([ffprobe_path, '-loglevel', 'quiet', '-select_streams', 'v:0', '-show_entries',\
'stream=width,height', '-of', 'json', '-i', f'/proc/self/fd/{infd:d}'],\
stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'streams' in out:
video = out['streams'][0]
if ('width' in video) and ('height' in video):
return int(video['width']), int(video['height'])
logger.error('Impossible to retrieve dimensions of video')
exit(-1)
@typechecked
def get_streams(ffprobe_path:str, input_file: IO[bytes]) -> list|None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_streams', '-of', 'json', '-i',
f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'streams' in out:
return out['streams']
else:
logger.error('Impossible to retrieve streams inside file')
return None
@typechecked
def with_subtitles(ffprobe_path:str, input_file: IO[bytes]) -> bool:
"""
Checks if a media file contains subtitles using the ffprobe tool.
This function runs the ffprobe binary with the specified input file and parses the output
to determine if the file contains subtitles.
It returns True if at least one subtitle stream is found, False otherwise.
Args:
ffprobe_path (str): The path to the ffprobe binary.
input_file (IO[bytes]): The input media file.
Returns:
bool:
- True if the media file contains at least one subtitle stream
- False if:
- the media file does not contain any subtitle streams
- an error occurs while running the ffprobe binary
- the streams information cannot be retrieved from the media file
"""
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_streams', '-of', 'json', '-i',
f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'streams' in out:
streams = out['streams']
for stream in streams:
if 'codec_type' in stream and stream['codec_type'] == 'subtitle':
return True
else:
logger.error('Impossible to retrieve streams inside file')
return False
@typechecked
def parse_timestamp(ts:str) -> timedelta|None:
"""
Parse a timestamp string into a timedelta object.
This function takes a string representing a timestamp in the format HH:MM:SS[.us] and returns
a timedelta object representing the corresponding time interval.
The timestamp string can have an optional microsecond component.
Args:
ts (str): The timestamp string to parse.
Returns:
timedelta | None:
- A timedelta object representing the parsed timestamp
- None if:
- the timestamp string is not in the correct format
- the timestamp values are out of range (e.g. hour > 23, minute > 59, etc.)
"""
logger = logging.getLogger(__name__)
ts_reg_exp = (r'^(?P<hour>[0-9]{1,2}):(?P<minute>[0-9]{1,2})'
r':(?P<second>[0-9]{1,2})(\.(?P<us>[0-9]{1,6}))?$')
p = re.compile(ts_reg_exp)
m = p.match(ts)
if m is None:
logger.warning("Impossible to parse timestamp: %s", ts)
return None
values = m.groupdict()
hour = 0
minute = 0
second = 0
us = 0
if values['hour'] is not None:
hour = int(values['hour'])
if values['minute'] is not None:
minute = int(values['minute'])
if values['second'] is not None:
second = int(values['second'])
if values['us'] is not None:
us = int(values['us'])
if hour < 0 or hour > 23:
logger.error("hour must be in [0,24[")
return None
if minute < 0 or minute > 59:
logger.error("minute must be in [0,60[")
return None
if second < 0 or second > 59:
logger.error("second must be in [0,60[")
return None
if us < 0 or us > 1000000:
logger.error("milliseconds must be in [0,1000000[")
return None
res = timedelta(hours=hour, minutes=minute, seconds=second, microseconds=us)
return res
@typechecked
def parse_time_interval(interval: str) -> tuple[timedelta, timedelta] | None:
"""
Parse a time interval string into a tuple of two timedelta objects.
This function takes a string representing a time interval in the format HH:MM:SS[.ms]-HH:MM:SS[.ms] and returns a tuple of two timedelta objects representing the start and end times of the interval.
The time interval string can have an optional millisecond component.
Args:
interval (str): The time interval string to parse.
Returns:
tuple[timedelta, timedelta] | None:
- A tuple of two timedelta objects representing the start and end times of the interval
- None if:
- the time interval string is not in the correct format
- the time values are out of range (e.g. hour > 23, minute > 59, etc.)
- the end time is before the start time (non-monotonic interval)
"""
logger = logging.getLogger(__name__)
interval_reg_exp = (r'^(?P<hour1>[0-9]{1,2}):(?P<minute1>[0-9]{1,2}):(?P<second1>[0-9]{1,2})'
r'(\.(?P<ms1>[0-9]{1,3}))?-(?P<hour2>[0-9]{1,2}):(?P<minute2>[0-9]{1,2})'
r':(?P<second2>[0-9]{1,2})(\.(?P<ms2>[0-9]{1,3}))?$')
p = re.compile(interval_reg_exp)
m = p.match(interval)
if m is None:
logger.error("Impossible to parse time interval")
return None
values = m.groupdict()
hour1 = 0
minute1 = 0
second1 = 0
ms1 = 0
hour2 = 0
minute2 = 0
second2 = 0
ms2 = 0
if values['hour1'] is not None:
hour1 = int(values['hour1'])
if values['minute1'] is not None:
minute1 = int(values['minute1'])
if values['second1'] is not None:
second1 = int(values['second1'])
if values['ms1'] is not None:
ms1 = int(values['ms1'])
if values['hour2'] is not None:
hour2 = int(values['hour2'])
if values['minute2'] is not None:
minute2 = int(values['minute2'])
if values['second2'] is not None:
second2 = int(values['second2'])
if values['ms2'] is not None:
ms2 = int(values['ms2'])
if hour1 < 0 or hour1 > 23:
logger.error("hour must be in [0,24[")
return None, None
if minute1 < 0 or minute1 > 59:
logger.error("minute must be in [0,60[")
return None, None
if second1 < 0 or second1 > 59:
logger.error("second must be in [0,60[")
return None, None
if ms1 < 0 or ms1 > 1000:
logger.error("milliseconds must be in [0,1000[")
return None, None
if hour2 < 0 or hour2 > 23:
logger.error("hour must be in [0,24[")
return None, None
if minute2 < 0 or minute2 > 59:
logger.error("minute must be in [0,60[")
return None, None
if second2 < 0 or second2 > 59:
logger.error("second must be in [0,60[")
return None, None
if ms2 < 0 or ms2 > 1000:
logger.error("milliseconds must be in [0,1000[")
return None, None
ts1 = timedelta(hours=hour1, minutes=minute1, seconds=second1, microseconds=ms1*1000)
ts2 = timedelta(hours=hour2, minutes=minute2, seconds=second2, microseconds=ms2*1000)
if ts2 < ts1:
logger.error("Non monotonic interval")
return None,None
return (ts1, ts2)
@typechecked
def compare_time_interval(interval1: tuple[timedelta, timedelta],
interval2: tuple[timedelta, timedelta]) -> int:
"""
Compare two time intervals.
This function compares two time intervals represented by tuples of two timedelta objects.
It returns an integer indicating the relationship between the two intervals:
- -1 if interval 1 is before interval 2
- 1 if interval 1 is after interval 2
- 0 if the two intervals overlap or are equal
Args:
interval1 (tuple[timedelta, timedelta]): The first time interval
interval2 (tuple[timedelta, timedelta]): The second time interval
Returns:
int: The relationship between the two time intervals
"""
ts11,ts12 = interval1
ts21,ts22 = interval2
if ts12 < ts21:
return -1
elif ts22 < ts11:
return 1
else:
return 0
@typechecked
def ffmpeg_convert(ffmpeg_path:str, ffprobe_path:str, input_file: IO[bytes], input_format:str,
output_file: IO[bytes], output_format:str, duration: timedelta):
logger = logging.getLogger(__name__)
width, height = get_video_dimensions(ffprobe_path, input_file)
subtitles = with_subtitles(ffprobe_path, input_file)
infd = input_file.fileno()
outfd = output_file.fileno()
set_inheritable(infd, True)
set_inheritable(outfd, True)
if logger.getEffectiveLevel() == logging.DEBUG:
log = []
else:
log = [ '-loglevel', 'quiet' ]
params = [ffmpeg_path, '-y',]+log+['-progress', '/dev/stdout', '-canvas_size',
f'{width:d}x{height:d}', '-f', input_format,
'-i', f'/proc/self/fd/{infd:d}', '-map', '0:v',
'-map', '0:a']
if subtitles:
params.extend(['-map', '0:s'])
params.extend(['-bsf:v', 'h264_mp4toannexb,dump_extra=freq=keyframe', '-vcodec', 'copy',
'-acodec', 'copy'])
if subtitles:
params.extend(['-scodec', 'dvdsub'])
params.extend(['-r:0', '25', '-f', output_format, f'/proc/self/fd/{outfd:d}'])
logger.debug('Executing %s', params)
with Popen(params, stdout=PIPE, close_fds=False) as ffmpeg:
pb = tqdm(TextIOWrapper(ffmpeg.stdout, encoding="utf-8"),
total=int(duration/timedelta(seconds=1)), unit='s', desc='Conversion')
for line in pb:
if line.startswith('out_time='):
ts = line.split('=')[1].strip()
ts = parse_timestamp(ts)
if ts is not None:
pb.n = int(ts/timedelta(seconds=1))
pb.update()
status = ffmpeg.wait()
if status != 0:
logger.error('Conversion failed with status code: %d', status)
@typechecked
def get_ts_frame(frame: dict) -> timedelta|None:
logger = logging.getLogger(__name__)
if 'pts_time' in frame:
pts_time = float(frame['pts_time'])
elif 'pkt_pts_time' in frame:
pts_time = float(frame['pkt_pts_time'])
else:
logger.error('Impossible to find timestamp of frame %s', frame)
return None
ts = timedelta(seconds=pts_time)
return ts
@typechecked
def get_packet_duration(packet: dict) -> int:
logger = logging.getLogger(__name__)
if 'duration' in packet:
duration = int(packet['duration'])
elif 'pkt_duration' in packet:
duration = int(packet['pkt_duration'])
else:
logger.error('Impossible to find duration of packet %s', packet)
return None
return duration
@typechecked
def get_frames_in_stream(ffprobe_path: str, input_file: IO[bytes], begin:timedelta, end:timedelta,
stream_kind:str, sub_stream_id:int=0) -> list[dict]|None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
set_inheritable(infd, True)
command = [ffprobe_path, '-loglevel', 'quiet', '-read_intervals', f'{begin}%{end}',
'-show_entries', 'frame', '-select_streams',
f'{stream_kind}:{sub_stream_id:d}','-of', 'json', f'/proc/self/fd/{infd:d}']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
frames = json.load(BytesIO(out))
status = ffprobe.wait()
if status != 0:
logger.error('ffprobe failed with status code: %d', status)
return None
# Sort frames by timestamp
tmp = {}
if 'frames' in frames:
frames = frames['frames']
for frame in frames:
ts = get_ts_frame(frame)
if ts is None:
return None
if begin <= ts <= end:
tmp[ts]=frame
res = []
for ts in sorted(tmp):
res.append(tmp[ts])
return res
else:
logger.error('Impossible to retrieve frames inside file around [%s,%s]', begin, end)
return None
# TODO: Finish implementation of this function and use it.
@typechecked
def get_nearest_idr_frame(ffprobe_path: str, input_file: IO[bytes], timestamp:timedelta,
before: bool=True, delta: timedelta=timedelta(seconds=2)):
# pylint: disable=W0613
logger = logging.getLogger(__name__)
zero = timedelta()
tbegin = timestamp-delta
tend = timestamp+delta
tbegin = max(tbegin, zero)
infd = input_file.fileno()
set_inheritable(infd, True)
logger.debug('Looking for IDR frame in [%s, %s]', tbegin, tend)
idrs = []
# Retains only IDR frame
with Popen([ffprobe_path, '-loglevel', 'quiet', '-read_intervals', f'{tbegin}%{tend}',
'-skip_frame', 'nokey', '-show_entries', 'frame', '-select_streams', 'v:0',
'-of', 'json', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
frames = json.load(BytesIO(out))
status = ffprobe.wait()
if status != 0:
logger.error('ffprobe failed with status code: %d', status)
return None
if 'frames' in frames:
frames = frames['frames']
for frame in frames:
ts = get_ts_frame(frame)
if ts is None:
return None
if tbegin <= ts <= tend:
idrs.append(frame)
else:
logger.error('Impossible to retrieve IDR frames inside file around [%s,%s]',
tbegin, tend)
return None
return None
@typechecked
def get_nearest_iframe(ffprobe_path:str, input_file: IO[bytes],
timestamp:timedelta, before:bool=True,
delta_max:timedelta=timedelta(seconds=15))-> tuple[int,dict]:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
set_inheritable(infd, True)
delta = timedelta(seconds=1)
iframe = None
while delta < delta_max:
zero = timedelta()
if before:
tbegin = timestamp-delta
else:
tbegin = timestamp
if not before:
tend = timestamp+delta
else:
tend = timestamp
tbegin = max(tbegin, zero)
logger.debug('Looking for an iframe in [%s, %s]', tbegin, tend)
frames = get_frames_in_stream(ffprobe_path, input_file=input_file, begin=tbegin, end=tend,
stream_kind='v')
if frames is None:
logger.debug('Found no frame in [%s, %s]', tbegin, tend)
delta+=timedelta(seconds=1)
continue
iframes = []
for frame in frames:
if frame['pict_type'] == 'I':
iframes.append(frame)
found = False
for frame in iframes:
ts = get_ts_frame(frame)
if ts is None:
logger.warning('I-frame with no timestamp: %s', frame)
continue
if before and ts <= timestamp:
found = True
iframe = frame
if not before and ts >= timestamp:
found = True
iframe = frame
break
if found:
logger.info("Found i-frame at: %s", iframe)
break
else:
delta+=timedelta(seconds=1)
continue
if iframe is not None:
its = get_ts_frame(iframe)
nb_frames = 0
for frame in frames:
ts = get_ts_frame(frame)
if ts is None:
logger.warning('Frame without timestamp: %s', frame)
continue
if before:
if its <= ts <= timestamp:
logger.info("Retrieve a frame between %s and %s at %s", its, timestamp, ts)
nb_frames = nb_frames+1
else:
if timestamp <= ts <= its:
logger.info("Retrieve a frame between %s and %s at %s", timestamp, ts, its)
nb_frames = nb_frames+1
else:
logger.error("Impossible to find I-frame between: %s and %s", tbegin, tend)
return 0, None
return(nb_frames, iframe)
@typechecked
def extract_mkv_part(mkvmerge_path:str, input_file:IO[bytes], output_file:IO[bytes],
begin:timedelta, end:timedelta) -> None:
logger = logging.getLogger(__name__)
logger.info('Extract video between I-frames at %s and %s', begin,end)
infd = input_file.fileno()
outfd = output_file.fileno()
lseek(infd, 0, SEEK_SET)
lseek(outfd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
env = {**os.environ, 'LANG': 'C'}
warnings = []
command = [mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '--split', f'parts:{begin}-{end}',
f'/proc/self/fd/{infd:d}']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
elif line.startswith('Warning'):
warnings.append(line)
pb.update(100-pb.n)
pb.refresh()
pb.close()
status = mkvmerge.wait()
if status == 1:
logger.warning('Extraction returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Extraction returns errors')
@typechecked
def extract_pictures(ffmpeg_path:str, input_file:IO[bytes], begin:timedelta, nb_frames:int,
width:int=640, height:int=480) -> tuple[bytes,int]|tuple[None,None]:
"""
Extract pictures from a video file using FFmpeg.
This function runs the FFmpeg binary to extract a specified number of frames from a video file,
starting at a given time.
The extracted frames are stored in memory as PPM images and returned as a tuple containing
the image data and a file descriptor to the memory created by memfd_create.
Args:
ffmpeg_path (str): The path to the FFmpeg binary.
input_file (IO[bytes]): The input video file.
begin (timedelta): The start time of the extraction.
nb_frames (int): The number of frames to extract.
width (int, optional): The width of the extracted images. Defaults to 640.
height (int, optional): The height of the extracted images. Defaults to 480.
Returns:
tuple[bytes, int] | tuple[None, None]:
- A tuple containing the extracted image data as bytes and a file descriptor
- A tuple containing None, None if the extraction fails
"""
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
outfd = memfd_create('pictures', flags=0)
set_inheritable(outfd, True)
# PPM header
# "P6\nWIDTH HEIGHT\n255\n"
header_len=2+1+ceil(log(width, 10))+1+ceil(log(height, 10))+1+3+1
logger.debug('Header length: %d', header_len)
image_length = width*height*3+header_len
length = image_length*nb_frames
logger.debug("Estimated length: %d", length)
command = [ffmpeg_path, '-loglevel', 'quiet' ,'-y', '-ss', f'{begin}', '-i',
f'/proc/self/fd/{infd}', '-s', f'{width:d}x{height:d}', '-vframes', f'{nb_frames:d}',
'-c:v', 'ppm','-f', 'image2pipe', f'/proc/self/fd/{outfd:d}']
logger.debug('Executing: %s', command)
images = b''
with Popen(command, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
logger.error('Conversion failed with status code: %d', status)
return None, None
lseek(outfd, 0, SEEK_SET)
images = read(outfd,length)
if len(images) != length:
logger.error("Received %d bytes but %d were expected.", len(images), length)
return None, None
lseek(outfd, 0, SEEK_SET)
return images, outfd
@typechecked
def extract_sound(ffmpeg_path:str, input_file: IO[bytes], begin:timedelta, output_filename:str,
packet_duration:int, sub_channel:int=0,
nb_packets:int=0, sample_rate:int=48000,
nb_channels:int=2) -> tuple[bytes,int]|tuple[None,None]:
logger = logging.getLogger(__name__)
outfd = memfd_create(output_filename, flags=0)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
sound = b''
length = int(nb_channels*sample_rate*4*nb_packets*packet_duration/1000)
command = [ffmpeg_path, '-y', '-loglevel', 'quiet', '-ss', f'{begin}',
'-i', f'/proc/self/fd/{infd}', f'-frames:a:{sub_channel:d}', f'{nb_packets+1:d}',
'-c:a', 'pcm_s32le', '-sample_rate', f'{sample_rate:d}',
'-channels', f'{nb_channels:d}', '-f', 's32le', f'/proc/self/fd/{outfd:d}']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
logger.error('Sound extraction returns error code: %d', status)
return None, None
lseek(outfd, 0, SEEK_SET)
sound = read(outfd, length)
if len(sound) != length:
logger.info("Received %d bytes but %d were expected (channels=%d, freq=%d, packets=%d,\
duration=%d ms).", len(sound), length, nb_channels, sample_rate, nb_packets,
packet_duration)
return None, None
return sound, outfd
@typechecked
def dump_ppm(pictures: bytes, prefix: str, temporaries: list[IO[bytes]]) -> None:
"""
Dump PPM pictures from a bytes buffer to files.
This function takes a bytes buffer containing PPM pictures, a prefix for the output file names, and a list of temporary files.
It extracts each PPM picture from the buffer, checks its validity, and writes it to a file.
The output files are named according to the prefix and a zero-padded three-digit number.
Args:
pictures (bytes): The bytes buffer containing the PPM pictures.
prefix (str): The prefix for the output file names.
temporaries (list[IO[bytes]]): A list of temporary files that will be used to store the output files.
Returns:
None
Raises:
None, but logs errors if:
- the PPM picture is not valid (e.g. wrong magic number, dimensions, or color encoding)
- an I/O error occurs while creating or writing to an output file
"""
logger = logging.getLogger(__name__)
# "P6\nWIDTH HEIGHT\n255\n"
pos = 0
picture = 0
logger.debug('Dumping %d pictures: %s', len(pictures),prefix)
while pos<len(pictures):
filename = f'{prefix}-{picture:03d}.ppm'
header = BytesIO(pictures[pos:])
magic = header.readline().decode('utf8')
dimensions = header.readline().decode('utf8')
max_value = int(header.readline().decode('utf8'))
if magic == 'P6\n':
pattern = re.compile('^(?P<width>[0-9]+) (?P<height>[0-9]+)\n$')
m = pattern.match(dimensions)
if m is not None:
width = int(m['width'])
height = int(m['height'])
else:
logger.error('Impossible to parse dimensions of picture')
return
else:
logger.error('Not a PPM picture')
return
if max_value != 255:
logger.error('Not a valid PPM picture. Color are not encoded on byte. Max value: %d',
max_value)
header_len=2+1+ceil(log(width, 10))+1+ceil(log(height, 10))+1+3+1
try:
with open(filename, 'wb') as out:
temporaries.append(out)
outfd = out.fileno()
length=header_len+3*width*height
nb_bytes = 0
while nb_bytes < length:
nb_bytes+=write(outfd, pictures[pos+nb_bytes:pos+length])
pos+=length
picture+=1
except OSError:
logger.error('Impossible to create file: %s', filename)
@typechecked
def extract_all_streams(ffmpeg_path:str, ffprobe_path:str, input_file:IO[bytes], begin:timedelta,
end:timedelta, streams, files_prefix, nb_frames:int, framerate:float,
width:int, height:int, temporaries, dump_mem_fd:bool=False):
logger = logging.getLogger(__name__)
# The command line for encoding only video track
video_encoder_params = [ ffmpeg_path, '-y', '-loglevel', 'quiet']
video_input_params = []
video_codec_params = []
# The command line to create a MKV file with the rest of tracks
generic_encoder_params = [ ffmpeg_path, '-y', '-loglevel', 'quiet' ]
generic_input_params = []
generic_codec_params = []
if begin < end:
video_id=0
audio_id=0
subtitle_id=0
memfds = []
for stream in streams:
if stream['codec_type'] == 'video':
logger.info("Extracting %d frames of video stream v:%d", nb_frames, video_id)
sar = stream['sample_aspect_ratio']
dar = stream['display_aspect_ratio']
pixel_format = stream['pix_fmt']
color_range = stream['color_range']
color_space =stream['color_space']
color_transfer = stream['color_transfer']
color_primaries = stream['color_primaries']
level = int(stream['level'])
level = f'{floor(level/10):d}.{level%10:d}'
chroma_location = stream['chroma_location']
field_order = stream
match field_order:
case 'progressive':
interlaced_options = ['-field_order', '0']
case 'tt':
interlaced_options = ['-top', '1', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order', '1']
case 'bb':
interlaced_options = ['-top', '0', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order','2']
case 'tb':
interlaced_options = ['-top', '1', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order', '3']
case 'bt':
interlaced_options = ['-top', '0', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order', '4']
case _:
interlaced_options = []
# ======================================= #
# TODO: adjust SAR and DAR
# https://superuser.com/questions/907933/correct-aspect-ratio-without-re-encoding-video-file
# SAR: -aspect width:height
# DAR: -bsf:v sample_aspect_ratio=1:video_format
logger.warning('Missing SAR adjustment for: %s', sar)
logger.warning('Missing DAR adjustment for: %s', dar)
logger.warning('Missing treatment for chroma location: %s', chroma_location)
codec = stream['codec_name']
images_bytes, memfd = extract_pictures(ffmpeg_path, input_file=input_file,
begin=begin, nb_frames=nb_frames,
width=width, height=height)
if images_bytes is None:
logger.error('Impossible to extract picture from video stream.')
exit(-1)
memfds.append(memfd)
if dump_mem_fd:
dump_ppm(images_bytes, f'{files_prefix}-{video_id:d}', temporaries)
# We rewind to zero the memory file descriptor
lseek(memfd, 0, SEEK_SET)
set_inheritable(memfd, True)
video_input_params.extend(['-framerate', f'{framerate:f}', '-f', 'image2pipe', '-i',
f'/proc/self/fd/{memfd:d}'])
video_codec_params.extend([f'-c:v:{video_id:d}', codec, f'-level:v:{video_id:d}',
level, '-pix_fmt', pixel_format])
video_codec_params.extend(interlaced_options)
video_codec_params.extend([f'-colorspace:v:{video_id}', color_space,
f'-color_primaries:v:{video_id:d}', color_primaries,
f'-color_trc:v:{video_id:d}', color_transfer,
f'-color_range:v:{video_id:d}', color_range])
video_id=video_id+1
elif stream['codec_type'] == 'audio':
logger.debug('Audio stream: %s', stream)
sample_rate = int(stream['sample_rate'])
nb_channels = int(stream['channels'])
if 'bit_rate' in stream:
bit_rate = int(stream['bit_rate'])
else:
bit_rate = 128000
codec = stream['codec_name']
if 'tags' in stream:
if 'language' in stream['tags']:
generic_codec_params.extend([f'-metadata:s:a:{audio_id:d}',
f"language={stream['tags']['language']}"])
packets = get_frames_in_stream(ffprobe_path, input_file=input_file, begin=begin,
end=end, stream_kind='a', sub_stream_id=audio_id)
nb_packets = len(packets)
logger.debug("Found %d packets to be extracted from audio track.", nb_packets)
if nb_packets > 0:
packet_duration = get_packet_duration(packets[0])
if packet_duration is None:
return None
else:
packet_duration = 0
logger.info("Extracting %d packets of audio stream: a:%d" , nb_packets, audio_id)
tmpname = f'{files_prefix}-{audio_id:d}.pcm'
sound_bytes, memfd = extract_sound(ffmpeg_path=ffmpeg_path, input_file=input_file,
begin=begin, nb_packets=nb_packets,
packet_duration=packet_duration,
output_filename=tmpname,
sample_rate=sample_rate, nb_channels=nb_channels)
if sound_bytes is None:
logger.error('Impossible to extract sound track')
exit(-1)
memfds.append(memfd)
if dump_mem_fd:
try:
with open(tmpname,'wb') as output:
temporaries.append(output)
outfd = output.fileno()
pos = 0
while pos < len(sound_bytes):
pos+=write(outfd, sound_bytes[pos:])
except OSError:
logger.error('Impossible to create file: %s', tmpname)
return None
# We rewind to zero the memory file descriptor
lseek(memfd, 0, SEEK_SET)
set_inheritable(memfd, True)
generic_input_params.extend(['-f', 's32le', '-ar', f'{sample_rate:d}', '-ac',
f'{nb_channels:d}', '-i', f'/proc/self/fd/{memfd:d}'])
generic_codec_params.extend([f'-c:a:{audio_id:d}', codec, f'-b:a:{audio_id:d}',
f'{bit_rate:d}'])
audio_id=audio_id+1
elif stream['codec_type'] == 'subtitle':
logger.info("Extracting a subtitle stream: s:%d", subtitle_id)
codec = stream['codec_name']
generic_input_params.extend(['-i', './empty.idx'])
if 'tags' in stream:
if 'language' in stream['tags']:
generic_codec_params.extend([f'-metadata:s:s:{subtitle_id:d}',
f"language={stream['tags']['language']}"])
generic_codec_params.extend([f'-c:s:{subtitle_id:d}', 'copy'])
subtitle_id=subtitle_id+1
else:
logger.error("Unknown stream type: %s", stream['codec_type'])
# Create a new MKV movie with all streams (except videos) that have been extracted.
generic_encoder_params.extend(generic_input_params)
for index in range(audio_id+subtitle_id):
generic_encoder_params.extend(['-map', f'{index:d}'])
generic_encoder_params.extend(generic_codec_params)
mkv_filename = f'{files_prefix}.mkv'
try:
mkv_output = open(mkv_filename,'wb+')
except OSError:
logger.error('Impossible to create file: %s', mkv_filename)
return None
mkvoutfd = mkv_output.fileno()
set_inheritable(mkvoutfd, True)
generic_encoder_params.extend(['-f', 'matroska', f'/proc/self/fd/{mkvoutfd:d}'])
logger.info('Encoding all streams (except video) into a MKV file: %s', mkv_filename)
logger.debug('Executing: %s', generic_encoder_params)
with Popen(generic_encoder_params, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
logger.error('Encoding failed with status code: %d', status)
return None
temporaries.append(mkv_output)
h264_filename = f'{files_prefix}.h264'
try:
h264_output = open(h264_filename,'wb+')
except OSError:
logger.error('Impossible to create file: %s', h264_filename)
return None
h264outfd = h264_output.fileno()
set_inheritable(h264outfd, True)
video_encoder_params.extend(video_input_params)
video_encoder_params.extend(video_codec_params)
video_encoder_params.extend([ '-x264opts', f'keyint=1:sps-id={1:d}','-bsf:v',
'h264_mp4toannexb,dump_extra=freq=keyframe,h264_metadata=\
overscan_appropriate_flag=1:sample_aspect_ratio=1:video_format=\
0:chroma_sample_loc_type=0','-f', 'h264',
f'/proc/self/fd/{h264outfd:d}'])
logger.info('Encoding video into a H264 file: %s', h264_filename)
logger.debug('Executing: %s', video_encoder_params)
with Popen(video_encoder_params, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
logger.error('Encoding failed with status code: %d', status)
return None
temporaries.append(h264_output)
h264_ts_filename = f'{files_prefix}-ts.txt'
try:
h264_ts_output = open(h264_ts_filename,'w+', encoding='utf8')
except OSError:
logger.error('Impossible to create file: %s', h264_ts_filename)
return None
h264_ts_output.write('# timestamp format v2\n')
ts = 0
for _ in range(nb_frames):
ts = ts+ceil(1000/framerate)
h264_ts_output.write(f'{ts:d}\n')
h264_ts_output.flush()
h264_ts_output.seek(0)
temporaries.append(h264_ts_output)
for memfd in memfds:
close(memfd)
return h264_output, h264_ts_output, mkv_output
else:
# Nothing to be done. We are already at a i-frame boundary.
return None, None
# Merge a list of mkv files passed as input, and produce a new MKV as output
@typechecked
def merge_mkvs(mkvmerge_path:str, inputs: list[IO[bytes]], output_name:str,
concatenate: bool=True, timestamps: dict[int, IO[str]] | None = None) -> IO[bytes]|None:
logger = logging.getLogger(__name__)
if timestamps is None:
timestamps = {}
fds = []
try:
out = open(output_name, 'wb+')
except OSError:
logger.error('Impossible to create file: %s', output_name)
return None
outfd = out.fileno()
lseek(outfd, 0, SEEK_SET)
fds.append(outfd)
set_inheritable(outfd, True)
# Timestamps of merged tracks are modified by the length of the preceding track.
# The default mode ('file') is using the largest timestamp of the whole file which may create
# desynchronize video and sound.
merge_params = [mkvmerge_path, '--append-mode', 'track']
first = True
partnum = 0
for mkv in inputs:
if mkv is not None:
fd = mkv.fileno()
fds.append(fd)
set_inheritable(fd, True)
# If we pass a timestamps file associated with the considered track, use it.
if partnum in timestamps:
tsfd = timestamps[partnum].fileno()
lseek(tsfd, 0, SEEK_SET)
fds.append(tsfd)
set_inheritable(tsfd, True)
merge_params.extend(['--timestamps', f'{partnum:d}:/proc/self/fd/{tsfd:d}'])
if first:
merge_params.append(f'/proc/self/fd/{fd:d}')
first = False
elif concatenate:
merge_params.append(f'+/proc/self/fd/{fd:d}')
else:
merge_params.append(f'/proc/self/fd/{fd:d}')
partnum+=1
merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}'])
# We merge all files.
warnings = []
env = {**os.environ, 'LANG': 'C'}
logger.debug('Executing: LANG=C %s', merge_params)
with Popen(merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Merging')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.n = int(m['progress'])
pb.update()
elif line.startswith('Warning'):
warnings.append(line)
status = mkvmerge.wait()
if status == 1:
logger.warning('Extraction returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Extraction returns errors')
for fd in fds:
set_inheritable(fd, False)
return out
def find_subtitles_tracks(ffprobe_path:str, input_file: IO[bytes]) -> dict|None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
command = [ffprobe_path, '-loglevel','quiet', '-i', f'/proc/self/fd/{infd:d}',
'-select_streams', 's', '-show_entries', 'stream=index:stream_tags=language',
'-of', 'json']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False) as ffprobe:
out, _ = ffprobe.communicate()
out = json.load(BytesIO(out))
if 'streams' in out:
return out['streams']
else:
logger.error('Impossible to retrieve format of file')
ffprobe.wait()
return None
@typechecked
def extract_track_from_mkv(mkvextract_path: str, input_file: IO[bytes], index,
output_file: IO[bytes], timestamps) -> None:
logger = logging.getLogger(__name__)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
outfd = output_file.fileno()
lseek(outfd, 0, SEEK_SET)
set_inheritable(outfd, True)
tsfd = timestamps.fileno()
lseek(tsfd, 0, SEEK_SET)
set_inheritable(tsfd, True)
params = [ mkvextract_path, f'/proc/self/fd/{infd:d}', 'tracks',
f'{index:d}:/proc/self/fd/{outfd:d}', 'timestamps_v2',
f'{index:d}:/proc/self/fd/{tsfd:d}']
env = {**os.environ, 'LANG': 'C'}
logger.debug('Executing: LANG=C %s', params)
with Popen(params, stdout=PIPE, close_fds=False, env=env) as extract:
pb = tqdm(TextIOWrapper(extract.stdout, encoding="utf-8"), total=100, unit='%',
desc='Extraction of track')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
extract.wait()
if extract.returncode != 0:
logger.error('Mkvextract returns an error code: %d', extract.returncode)
else:
logger.info('Track %d was succesfully extracted.', index)
@typechecked
def remove_video_tracks_from_mkv(mkvmerge_path:str, input_file: IO[bytes],
output_file: IO[bytes]) -> None:
logger = logging.getLogger(__name__)
outfd = output_file.fileno()
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
lseek(outfd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
params = [ mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '-D', f'/proc/self/fd/{infd:d}']
logger.debug('Executing: LANG=C %s', params)
env = {**os.environ, 'LANG': 'C'}
with Popen(params, stdout=PIPE, close_fds=False, env=env) as remove:
pb = tqdm(TextIOWrapper(remove.stdout, encoding="utf-8"), total=100, unit='%',
desc='Removal of video track:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.update(int(m['progress'])-pb.n)
pb.update(100-pb.n)
pb.refresh()
pb.close()
remove.wait()
if remove.returncode != 0:
logger.error('Mkvmerge returns an error code: %d', remove.returncode)
else:
logger.info('Video tracks were succesfully extracted.')
@typechecked
def remux_srt_subtitles(mkvmerge_path:str, input_file: IO[bytes], output_filename: str,
subtitles) -> None:
logger = logging.getLogger(__name__)
try:
out = open(output_filename, 'w', encoding='utf8')
except OSError:
logger.error('Impossible to create file: %s', output_filename)
return None
outfd = out.fileno()
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
mkv_merge_params = [mkvmerge_path, f'/proc/self/fd/{infd:d}']
for fd, lang in subtitles:
lseek(fd, 0, SEEK_SET)
set_inheritable(fd, True)
mkv_merge_params.extend(['--language', f'0:{lang}', f'/proc/self/fd/{fd:d}'])
mkv_merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}'])
warnings = []
env = {**os.environ, 'LANG': 'C'}
logger.info('Remux subtitles: %s', mkv_merge_params)
with Popen(mkv_merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge:
pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%',
desc='Remux subtitles:')
for line in pb:
if line.startswith('Progress :'):
p = re.compile('^Progress : (?P<progress>[0-9]{1,3})%$')
m = p.match(line)
if m is None:
logger.error('Impossible to parse progress')
pb.n = int(m['progress'])
pb.update()
elif line.startswith('Warning'):
warnings.append(line)
status = mkvmerge.wait()
if status == 1:
logger.warning('Remux subtitles returns warning')
for w in warnings:
logger.warning(w)
elif status == 2:
logger.error('Remux subtitles returns errors')
return None
@typechecked
def concatenate_h264_parts(h264parts: list[IO[bytes]], output: IO[bytes]) -> None:
logger = logging.getLogger(__name__)
total_length = 0
for h264 in h264parts:
fd = h264.fileno()
total_length += fstat(fd).st_size
logger.info('Total length: %d', total_length)
outfd = output.fileno()
lseek(outfd, 0, SEEK_SET)
pb = tqdm(total=total_length, unit='bytes', desc='Concatenation')
for h264 in h264parts:
fd = h264.fileno()
lseek(fd, 0, SEEK_SET)
while True:
buf = read(fd, 1000000)
if buf is None or len(buf) == 0:
break
pos = 0
while pos < len(buf):
nb_bytes = write(outfd, buf[pos:])
pb.update(nb_bytes)
pos += nb_bytes
def concatenate_h264_ts_parts(h264_ts_parts: list[IO[bytes]], output: IO[bytes]) -> None:
logger = logging.getLogger(__name__)
header = '# timestamp format v2\n'
output.write(header)
last = 0.
first = True
for part in h264_ts_parts:
if first:
offset = last
else:
# TODO: take framerate into account
offset = last + 40
logger.debug('Parsing file: %s. Offset=%d', part, offset)
isheader = part.readline()
if (not isheader) or (isheader != header):
logger.error('Impossible to find a valid header: "%s"', isheader)
exit(-1)
while True:
line = part.readline()
if not line:
break
ts = offset + float(line)
last = max(last,ts)
output.write(f'{ts:f}\n')
if first:
first = False
# TODO: finish this procedure
def do_coarse_processing(ffmpeg_path:str, ffprobe_path:str, mkvmerge_path:str,
input_file: IO[bytes], begin, end, nb_frames, framerate,
files_prefix, streams, width, height, temporaries, dump_mem_fd) -> None:
# pylint: disable=W0613
logger = logging.getLogger(__name__)
# Internal video with all streams (video, audio and subtitles)
internal_mkv_name = f'{files_prefix}.mkv'
try:
internal_mkv = open(internal_mkv_name, 'wb+')
except OSError:
logger.error('Impossible to create file: %s', internal_mkv_name)
exit(-1)
# Extract internal part of MKV
extract_mkv_part(mkvmerge_path=mkvmerge_path, input_file=input_file, output_file=internal_mkv,
begin=begin, end=end)
temporaries.append(internal_mkv)