#!/usr/bin/env python3 '''A module to remove parts of video (.e.g advertisements) with single frame precision.''' # Standard modules import json import logging import os.path import re from datetime import timedelta from enum import IntEnum, unique from io import BytesIO, TextIOWrapper from math import ceil, floor, log from os import ( SEEK_SET, close, fstat, ftruncate, lseek, memfd_create, read, set_inheritable, write, ) from shutil import which from subprocess import PIPE, Popen from sys import exit from typing import IO # Third party libraries import hexdump from iso639 import Lang from iso639.exceptions import InvalidLanguageValue from tqdm import tqdm from typeguard import typechecked from tscut.h264.avc import (dump_codec_private_data, get_avc_config_from_h264, parse_codec_private) # Useful SPS/PPS discussion. # https://copyprogramming.com/howto/including-sps-and-pps-in-a-raw-h264-track # https://gitlab.com/mbunkus/mkvtoolnix/-/issues/2390 # New strategy: a possible way of handling multiple SPS/PPS gracefully. # Encode each head and trailer with FFMPEG using only I-frame (to be sure the NAL unit will never # refer to another image). # Encode using an different SPS-ID all of them (using sps-id parameter of libx264 library, e.g # 1 instead of 0). # For the video track produce only a raw H264 file and a file containing timestamps of the # different frames. # For the rest of the tracks (audio, subtitles) produce directly a MKV (this is already done). # Concatenate all raw H264 in a giant one (like cat), and the same for timestamps of video frames # (to keep sound and video synchronized). # Then use mkvmerge to remux the H264 track and the rest of tracks. # MKVmerge "concatenate" subcommand is able to concatenate different SPS/PPS data into a bigger # Private Codec Data. # However, this is proved to be not reliable. Sometimes it results in a AVC context containing # a single SPS/PPS. # So we have to rely on a manual parsing of the H264 AVC context of original movie # and the ones produced for headers and trailers, and then merging them into a bigger AVC context. # Then finally, change the Private Codec Data in the final MKV. @typechecked def check_required_tools() -> tuple[bool,dict[str,str]]: """ Checks if all required external tools are installed. This function verifies the presence of required and optional external tools on the system. It returns a tuple containing a boolean indicating whether all optional tools are installed, along with a dictionary containing the paths to all tools. Args: None Returns: tuple[bool, dict[str, str]]: - bool: True if all optional tools are installed, False otherwise - dict[str, str]: dictionary containing the paths to all tools """ logger = logging.getLogger(__name__) all_optional_tools = True paths = {} required = ['ffmpeg', 'ffprobe', 'mkvmerge', 'mkvinfo'] optional = ['mkvextract', 'vobsubocr','tesseract'] for tool in required: path = which(tool) if path is None: logger.error('Required tool: %s is missing.',tool) exit(-1) else: paths[tool] = path for tool in optional: path = which(tool) if path is None: logger.info('Optional tool: %s is missing.',tool) all_optional_tools = False else: paths[tool] = path return all_optional_tools, paths @typechecked def get_tesseract_supported_lang(tesseract_path:str) -> dict[Lang, str]|None: """ Retrieves the set of natural languages supported by the Tesseract OCR tool. This function runs the Tesseract binary with the --list-langs option and parses the output to extract the supported languages. Args: tesseract_path (str): The path to the Tesseract binary. Returns: dict[Lang, str] | None: - A dictionary mapping Lang objects to their corresponding language codes (e.g., "eng" for English) - None if an error occurs while running the Tesseract binary """ logger = logging.getLogger(__name__) res = {} with Popen([tesseract_path, '--list-langs'], stdout=PIPE) as tesseract: for line in tesseract.stdout: line = line.decode('utf8') p = re.compile('(?P[a-z]{3})\n') m = re.match(p,line) if m is not None: try: lang = m.group('lang') key = Lang(lang) res[key] = lang except InvalidLanguageValue as e: logger.warning('Invalid language: %s', e) pass tesseract.wait() if tesseract.returncode != 0: logger.error("Tesseract returns an error code: %d",tesseract.returncode) return None return res @typechecked def get_frame_rate(ffprobe_path:str, input_file: IO[bytes]) -> float|None: """ Retrieves the frame rate of a video file using the ffprobe tool. This function runs the ffprobe binary with the specified input file and parses the output to extract the frame rate. It uses two methods to calculate the frame rate: one based on the timestamp of the frames and another based on the duration of the frames. If the two calculated frame rates are significantly different, the function returns an error Args: ffprobe_path (str): The path to the ffprobe binary. input_file (IO[bytes]): The input video file. Returns: float | None: - The frame rate of the video file as a floating-point number - None if an error occurs while running the ffprobe binary or if the calculated frame rates are inconsistent """ logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) mean_duration = 0. nb_frames1 = 0 nb_frames2 = 0 min_ts = None max_ts = None interlaced = False params = [ffprobe_path, '-loglevel', 'quiet', '-select_streams', 'v', '-show_frames', '-read_intervals', '00%+30', '-of', 'json', f'/proc/self/fd/{infd:d}'] env = {**os.environ, 'LANG': 'C'} with Popen(params, stdout=PIPE, close_fds=False, env=env) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'frames' in out: for frame in out['frames']: if 'interlaced_frame' in frame: if frame['interlaced_frame'] == 1: interlaced = True if 'pts_time' in frame: ts = float(frame['pts_time']) if min_ts is None: min_ts = ts if max_ts is None: max_ts = ts min_ts = min(min_ts, ts) max_ts = max(max_ts, ts) nb_frames1+=1 if 'duration_time' in frame: mean_duration+=float(frame['duration_time']) nb_frames2+=1 else: return None ffprobe.wait() if ffprobe.returncode != 0: logger.error("ffprobe returns an error code: %d", ffprobe.returncode) return None frame_rate1 = nb_frames1/(max_ts-min_ts) frame_rate2 = nb_frames2 / mean_duration if abs(frame_rate1 - frame_rate2) > 0.2: if not interlaced: logger.error('Video is not interlaced and the disperancy between frame rates is too \ big: %f / %f', frame_rate1, frame_rate2) return None if abs(frame_rate1*2 - frame_rate2) < 0.2: return frame_rate2/2 logger.error('Video is interlaced and the disperancy between frame rates is too big:\ %f / %f', frame_rate1, frame_rate2) return None return frame_rate2 @typechecked def get_subtitles_tracks(ffprobe_path:str, mkv_path: str) -> dict[str,str]|None: logger = logging.getLogger(__name__) tracks={} with Popen([ffprobe_path, '-loglevel', 'quiet', '-select_streams', 's', '-show_entries', 'stream=index,codec_name:stream_tags=language', '-of', 'json', mkv_path], stdout=PIPE) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'streams' in out: for stream in out['streams']: index = stream['index'] codec = stream['codec'] lang = stream['tags']['language'] if codec == 'dvd_subtitle': if lang not in tracks: tracks[lang] = [index] else: current_langs = tracks[lang] current_langs.append(index) tracks[lang] = current_langs else: return None ffprobe.wait() if ffprobe.returncode != 0: logger.error("ffprobe returns an error code: %d", ffprobe.returncode) return None return tracks @typechecked def extract_srt(mkvextract:str, filename:str, subtitles:dict[str, list[int]], langs:dict[Lang,str]) -> list[tuple[str,str,str,str]]|None: logger = logging.getLogger(__name__) params = [mkvextract, filename, 'tracks'] res = [] for lang in subtitles: iso = Lang(lang) if iso in langs: ocrlang = langs[iso] else: logger.warning("Language not supported by Tesseract: %s", iso.name) ocrlang ='osd' if len(subtitles[lang]) == 1: params.append(f'{subtitles[lang][0]:d}:{lang}') res.append((f'{lang}.idx', f'{lang}.sub', lang, ocrlang)) else: count = 1 for track in subtitles[lang]: params.append(f'{track:d}:{lang}-{count:d}') res.append((f'{lang}-{count:d}.idx', f'{lang}-{count:d}.sub', lang, ocrlang)) count = count+1 logger.debug('Executing %s', params) env = {**os.environ, 'LANG': 'C'} with Popen(params, stdout=PIPE, close_fds=False, env=env) as extract: pb = tqdm(TextIOWrapper(extract.stdout, encoding="utf-8"), total=100, unit='%', desc='Extraction:') for line in pb: if line.startswith('Progress :'): p = re.compile('^Progress : (?P[0-9]{1,3})%$') m = p.match(line) if m is None: logger.error('Impossible to parse progress') pb.update(int(m['progress'])-pb.n) pb.update(100-pb.n) pb.refresh() pb.close() extract.wait() # mkvextract returns 0, 1 or 2 as error code. match extract.returncode: case 0: logger.info('Subtitle tracks were succesfully extracted.') case 1: logger.warning('Mkvextract returns warning') case 2: logger.error('Mkvextract returns an error code: %d', extract.returncode) res = None return res @typechecked def do_ocr(vobsubocr:str, idxs: list[tuple[str,str,str,str]], duration:timedelta, temporaries:list[IO[bytes]], dump_mem_fd:bool=False): logger = logging.getLogger(__name__) res = [] for idx_name, _, lang, iso in idxs: srtname = f'{os.path.splitext(idx_name)[0]}.srt' # Tesseract seems to recognize the three dots ... as "su" ldots = re.compile('^su\n$') # Timestamps produced by vobsubocr: 01:52:19,861 --> 01:52:21,641 timestamps = re.compile((r'^[0-9]{2}:[0-9]{2}:[0-9]{2},[0-9]{3} \-\-> (?P[0-9]{2}):' r'(?P[0-9]{2}):(?P[0-9]{2}),[0-9]{3}$')) srtfd = memfd_create(srtname, flags=0) with Popen([vobsubocr, '--lang', iso, idx_name], stdout=PIPE) as ocr: pb = tqdm(TextIOWrapper(ocr.stdout, encoding="utf-8"), total= int(duration/timedelta(seconds=1)), unit='s', desc='OCR') for line in pb: m = re.match(ldots,line) if m is not None: write(srtfd, '...'.encode(encoding='UTF-8')) else: write(srtfd, line.encode(encoding='UTF-8')) m = re.match(timestamps, line) if m is not None: hours = int(m.group('hours')) minutes = int(m.group('hours')) seconds = int(m.group('seconds')) ts = timedelta(hours=hours, minutes=minutes, seconds=seconds) pb.n = int(ts/timedelta(seconds=1)) pb.update() status = ocr.wait() if status != 0: logger.error('OCR failed with status code: %d', status) if dump_mem_fd: try: with open(srtname,'w', encoding='utf8') as dump_srt: lseek(srtfd, 0, SEEK_SET) srt_length = fstat(srtfd).st_size buf = read(srtfd, srt_length) outfd = dump_srt.fileno() pos = 0 while pos < srt_length: pos+=write(outfd, buf[pos:]) temporaries.append(dump_srt) except OSError: logger.error('Impossible to create file: %s', srtname) return None srt_length = fstat(srtfd).st_size if srt_length > 0: res.append((srtfd, lang)) return res @unique class SupportedFormat(IntEnum): TS = 1 MP4 = 2 MATROSKA = 3 def __str__(self): match self: case SupportedFormat.TS: return 'mpegts' case SupportedFormat.MP4: return 'mov,mp4,m4a,3gp,3g2,mj2' case SupportedFormat.MATROSKA: return 'matroska,webm' case _: return 'Unsupported format' # Extract SPS/PPS # https://gitlab.com/mbunkus/mkvtoolnix/-/issues/2390 # ffmpeg -i -c:v copy -an -sn -bsf:v trace_headers -t 0.01\ # -report -loglevel 0 -f null - # Found codec private data using mkvinfo @typechecked def get_codec_private_data_from_mkv(mkvinfo_path:str, input_file: IO[bytes]) -> tuple[int, bytes]|tuple[None,None]: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) found = False env = {**os.environ, 'LANG': 'C'} # Output example # Codec's private data: size 48 (H.264 profile: High @L4.0) hexdump 01 64 00 28 ff e1 00 1b 67\ # 64 00 28 ac d9 40 78 04 4f dc d4 04 04 05 00 00 92 ef 00 1d ad a6 1f 16 2d 96 01 00 06 68 fb\ # a3 cb 22 c0 fd f8 f8 00 at 406 size 51 data size 48 with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False, env=env) as mkvinfo: out, _ = mkvinfo.communicate() out = out.decode('utf8') reg_exp = (r"^.*Codec's private data: size ([0-9]+) \(H.264.*\) hexdump " r"(?P([0-9a-f]{2} )+)at (?P[0-9]+) size (?P[0-9]+).*$") p = re.compile(reg_exp) for line in out.splitlines(): m = p.match(line) if m is not None: size = int(m.group('size')) position = int(m.group('position')) logger.debug("Found codec private data at position: %s, size: %d", position, size) found = True mkvinfo.wait() break if found: lseek(infd, position, SEEK_SET) data = read(infd, size) return position, data return None, None @typechecked def parse_mkv_tree(mkvinfo_path:str, input_file: IO[bytes]) -> dict[str,tuple[int,int]]: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) env = {**os.environ, 'LANG': 'C'} elements = {} with Popen([mkvinfo_path, '-z', '-X', '-P', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False, env=env) as mkvinfo: out, _ = mkvinfo.communicate() out = out.decode('utf8') prefix = [] reg_exp = (r"(^(?P\+)|(\|(?P[ ]*\+))).*at (?P[0-9]+)" r" size (?P[0-9]+).*$") p = re.compile(reg_exp) prev_depth = -1 for line in out.splitlines(): m = p.match(line) if m is None: logger.error("Impossible to match line: %s", line) else: position = int(m.group('position')) size = int(m.group('size')) root = m.group('root') is not None if root: depth = 0 else: depth = len(m.group('depth')) if depth > prev_depth: for _ in range(depth-prev_depth): prefix.append(1) elif depth == prev_depth: subid = prefix[-1] subid+=1 prefix.pop() prefix.append(subid) else: for _ in range(prev_depth-depth): prefix.pop() subid = prefix[-1] subid+=1 prefix.pop() prefix.append(subid) prev_depth = depth key=".".join(map(str, prefix)) elements[key] = (position, size) mkvinfo.wait() return elements @typechecked def change_codec_private_data(mkvinfo_path:str, input_file: IO[bytes], codec_data:bytes) -> None: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) current_length = fstat(infd).st_size logger.info('Current size of file: %d', current_length) position, current_data = get_codec_private_data_from_mkv(mkvinfo_path, input_file) current_data_length = len(current_data) future_length = current_length - current_data_length + len(codec_data) logger.info('Expected size of file: %d', future_length) logger.info('Current data at position %d: %s', position, hexdump.dump(current_data, sep=":")) logger.info('Future data: %s', hexdump.dump(codec_data, sep=":")) elements = parse_mkv_tree(mkvinfo_path, input_file) found = False for key, (pos,size) in elements.items(): if pos == position: logger.info('Codec private data key: %s', key) found = True break if not found: logger.error('Impossible to retrieve the key of codec private data') exit(-1) if current_length < future_length: lseek(infd, position+current_data_length, SEEK_SET) tail = read(infd, current_length-(position+current_data_length)) # We extend the file at the end with zeroes ftruncate(infd, future_length) lseek(infd, position+len(codec_data), SEEK_SET) write(infd, tail) lseek(infd, position, SEEK_SET) write(infd, codec_data) elif current_length == future_length: # Almost nothing to do except overwriting old private codec data with new ones. lseek(infd, position, SEEK_SET) write(infd, codec_data) else: lseek(infd, position+current_data_length, SEEK_SET) tail = read(infd, current_length-(position+current_data_length)) lseek(infd, position+len(codec_data), SEEK_SET) write(infd, tail) lseek(infd, position, SEEK_SET) write(infd, codec_data) # We reduce the length of file. ftruncate(infd, future_length) # We have to modify the tree elements up to the root that contains the codec private data. keys = key.split('.') logger.info(keys) delta = future_length-current_length # if there is no modification of the private codec data, no need to change anything. if delta != 0: for _ in range(len(keys)-1): keys.pop() key=".".join(map(str, keys)) pos, size = elements[key] logger.info('Trying to fix element with key: %s at position: %d with actual size: %d.', key, pos, size) # Changing an element can increase its size (in very rare case). # In that case, we update the new delta that will be larger (because the element has # been resized). delta+=change_ebml_element_size(input_file, pos, delta) @typechecked def get_format(ffprobe_path:str, input_file: IO[bytes]) -> dict|None: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_format', '-of', 'json', '-i', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'format' in out: return out['format'] else: logger.error('Impossible to retrieve format of file') return None @typechecked def get_movie_duration(ffprobe_path:str, input_file: IO[bytes]) -> timedelta|None: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_format', '-of', 'json', '-i', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'format' in out and 'duration' in out['format']: duration = floor(float(out['format']['duration'])) ts = timedelta(seconds=duration) return ts else: logger.error('Impossible to retrieve duration of movie') return None # ffprobe -loglevel quiet -select_streams v:0 -show_entries stream=width,height -of json sample.ts @typechecked def get_video_dimensions(ffprobe_path:str, input_file: IO[bytes]) -> tuple[int,int]: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) with Popen([ffprobe_path, '-loglevel', 'quiet', '-select_streams', 'v:0', '-show_entries',\ 'stream=width,height', '-of', 'json', '-i', f'/proc/self/fd/{infd:d}'],\ stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'streams' in out: video = out['streams'][0] if ('width' in video) and ('height' in video): return int(video['width']), int(video['height']) logger.error('Impossible to retrieve dimensions of video') exit(-1) @typechecked def get_streams(ffprobe_path:str, input_file: IO[bytes]) -> list|None: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_streams', '-of', 'json', '-i', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'streams' in out: return out['streams'] else: logger.error('Impossible to retrieve streams inside file') return None @typechecked def with_subtitles(ffprobe_path:str, input_file: IO[bytes]) -> bool: """ Checks if a media file contains subtitles using the ffprobe tool. This function runs the ffprobe binary with the specified input file and parses the output to determine if the file contains subtitles. It returns True if at least one subtitle stream is found, False otherwise. Args: ffprobe_path (str): The path to the ffprobe binary. input_file (IO[bytes]): The input media file. Returns: bool: - True if the media file contains at least one subtitle stream - False if: - the media file does not contain any subtitle streams - an error occurs while running the ffprobe binary - the streams information cannot be retrieved from the media file """ logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) with Popen([ffprobe_path, '-loglevel', 'quiet', '-show_streams', '-of', 'json', '-i', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'streams' in out: streams = out['streams'] for stream in streams: if 'codec_type' in stream and stream['codec_type'] == 'subtitle': return True else: logger.error('Impossible to retrieve streams inside file') return False @typechecked def parse_timestamp(ts:str) -> timedelta|None: """ Parse a timestamp string into a timedelta object. This function takes a string representing a timestamp in the format HH:MM:SS[.us] and returns a timedelta object representing the corresponding time interval. The timestamp string can have an optional microsecond component. Args: ts (str): The timestamp string to parse. Returns: timedelta | None: - A timedelta object representing the parsed timestamp - None if: - the timestamp string is not in the correct format - the timestamp values are out of range (e.g. hour > 23, minute > 59, etc.) """ logger = logging.getLogger(__name__) ts_reg_exp = (r'^(?P[0-9]{1,2}):(?P[0-9]{1,2})' r':(?P[0-9]{1,2})(\.(?P[0-9]{1,6}))?$') p = re.compile(ts_reg_exp) m = p.match(ts) if m is None: logger.warning("Impossible to parse timestamp: %s", ts) return None values = m.groupdict() hour = 0 minute = 0 second = 0 us = 0 if values['hour'] is not None: hour = int(values['hour']) if values['minute'] is not None: minute = int(values['minute']) if values['second'] is not None: second = int(values['second']) if values['us'] is not None: us = int(values['us']) if hour < 0 or hour > 23: logger.error("hour must be in [0,24[") return None if minute < 0 or minute > 59: logger.error("minute must be in [0,60[") return None if second < 0 or second > 59: logger.error("second must be in [0,60[") return None if us < 0 or us > 1000000: logger.error("milliseconds must be in [0,1000000[") return None res = timedelta(hours=hour, minutes=minute, seconds=second, microseconds=us) return res @typechecked def parse_time_interval(interval: str) -> tuple[timedelta, timedelta] | None: """ Parse a time interval string into a tuple of two timedelta objects. This function takes a string representing a time interval in the format HH:MM:SS[.ms]-HH:MM:SS[.ms] and returns a tuple of two timedelta objects representing the start and end times of the interval. The time interval string can have an optional millisecond component. Args: interval (str): The time interval string to parse. Returns: tuple[timedelta, timedelta] | None: - A tuple of two timedelta objects representing the start and end times of the interval - None if: - the time interval string is not in the correct format - the time values are out of range (e.g. hour > 23, minute > 59, etc.) - the end time is before the start time (non-monotonic interval) """ logger = logging.getLogger(__name__) interval_reg_exp = (r'^(?P[0-9]{1,2}):(?P[0-9]{1,2}):(?P[0-9]{1,2})' r'(\.(?P[0-9]{1,3}))?-(?P[0-9]{1,2}):(?P[0-9]{1,2})' r':(?P[0-9]{1,2})(\.(?P[0-9]{1,3}))?$') p = re.compile(interval_reg_exp) m = p.match(interval) if m is None: logger.error("Impossible to parse time interval") return None values = m.groupdict() hour1 = 0 minute1 = 0 second1 = 0 ms1 = 0 hour2 = 0 minute2 = 0 second2 = 0 ms2 = 0 if values['hour1'] is not None: hour1 = int(values['hour1']) if values['minute1'] is not None: minute1 = int(values['minute1']) if values['second1'] is not None: second1 = int(values['second1']) if values['ms1'] is not None: ms1 = int(values['ms1']) if values['hour2'] is not None: hour2 = int(values['hour2']) if values['minute2'] is not None: minute2 = int(values['minute2']) if values['second2'] is not None: second2 = int(values['second2']) if values['ms2'] is not None: ms2 = int(values['ms2']) if hour1 < 0 or hour1 > 23: logger.error("hour must be in [0,24[") return None, None if minute1 < 0 or minute1 > 59: logger.error("minute must be in [0,60[") return None, None if second1 < 0 or second1 > 59: logger.error("second must be in [0,60[") return None, None if ms1 < 0 or ms1 > 1000: logger.error("milliseconds must be in [0,1000[") return None, None if hour2 < 0 or hour2 > 23: logger.error("hour must be in [0,24[") return None, None if minute2 < 0 or minute2 > 59: logger.error("minute must be in [0,60[") return None, None if second2 < 0 or second2 > 59: logger.error("second must be in [0,60[") return None, None if ms2 < 0 or ms2 > 1000: logger.error("milliseconds must be in [0,1000[") return None, None ts1 = timedelta(hours=hour1, minutes=minute1, seconds=second1, microseconds=ms1*1000) ts2 = timedelta(hours=hour2, minutes=minute2, seconds=second2, microseconds=ms2*1000) if ts2 < ts1: logger.error("Non monotonic interval") return None,None return (ts1, ts2) @typechecked def compare_time_interval(interval1: tuple[timedelta, timedelta], interval2: tuple[timedelta, timedelta]) -> int: """ Compare two time intervals. This function compares two time intervals represented by tuples of two timedelta objects. It returns an integer indicating the relationship between the two intervals: - -1 if interval 1 is before interval 2 - 1 if interval 1 is after interval 2 - 0 if the two intervals overlap or are equal Args: interval1 (tuple[timedelta, timedelta]): The first time interval interval2 (tuple[timedelta, timedelta]): The second time interval Returns: int: The relationship between the two time intervals """ ts11,ts12 = interval1 ts21,ts22 = interval2 if ts12 < ts21: return -1 elif ts22 < ts11: return 1 else: return 0 @typechecked def ffmpeg_convert(ffmpeg_path:str, ffprobe_path:str, input_file: IO[bytes], input_format:str, output_file: IO[bytes], output_format:str, duration: timedelta): logger = logging.getLogger(__name__) width, height = get_video_dimensions(ffprobe_path, input_file) subtitles = with_subtitles(ffprobe_path, input_file) infd = input_file.fileno() outfd = output_file.fileno() set_inheritable(infd, True) set_inheritable(outfd, True) if logger.getEffectiveLevel() == logging.DEBUG: log = [] else: log = [ '-loglevel', 'quiet' ] params = [ffmpeg_path, '-y',]+log+['-progress', '/dev/stdout', '-canvas_size', f'{width:d}x{height:d}', '-f', input_format, '-i', f'/proc/self/fd/{infd:d}', '-map', '0:v', '-map', '0:a'] if subtitles: params.extend(['-map', '0:s']) params.extend(['-bsf:v', 'h264_mp4toannexb,dump_extra=freq=keyframe', '-vcodec', 'copy', '-acodec', 'copy']) if subtitles: params.extend(['-scodec', 'dvdsub']) params.extend(['-r:0', '25', '-f', output_format, f'/proc/self/fd/{outfd:d}']) logger.debug('Executing %s', params) with Popen(params, stdout=PIPE, close_fds=False) as ffmpeg: pb = tqdm(TextIOWrapper(ffmpeg.stdout, encoding="utf-8"), total=int(duration/timedelta(seconds=1)), unit='s', desc='Conversion') for line in pb: if line.startswith('out_time='): ts = line.split('=')[1].strip() ts = parse_timestamp(ts) if ts is not None: pb.n = int(ts/timedelta(seconds=1)) pb.update() status = ffmpeg.wait() if status != 0: logger.error('Conversion failed with status code: %d', status) @typechecked def get_ts_frame(frame: dict) -> timedelta|None: logger = logging.getLogger(__name__) if 'pts_time' in frame: pts_time = float(frame['pts_time']) elif 'pkt_pts_time' in frame: pts_time = float(frame['pkt_pts_time']) else: logger.error('Impossible to find timestamp of frame %s', frame) return None ts = timedelta(seconds=pts_time) return ts @typechecked def get_packet_duration(packet: dict) -> int: logger = logging.getLogger(__name__) if 'duration' in packet: duration = int(packet['duration']) elif 'pkt_duration' in packet: duration = int(packet['pkt_duration']) else: logger.error('Impossible to find duration of packet %s', packet) return None return duration @typechecked def get_frames_in_stream(ffprobe_path: str, input_file: IO[bytes], begin:timedelta, end:timedelta, stream_kind:str, sub_stream_id:int=0) -> list[dict]|None: logger = logging.getLogger(__name__) infd = input_file.fileno() set_inheritable(infd, True) command = [ffprobe_path, '-loglevel', 'quiet', '-read_intervals', f'{begin}%{end}', '-show_entries', 'frame', '-select_streams', f'{stream_kind}:{sub_stream_id:d}','-of', 'json', f'/proc/self/fd/{infd:d}'] logger.debug('Executing: %s', command) with Popen(command, stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() frames = json.load(BytesIO(out)) status = ffprobe.wait() if status != 0: logger.error('ffprobe failed with status code: %d', status) return None # Sort frames by timestamp tmp = {} if 'frames' in frames: frames = frames['frames'] for frame in frames: ts = get_ts_frame(frame) if ts is None: return None if begin <= ts <= end: tmp[ts]=frame res = [] for ts in sorted(tmp): res.append(tmp[ts]) return res else: logger.error('Impossible to retrieve frames inside file around [%s,%s]', begin, end) return None # TODO: Finish implementation of this function and use it. @typechecked def get_nearest_idr_frame(ffprobe_path: str, input_file: IO[bytes], timestamp:timedelta, before: bool=True, delta: timedelta=timedelta(seconds=2)): # pylint: disable=W0613 logger = logging.getLogger(__name__) zero = timedelta() tbegin = timestamp-delta tend = timestamp+delta tbegin = max(tbegin, zero) infd = input_file.fileno() set_inheritable(infd, True) logger.debug('Looking for IDR frame in [%s, %s]', tbegin, tend) idrs = [] # Retains only IDR frame with Popen([ffprobe_path, '-loglevel', 'quiet', '-read_intervals', f'{tbegin}%{tend}', '-skip_frame', 'nokey', '-show_entries', 'frame', '-select_streams', 'v:0', '-of', 'json', f'/proc/self/fd/{infd:d}'], stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() frames = json.load(BytesIO(out)) status = ffprobe.wait() if status != 0: logger.error('ffprobe failed with status code: %d', status) return None if 'frames' in frames: frames = frames['frames'] for frame in frames: ts = get_ts_frame(frame) if ts is None: return None if tbegin <= ts <= tend: idrs.append(frame) else: logger.error('Impossible to retrieve IDR frames inside file around [%s,%s]', tbegin, tend) return None return None @typechecked def get_nearest_iframe(ffprobe_path:str, input_file: IO[bytes], timestamp:timedelta, before:bool=True, delta_max:timedelta=timedelta(seconds=15))-> tuple[int,dict]: logger = logging.getLogger(__name__) infd = input_file.fileno() set_inheritable(infd, True) delta = timedelta(seconds=1) iframe = None while delta < delta_max: zero = timedelta() if before: tbegin = timestamp-delta else: tbegin = timestamp if not before: tend = timestamp+delta else: tend = timestamp tbegin = max(tbegin, zero) logger.debug('Looking for an iframe in [%s, %s]', tbegin, tend) frames = get_frames_in_stream(ffprobe_path, input_file=input_file, begin=tbegin, end=tend, stream_kind='v') if frames is None: logger.debug('Found no frame in [%s, %s]', tbegin, tend) delta+=timedelta(seconds=1) continue iframes = [] for frame in frames: if frame['pict_type'] == 'I': iframes.append(frame) found = False for frame in iframes: ts = get_ts_frame(frame) if ts is None: logger.warning('I-frame with no timestamp: %s', frame) continue if before and ts <= timestamp: found = True iframe = frame if not before and ts >= timestamp: found = True iframe = frame break if found: logger.info("Found i-frame at: %s", iframe) break else: delta+=timedelta(seconds=1) continue if iframe is not None: its = get_ts_frame(iframe) nb_frames = 0 for frame in frames: ts = get_ts_frame(frame) if ts is None: logger.warning('Frame without timestamp: %s', frame) continue if before: if its <= ts <= timestamp: logger.info("Retrieve a frame between %s and %s at %s", its, timestamp, ts) nb_frames = nb_frames+1 else: if timestamp <= ts <= its: logger.info("Retrieve a frame between %s and %s at %s", timestamp, ts, its) nb_frames = nb_frames+1 else: logger.error("Impossible to find I-frame between: %s and %s", tbegin, tend) return 0, None return(nb_frames, iframe) @typechecked def extract_mkv_part(mkvmerge_path:str, input_file:IO[bytes], output_file:IO[bytes], begin:timedelta, end:timedelta) -> None: logger = logging.getLogger(__name__) logger.info('Extract video between I-frames at %s and %s', begin,end) infd = input_file.fileno() outfd = output_file.fileno() lseek(infd, 0, SEEK_SET) lseek(outfd, 0, SEEK_SET) set_inheritable(infd, True) set_inheritable(outfd, True) env = {**os.environ, 'LANG': 'C'} warnings = [] command = [mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '--split', f'parts:{begin}-{end}', f'/proc/self/fd/{infd:d}'] logger.debug('Executing: %s', command) with Popen(command, stdout=PIPE, close_fds=False, env=env) as mkvmerge: pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%', desc='Extraction') for line in pb: if line.startswith('Progress :'): p = re.compile('^Progress : (?P[0-9]{1,3})%$') m = p.match(line) if m is None: logger.error('Impossible to parse progress') pb.update(int(m['progress'])-pb.n) elif line.startswith('Warning'): warnings.append(line) pb.update(100-pb.n) pb.refresh() pb.close() status = mkvmerge.wait() if status == 1: logger.warning('Extraction returns warning') for w in warnings: logger.warning(w) elif status == 2: logger.error('Extraction returns errors') @typechecked def extract_pictures(ffmpeg_path:str, input_file:IO[bytes], begin:timedelta, nb_frames:int, width:int=640, height:int=480) -> tuple[bytes,int]|tuple[None,None]: """ Extract pictures from a video file using FFmpeg. This function runs the FFmpeg binary to extract a specified number of frames from a video file, starting at a given time. The extracted frames are stored in memory as PPM images and returned as a tuple containing the image data and a file descriptor to the memory created by memfd_create. Args: ffmpeg_path (str): The path to the FFmpeg binary. input_file (IO[bytes]): The input video file. begin (timedelta): The start time of the extraction. nb_frames (int): The number of frames to extract. width (int, optional): The width of the extracted images. Defaults to 640. height (int, optional): The height of the extracted images. Defaults to 480. Returns: tuple[bytes, int] | tuple[None, None]: - A tuple containing the extracted image data as bytes and a file descriptor - A tuple containing None, None if the extraction fails """ logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) outfd = memfd_create('pictures', flags=0) set_inheritable(outfd, True) # PPM header # "P6\nWIDTH HEIGHT\n255\n" header_len=2+1+ceil(log(width, 10))+1+ceil(log(height, 10))+1+3+1 logger.debug('Header length: %d', header_len) image_length = width*height*3+header_len length = image_length*nb_frames logger.debug("Estimated length: %d", length) command = [ffmpeg_path, '-loglevel', 'quiet' ,'-y', '-ss', f'{begin}', '-i', f'/proc/self/fd/{infd}', '-s', f'{width:d}x{height:d}', '-vframes', f'{nb_frames:d}', '-c:v', 'ppm','-f', 'image2pipe', f'/proc/self/fd/{outfd:d}'] logger.debug('Executing: %s', command) images = b'' with Popen(command, stdout=PIPE, close_fds=False) as ffmpeg: status = ffmpeg.wait() if status != 0: logger.error('Conversion failed with status code: %d', status) return None, None lseek(outfd, 0, SEEK_SET) images = read(outfd,length) if len(images) != length: logger.error("Received %d bytes but %d were expected.", len(images), length) return None, None lseek(outfd, 0, SEEK_SET) return images, outfd @typechecked def extract_sound(ffmpeg_path:str, input_file: IO[bytes], begin:timedelta, output_filename:str, packet_duration:int, sub_channel:int=0, nb_packets:int=0, sample_rate:int=48000, nb_channels:int=2) -> tuple[bytes,int]|tuple[None,None]: logger = logging.getLogger(__name__) outfd = memfd_create(output_filename, flags=0) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) set_inheritable(outfd, True) sound = b'' length = int(nb_channels*sample_rate*4*nb_packets*packet_duration/1000) command = [ffmpeg_path, '-y', '-loglevel', 'quiet', '-ss', f'{begin}', '-i', f'/proc/self/fd/{infd}', f'-frames:a:{sub_channel:d}', f'{nb_packets+1:d}', '-c:a', 'pcm_s32le', '-sample_rate', f'{sample_rate:d}', '-channels', f'{nb_channels:d}', '-f', 's32le', f'/proc/self/fd/{outfd:d}'] logger.debug('Executing: %s', command) with Popen(command, stdout=PIPE, close_fds=False) as ffmpeg: status = ffmpeg.wait() if status != 0: logger.error('Sound extraction returns error code: %d', status) return None, None lseek(outfd, 0, SEEK_SET) sound = read(outfd, length) if len(sound) != length: logger.info("Received %d bytes but %d were expected (channels=%d, freq=%d, packets=%d,\ duration=%d ms).", len(sound), length, nb_channels, sample_rate, nb_packets, packet_duration) return None, None return sound, outfd @typechecked def dump_ppm(pictures: bytes, prefix: str, temporaries: list[IO[bytes]]) -> None: """ Dump PPM pictures from a bytes buffer to files. This function takes a bytes buffer containing PPM pictures, a prefix for the output file names, and a list of temporary files. It extracts each PPM picture from the buffer, checks its validity, and writes it to a file. The output files are named according to the prefix and a zero-padded three-digit number. Args: pictures (bytes): The bytes buffer containing the PPM pictures. prefix (str): The prefix for the output file names. temporaries (list[IO[bytes]]): A list of temporary files that will be used to store the output files. Returns: None Raises: None, but logs errors if: - the PPM picture is not valid (e.g. wrong magic number, dimensions, or color encoding) - an I/O error occurs while creating or writing to an output file """ logger = logging.getLogger(__name__) # "P6\nWIDTH HEIGHT\n255\n" pos = 0 picture = 0 logger.debug('Dumping %d pictures: %s', len(pictures),prefix) while pos[0-9]+) (?P[0-9]+)\n$') m = pattern.match(dimensions) if m is not None: width = int(m['width']) height = int(m['height']) else: logger.error('Impossible to parse dimensions of picture') return else: logger.error('Not a PPM picture') return if max_value != 255: logger.error('Not a valid PPM picture. Color are not encoded on byte. Max value: %d', max_value) header_len=2+1+ceil(log(width, 10))+1+ceil(log(height, 10))+1+3+1 try: with open(filename, 'wb') as out: temporaries.append(out) outfd = out.fileno() length=header_len+3*width*height nb_bytes = 0 while nb_bytes < length: nb_bytes+=write(outfd, pictures[pos+nb_bytes:pos+length]) pos+=length picture+=1 except OSError: logger.error('Impossible to create file: %s', filename) @typechecked def extract_all_streams(ffmpeg_path:str, ffprobe_path:str, input_file:IO[bytes], begin:timedelta, end:timedelta, streams, files_prefix, nb_frames:int, framerate:float, width:int, height:int, temporaries, dump_mem_fd:bool=False): logger = logging.getLogger(__name__) # The command line for encoding only video track video_encoder_params = [ ffmpeg_path, '-y', '-loglevel', 'quiet'] video_input_params = [] video_codec_params = [] # The command line to create a MKV file with the rest of tracks generic_encoder_params = [ ffmpeg_path, '-y', '-loglevel', 'quiet' ] generic_input_params = [] generic_codec_params = [] if begin < end: video_id=0 audio_id=0 subtitle_id=0 memfds = [] for stream in streams: if stream['codec_type'] == 'video': logger.info("Extracting %d frames of video stream v:%d", nb_frames, video_id) sar = stream['sample_aspect_ratio'] dar = stream['display_aspect_ratio'] pixel_format = stream['pix_fmt'] color_range = stream['color_range'] color_space =stream['color_space'] color_transfer = stream['color_transfer'] color_primaries = stream['color_primaries'] level = int(stream['level']) level = f'{floor(level/10):d}.{level%10:d}' chroma_location = stream['chroma_location'] field_order = stream match field_order: case 'progressive': interlaced_options = ['-field_order', '0'] case 'tt': interlaced_options = ['-top', '1', f'-flags:v:{video_id:d}', '+ilme+ildct', '-field_order', '1'] case 'bb': interlaced_options = ['-top', '0', f'-flags:v:{video_id:d}', '+ilme+ildct', '-field_order','2'] case 'tb': interlaced_options = ['-top', '1', f'-flags:v:{video_id:d}', '+ilme+ildct', '-field_order', '3'] case 'bt': interlaced_options = ['-top', '0', f'-flags:v:{video_id:d}', '+ilme+ildct', '-field_order', '4'] case _: interlaced_options = [] # ======================================= # # TODO: adjust SAR and DAR # https://superuser.com/questions/907933/correct-aspect-ratio-without-re-encoding-video-file # SAR: -aspect width:height # DAR: -bsf:v sample_aspect_ratio=1:video_format logger.warning('Missing SAR adjustment for: %s', sar) logger.warning('Missing DAR adjustment for: %s', dar) logger.warning('Missing treatment for chroma location: %s', chroma_location) codec = stream['codec_name'] images_bytes, memfd = extract_pictures(ffmpeg_path, input_file=input_file, begin=begin, nb_frames=nb_frames, width=width, height=height) if images_bytes is None: logger.error('Impossible to extract picture from video stream.') exit(-1) memfds.append(memfd) if dump_mem_fd: dump_ppm(images_bytes, f'{files_prefix}-{video_id:d}', temporaries) # We rewind to zero the memory file descriptor lseek(memfd, 0, SEEK_SET) set_inheritable(memfd, True) video_input_params.extend(['-framerate', f'{framerate:f}', '-f', 'image2pipe', '-i', f'/proc/self/fd/{memfd:d}']) video_codec_params.extend([f'-c:v:{video_id:d}', codec, f'-level:v:{video_id:d}', level, '-pix_fmt', pixel_format]) video_codec_params.extend(interlaced_options) video_codec_params.extend([f'-colorspace:v:{video_id}', color_space, f'-color_primaries:v:{video_id:d}', color_primaries, f'-color_trc:v:{video_id:d}', color_transfer, f'-color_range:v:{video_id:d}', color_range]) video_id=video_id+1 elif stream['codec_type'] == 'audio': logger.debug('Audio stream: %s', stream) sample_rate = int(stream['sample_rate']) nb_channels = int(stream['channels']) if 'bit_rate' in stream: bit_rate = int(stream['bit_rate']) else: bit_rate = 128000 codec = stream['codec_name'] if 'tags' in stream: if 'language' in stream['tags']: generic_codec_params.extend([f'-metadata:s:a:{audio_id:d}', f"language={stream['tags']['language']}"]) packets = get_frames_in_stream(ffprobe_path, input_file=input_file, begin=begin, end=end, stream_kind='a', sub_stream_id=audio_id) nb_packets = len(packets) logger.debug("Found %d packets to be extracted from audio track.", nb_packets) if nb_packets > 0: packet_duration = get_packet_duration(packets[0]) if packet_duration is None: return None else: packet_duration = 0 logger.info("Extracting %d packets of audio stream: a:%d" , nb_packets, audio_id) tmpname = f'{files_prefix}-{audio_id:d}.pcm' sound_bytes, memfd = extract_sound(ffmpeg_path=ffmpeg_path, input_file=input_file, begin=begin, nb_packets=nb_packets, packet_duration=packet_duration, output_filename=tmpname, sample_rate=sample_rate, nb_channels=nb_channels) if sound_bytes is None: logger.error('Impossible to extract sound track') exit(-1) memfds.append(memfd) if dump_mem_fd: try: with open(tmpname,'wb') as output: temporaries.append(output) outfd = output.fileno() pos = 0 while pos < len(sound_bytes): pos+=write(outfd, sound_bytes[pos:]) except OSError: logger.error('Impossible to create file: %s', tmpname) return None # We rewind to zero the memory file descriptor lseek(memfd, 0, SEEK_SET) set_inheritable(memfd, True) generic_input_params.extend(['-f', 's32le', '-ar', f'{sample_rate:d}', '-ac', f'{nb_channels:d}', '-i', f'/proc/self/fd/{memfd:d}']) generic_codec_params.extend([f'-c:a:{audio_id:d}', codec, f'-b:a:{audio_id:d}', f'{bit_rate:d}']) audio_id=audio_id+1 elif stream['codec_type'] == 'subtitle': logger.info("Extracting a subtitle stream: s:%d", subtitle_id) codec = stream['codec_name'] generic_input_params.extend(['-i', './empty.idx']) if 'tags' in stream: if 'language' in stream['tags']: generic_codec_params.extend([f'-metadata:s:s:{subtitle_id:d}', f"language={stream['tags']['language']}"]) generic_codec_params.extend([f'-c:s:{subtitle_id:d}', 'copy']) subtitle_id=subtitle_id+1 else: logger.error("Unknown stream type: %s", stream['codec_type']) # Create a new MKV movie with all streams (except videos) that have been extracted. generic_encoder_params.extend(generic_input_params) for index in range(audio_id+subtitle_id): generic_encoder_params.extend(['-map', f'{index:d}']) generic_encoder_params.extend(generic_codec_params) mkv_filename = f'{files_prefix}.mkv' try: mkv_output = open(mkv_filename,'wb+') except OSError: logger.error('Impossible to create file: %s', mkv_filename) return None mkvoutfd = mkv_output.fileno() set_inheritable(mkvoutfd, True) generic_encoder_params.extend(['-f', 'matroska', f'/proc/self/fd/{mkvoutfd:d}']) logger.info('Encoding all streams (except video) into a MKV file: %s', mkv_filename) logger.debug('Executing: %s', generic_encoder_params) with Popen(generic_encoder_params, stdout=PIPE, close_fds=False) as ffmpeg: status = ffmpeg.wait() if status != 0: logger.error('Encoding failed with status code: %d', status) return None temporaries.append(mkv_output) h264_filename = f'{files_prefix}.h264' try: h264_output = open(h264_filename,'wb+') except OSError: logger.error('Impossible to create file: %s', h264_filename) return None h264outfd = h264_output.fileno() set_inheritable(h264outfd, True) video_encoder_params.extend(video_input_params) video_encoder_params.extend(video_codec_params) video_encoder_params.extend([ '-x264opts', f'keyint=1:sps-id={1:d}','-bsf:v', 'h264_mp4toannexb,dump_extra=freq=keyframe,h264_metadata=\ overscan_appropriate_flag=1:sample_aspect_ratio=1:video_format=\ 0:chroma_sample_loc_type=0','-f', 'h264', f'/proc/self/fd/{h264outfd:d}']) logger.info('Encoding video into a H264 file: %s', h264_filename) logger.debug('Executing: %s', video_encoder_params) with Popen(video_encoder_params, stdout=PIPE, close_fds=False) as ffmpeg: status = ffmpeg.wait() if status != 0: logger.error('Encoding failed with status code: %d', status) return None temporaries.append(h264_output) h264_ts_filename = f'{files_prefix}-ts.txt' try: h264_ts_output = open(h264_ts_filename,'w+', encoding='utf8') except OSError: logger.error('Impossible to create file: %s', h264_ts_filename) return None h264_ts_output.write('# timestamp format v2\n') ts = 0 for _ in range(nb_frames): ts = ts+ceil(1000/framerate) h264_ts_output.write(f'{ts:d}\n') h264_ts_output.flush() h264_ts_output.seek(0) temporaries.append(h264_ts_output) for memfd in memfds: close(memfd) return h264_output, h264_ts_output, mkv_output else: # Nothing to be done. We are already at a i-frame boundary. return None, None # Merge a list of mkv files passed as input, and produce a new MKV as output @typechecked def merge_mkvs(mkvmerge_path:str, inputs: list[IO[bytes]], output_name:str, concatenate: bool=True, timestamps: dict[int, IO[str]] | None = None) -> IO[bytes]|None: logger = logging.getLogger(__name__) if timestamps is None: timestamps = {} fds = [] try: out = open(output_name, 'wb+') except OSError: logger.error('Impossible to create file: %s', output_name) return None outfd = out.fileno() lseek(outfd, 0, SEEK_SET) fds.append(outfd) set_inheritable(outfd, True) # Timestamps of merged tracks are modified by the length of the preceding track. # The default mode ('file') is using the largest timestamp of the whole file which may create # desynchronize video and sound. merge_params = [mkvmerge_path, '--append-mode', 'track'] first = True partnum = 0 for mkv in inputs: if mkv is not None: fd = mkv.fileno() fds.append(fd) set_inheritable(fd, True) # If we pass a timestamps file associated with the considered track, use it. if partnum in timestamps: tsfd = timestamps[partnum].fileno() lseek(tsfd, 0, SEEK_SET) fds.append(tsfd) set_inheritable(tsfd, True) merge_params.extend(['--timestamps', f'{partnum:d}:/proc/self/fd/{tsfd:d}']) if first: merge_params.append(f'/proc/self/fd/{fd:d}') first = False elif concatenate: merge_params.append(f'+/proc/self/fd/{fd:d}') else: merge_params.append(f'/proc/self/fd/{fd:d}') partnum+=1 merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}']) # We merge all files. warnings = [] env = {**os.environ, 'LANG': 'C'} logger.debug('Executing: LANG=C %s', merge_params) with Popen(merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge: pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%', desc='Merging') for line in pb: if line.startswith('Progress :'): p = re.compile('^Progress : (?P[0-9]{1,3})%$') m = p.match(line) if m is None: logger.error('Impossible to parse progress') pb.n = int(m['progress']) pb.update() elif line.startswith('Warning'): warnings.append(line) status = mkvmerge.wait() if status == 1: logger.warning('Extraction returns warning') for w in warnings: logger.warning(w) elif status == 2: logger.error('Extraction returns errors') for fd in fds: set_inheritable(fd, False) return out def find_subtitles_tracks(ffprobe_path:str, input_file: IO[bytes]) -> dict|None: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) command = [ffprobe_path, '-loglevel','quiet', '-i', f'/proc/self/fd/{infd:d}', '-select_streams', 's', '-show_entries', 'stream=index:stream_tags=language', '-of', 'json'] logger.debug('Executing: %s', command) with Popen(command, stdout=PIPE, close_fds=False) as ffprobe: out, _ = ffprobe.communicate() out = json.load(BytesIO(out)) if 'streams' in out: return out['streams'] else: logger.error('Impossible to retrieve format of file') ffprobe.wait() return None @typechecked def extract_track_from_mkv(mkvextract_path: str, input_file: IO[bytes], index, output_file: IO[bytes], timestamps) -> None: logger = logging.getLogger(__name__) infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) outfd = output_file.fileno() lseek(outfd, 0, SEEK_SET) set_inheritable(outfd, True) tsfd = timestamps.fileno() lseek(tsfd, 0, SEEK_SET) set_inheritable(tsfd, True) params = [ mkvextract_path, f'/proc/self/fd/{infd:d}', 'tracks', f'{index:d}:/proc/self/fd/{outfd:d}', 'timestamps_v2', f'{index:d}:/proc/self/fd/{tsfd:d}'] env = {**os.environ, 'LANG': 'C'} logger.debug('Executing: LANG=C %s', params) with Popen(params, stdout=PIPE, close_fds=False, env=env) as extract: pb = tqdm(TextIOWrapper(extract.stdout, encoding="utf-8"), total=100, unit='%', desc='Extraction of track') for line in pb: if line.startswith('Progress :'): p = re.compile('^Progress : (?P[0-9]{1,3})%$') m = p.match(line) if m is None: logger.error('Impossible to parse progress') pb.update(int(m['progress'])-pb.n) pb.update(100-pb.n) pb.refresh() pb.close() extract.wait() if extract.returncode != 0: logger.error('Mkvextract returns an error code: %d', extract.returncode) else: logger.info('Track %d was succesfully extracted.', index) @typechecked def remove_video_tracks_from_mkv(mkvmerge_path:str, input_file: IO[bytes], output_file: IO[bytes]) -> None: logger = logging.getLogger(__name__) outfd = output_file.fileno() infd = input_file.fileno() lseek(infd, 0, SEEK_SET) lseek(outfd, 0, SEEK_SET) set_inheritable(infd, True) set_inheritable(outfd, True) params = [ mkvmerge_path, '-o', f'/proc/self/fd/{outfd:d}', '-D', f'/proc/self/fd/{infd:d}'] logger.debug('Executing: LANG=C %s', params) env = {**os.environ, 'LANG': 'C'} with Popen(params, stdout=PIPE, close_fds=False, env=env) as remove: pb = tqdm(TextIOWrapper(remove.stdout, encoding="utf-8"), total=100, unit='%', desc='Removal of video track:') for line in pb: if line.startswith('Progress :'): p = re.compile('^Progress : (?P[0-9]{1,3})%$') m = p.match(line) if m is None: logger.error('Impossible to parse progress') pb.update(int(m['progress'])-pb.n) pb.update(100-pb.n) pb.refresh() pb.close() remove.wait() if remove.returncode != 0: logger.error('Mkvmerge returns an error code: %d', remove.returncode) else: logger.info('Video tracks were succesfully extracted.') @typechecked def remux_srt_subtitles(mkvmerge_path:str, input_file: IO[bytes], output_filename: str, subtitles) -> None: logger = logging.getLogger(__name__) try: out = open(output_filename, 'w', encoding='utf8') except OSError: logger.error('Impossible to create file: %s', output_filename) return None outfd = out.fileno() infd = input_file.fileno() lseek(infd, 0, SEEK_SET) set_inheritable(infd, True) set_inheritable(outfd, True) mkv_merge_params = [mkvmerge_path, f'/proc/self/fd/{infd:d}'] for fd, lang in subtitles: lseek(fd, 0, SEEK_SET) set_inheritable(fd, True) mkv_merge_params.extend(['--language', f'0:{lang}', f'/proc/self/fd/{fd:d}']) mkv_merge_params.extend(['-o', f'/proc/self/fd/{outfd:d}']) warnings = [] env = {**os.environ, 'LANG': 'C'} logger.info('Remux subtitles: %s', mkv_merge_params) with Popen(mkv_merge_params, stdout=PIPE, close_fds=False, env=env) as mkvmerge: pb = tqdm(TextIOWrapper(mkvmerge.stdout, encoding="utf-8"), total=100, unit='%', desc='Remux subtitles:') for line in pb: if line.startswith('Progress :'): p = re.compile('^Progress : (?P[0-9]{1,3})%$') m = p.match(line) if m is None: logger.error('Impossible to parse progress') pb.n = int(m['progress']) pb.update() elif line.startswith('Warning'): warnings.append(line) status = mkvmerge.wait() if status == 1: logger.warning('Remux subtitles returns warning') for w in warnings: logger.warning(w) elif status == 2: logger.error('Remux subtitles returns errors') return None @typechecked def concatenate_h264_parts(h264parts: list[IO[bytes]], output: IO[bytes]) -> None: logger = logging.getLogger(__name__) total_length = 0 for h264 in h264parts: fd = h264.fileno() total_length += fstat(fd).st_size logger.info('Total length: %d', total_length) outfd = output.fileno() lseek(outfd, 0, SEEK_SET) pb = tqdm(total=total_length, unit='bytes', desc='Concatenation') for h264 in h264parts: fd = h264.fileno() lseek(fd, 0, SEEK_SET) while True: buf = read(fd, 1000000) if buf is None or len(buf) == 0: break pos = 0 while pos < len(buf): nb_bytes = write(outfd, buf[pos:]) pb.update(nb_bytes) pos += nb_bytes def concatenate_h264_ts_parts(h264_ts_parts: list[IO[bytes]], output: IO[bytes]) -> None: logger = logging.getLogger(__name__) header = '# timestamp format v2\n' output.write(header) last = 0. first = True for part in h264_ts_parts: if first: offset = last else: # TODO: take framerate into account offset = last + 40 logger.debug('Parsing file: %s. Offset=%d', part, offset) isheader = part.readline() if (not isheader) or (isheader != header): logger.error('Impossible to find a valid header: "%s"', isheader) exit(-1) while True: line = part.readline() if not line: break ts = offset + float(line) last = max(last,ts) output.write(f'{ts:f}\n') if first: first = False # TODO: finish this procedure def do_coarse_processing(ffmpeg_path:str, ffprobe_path:str, mkvmerge_path:str, input_file: IO[bytes], begin, end, nb_frames, framerate, files_prefix, streams, width, height, temporaries, dump_mem_fd) -> None: # pylint: disable=W0613 logger = logging.getLogger(__name__) # Internal video with all streams (video, audio and subtitles) internal_mkv_name = f'{files_prefix}.mkv' try: internal_mkv = open(internal_mkv_name, 'wb+') except OSError: logger.error('Impossible to create file: %s', internal_mkv_name) exit(-1) # Extract internal part of MKV extract_mkv_part(mkvmerge_path=mkvmerge_path, input_file=input_file, output_file=internal_mkv, begin=begin, end=end) temporaries.append(internal_mkv)