Files
removeads/src/tscut/tools/ffmpeg.py
T

389 lines
18 KiB
Python

# SPDX-License-Identifier: GPL-2.0-or-later
#
# Copyright (C) 2026 Frédéric Tronel
import logging
from datetime import timedelta
from io import TextIOWrapper
from math import ceil, floor, log
from os import SEEK_SET, close, lseek, memfd_create, read, set_inheritable, write
from subprocess import PIPE, Popen
from typing import IO, BinaryIO
from tqdm import tqdm
from typeguard import typechecked
from tscut.exceptions import ExternalToolError, InvalidMediaError, TemporaryFileError
from tscut.temporaries import TemporaryFiles
from tscut.tools.ffprobe import (
get_frames_in_stream,
get_video_dimensions,
with_subtitles,
)
from tscut.tools.ppm import dump_ppm
from tscut.tools.timeframe import get_packet_duration, parse_timestamp
logger = logging.getLogger(__name__)
@typechecked
def ffmpeg_convert(ffmpeg_path:str, ffprobe_path:str, input_file: IO[bytes], input_format:str,
output_file: IO[bytes], output_format:str, duration: timedelta) -> None:
width, height = get_video_dimensions(ffprobe_path, input_file)
if width is None or height is None:
return
subtitles = with_subtitles(ffprobe_path, input_file)
infd = input_file.fileno()
outfd = output_file.fileno()
set_inheritable(infd, True)
set_inheritable(outfd, True)
log_level = [] if logger.getEffectiveLevel() == logging.DEBUG else ['-loglevel', 'quiet']
params = [ffmpeg_path, '-y',]+log_level+['-progress', '/dev/stdout', '-canvas_size',
f'{width:d}x{height:d}', '-f', input_format,
'-i', f'/proc/self/fd/{infd:d}', '-map', '0:v',
'-map', '0:a']
if subtitles:
params.extend(['-map', '0:s'])
params.extend(['-bsf:v', 'h264_mp4toannexb,dump_extra=freq=keyframe', '-vcodec', 'copy',
'-acodec', 'copy'])
if subtitles:
params.extend(['-scodec', 'dvdsub'])
params.extend(['-r:0', '25', '-f', output_format, f'/proc/self/fd/{outfd:d}'])
logger.debug('Executing %s', params)
with Popen(params, stdout=PIPE, close_fds=False) as ffmpeg:
assert ffmpeg.stdout is not None
pb = tqdm(TextIOWrapper(ffmpeg.stdout, encoding="utf-8"),
total=int(duration/timedelta(seconds=1)), unit='s', desc='Conversion')
for line in pb:
if line.startswith('out_time='):
ts_str = line.split('=')[1].strip()
ts = parse_timestamp(ts_str)
if ts is not None:
pb.n = int(ts/timedelta(seconds=1))
pb.update()
status = ffmpeg.wait()
if status != 0:
raise ExternalToolError(f"Conversion failed with status code: {status:d}")
@typechecked
def extract_pictures(ffmpeg_path:str, input_file:IO[bytes], begin:timedelta, nb_frames:int,
width:int=640, height:int=480) -> tuple[bytes,int]:
"""
Extract pictures from a video file using FFmpeg.
This function runs the FFmpeg binary to extract a specified number of frames from a video file,
starting at a given time.
The extracted frames are stored in memory as PPM images and returned as a tuple containing
the image data and a file descriptor to the memory created by memfd_create.
Args:
ffmpeg_path (str): The path to the FFmpeg binary.
input_file (IO[bytes]): The input video file.
begin (timedelta): The start time of the extraction.
nb_frames (int): The number of frames to extract.
width (int, optional): The width of the extracted images. Defaults to 640.
height (int, optional): The height of the extracted images. Defaults to 480.
Returns:
tuple[bytes, int] | tuple[None, None]:
- A tuple containing the extracted image data as bytes and a file descriptor
- A tuple containing None, None if the extraction fails
"""
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
outfd = memfd_create('pictures', flags=0)
set_inheritable(outfd, True)
# PPM header
# "P6\nWIDTH HEIGHT\n255\n"
header_len=2+1+ceil(log(width, 10))+1+ceil(log(height, 10))+1+3+1
logger.debug('Header length: %d', header_len)
image_length = width*height*3+header_len
length = image_length*nb_frames
logger.debug("Estimated length: %d", length)
command = [ffmpeg_path, '-loglevel', 'quiet' ,'-y', '-ss', f'{begin}', '-i',
f'/proc/self/fd/{infd}', '-s', f'{width:d}x{height:d}', '-vframes', f'{nb_frames:d}',
'-c:v', 'ppm','-f', 'image2pipe', f'/proc/self/fd/{outfd:d}']
logger.debug('Executing: %s', command)
images = b''
with Popen(command, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
raise ExternalToolError(f"Conversion failed with status code: {status:d}")
lseek(outfd, 0, SEEK_SET)
images = read(outfd,length)
if len(images) != length:
raise InvalidMediaError(f"Received {len(images)} bytes but {length} were expected.")
lseek(outfd, 0, SEEK_SET)
return images, outfd
@typechecked
def extract_sound(ffmpeg_path:str, input_file: IO[bytes], begin:timedelta, output_filename:str,
packet_duration:int, sub_channel:int=0,
nb_packets:int=0, sample_rate:int=48000,
nb_channels:int=2) -> tuple[bytes,int]:
outfd = memfd_create(output_filename, flags=0)
infd = input_file.fileno()
lseek(infd, 0, SEEK_SET)
set_inheritable(infd, True)
set_inheritable(outfd, True)
sound = b''
length = int(nb_channels*sample_rate*4*nb_packets*packet_duration/1000)
command = [ffmpeg_path, '-y', '-loglevel', 'quiet', '-ss', f'{begin}',
'-i', f'/proc/self/fd/{infd}', f'-frames:a:{sub_channel:d}', f'{nb_packets+1:d}',
'-c:a', 'pcm_s32le', '-sample_rate', f'{sample_rate:d}',
'-channels', f'{nb_channels:d}', '-f', 's32le', f'/proc/self/fd/{outfd:d}']
logger.debug('Executing: %s', command)
with Popen(command, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
raise ExternalToolError(f"Sound extraction returns error code: {status}")
lseek(outfd, 0, SEEK_SET)
sound = read(outfd, length)
if len(sound) != length:
raise InvalidMediaError(f"Received {len(sound)} bytes but {length} were expected (\
channels={nb_channels}, freq={sample_rate} packets={nb_packets},\
duration={packet_duration} ms).")
return sound, outfd
@typechecked
def extract_all_streams(ffmpeg_path:str, ffprobe_path:str, input_file:IO[bytes], begin:timedelta,
end:timedelta, streams, files_prefix, nb_frames:int, framerate:float,
width:int, height:int, temporaries:TemporaryFiles,
dump_mem_fd:bool=False) -> tuple[BinaryIO|None,
TextIOWrapper|None,
BinaryIO|None]:
# The command line for encoding only video track
video_encoder_params = [ ffmpeg_path, '-y', '-loglevel', 'quiet']
video_input_params = []
video_codec_params = []
# The command line to create a MKV file with the rest of tracks
generic_encoder_params = [ ffmpeg_path, '-y', '-loglevel', 'quiet' ]
generic_input_params = []
generic_codec_params = []
if begin < end:
video_id=0
audio_id=0
subtitle_id=0
memfds = []
for stream in streams:
if stream['codec_type'] == 'video':
logger.info("Extracting %d frames of video stream v:%d", nb_frames, video_id)
sar = stream['sample_aspect_ratio']
dar = stream['display_aspect_ratio']
pixel_format = stream['pix_fmt']
color_range = stream['color_range']
color_space =stream['color_space']
color_transfer = stream['color_transfer']
color_primaries = stream['color_primaries']
level_int = int(stream['level'])
level = f'{floor(level_int/10):d}.{level_int%10:d}'
chroma_location = stream['chroma_location']
field_order = stream
match field_order:
case 'progressive':
interlaced_options = ['-field_order', '0']
case 'tt':
interlaced_options = ['-top', '1', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order', '1']
case 'bb':
interlaced_options = ['-top', '0', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order','2']
case 'tb':
interlaced_options = ['-top', '1', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order', '3']
case 'bt':
interlaced_options = ['-top', '0', f'-flags:v:{video_id:d}', '+ilme+ildct',
'-field_order', '4']
case _:
interlaced_options = []
# ======================================= #
# TODO: adjust SAR and DAR
# https://superuser.com/questions/907933/correct-aspect-ratio-without-re-encoding-video-file
# SAR: -aspect width:height
# DAR: -bsf:v sample_aspect_ratio=1:video_format
logger.warning('Missing SAR adjustment for: %s', sar)
logger.warning('Missing DAR adjustment for: %s', dar)
logger.warning('Missing treatment for chroma location: %s', chroma_location)
codec = stream['codec_name']
images_bytes, memfd = extract_pictures(ffmpeg_path, input_file=input_file,
begin=begin, nb_frames=nb_frames,
width=width, height=height)
memfds.append(memfd)
if dump_mem_fd:
dump_ppm(images_bytes, f'{files_prefix}-{video_id:d}', temporaries)
# We rewind to zero the memory file descriptor
lseek(memfd, 0, SEEK_SET)
set_inheritable(memfd, True)
video_input_params.extend(['-framerate', f'{framerate:f}', '-f', 'image2pipe', '-i',
f'/proc/self/fd/{memfd:d}'])
video_codec_params.extend([f'-c:v:{video_id:d}', codec, f'-level:v:{video_id:d}',
level, '-pix_fmt', pixel_format])
video_codec_params.extend(interlaced_options)
video_codec_params.extend([f'-colorspace:v:{video_id:d}', color_space,
f'-color_primaries:v:{video_id:d}', color_primaries,
f'-color_trc:v:{video_id:d}', color_transfer,
f'-color_range:v:{video_id:d}', color_range])
video_id=video_id+1
elif stream['codec_type'] == 'audio':
logger.debug('Audio stream: %s', stream)
sample_rate = int(stream['sample_rate'])
nb_channels = int(stream['channels'])
bit_rate = int(stream['bit_rate']) if 'bit_rate' in stream else 128000
codec = stream['codec_name']
if 'tags' in stream and 'language' in stream['tags']:
generic_codec_params.extend([f'-metadata:s:a:{audio_id:d}',
f"language={stream['tags']['language']}"])
packets = get_frames_in_stream(ffprobe_path, input_file=input_file, begin=begin,
end=end, stream_kind='a', sub_stream_id=audio_id)
if packets is None:
raise InvalidMediaError("Impossible to retrieve audio packets")
nb_packets = len(packets)
logger.debug("Found %d packets to be extracted from audio track.", nb_packets)
if nb_packets > 0:
packet_duration = get_packet_duration(packets[0])
else:
packet_duration = 0
logger.info("Extracting %d packets of audio stream: a:%d" , nb_packets, audio_id)
tmpname = f'{files_prefix}-{audio_id:d}.pcm'
sound_bytes, memfd = extract_sound(ffmpeg_path=ffmpeg_path, input_file=input_file,
begin=begin, nb_packets=nb_packets,
packet_duration=packet_duration,
output_filename=tmpname,
sample_rate=sample_rate, nb_channels=nb_channels)
memfds.append(memfd)
if dump_mem_fd:
try:
with open(tmpname,'wb') as output:
temporaries.add(output)
outfd = output.fileno()
pos = 0
while pos < len(sound_bytes):
pos+=write(outfd, sound_bytes[pos:])
except OSError as e:
raise TemporaryFileError(f"Impossible to create file: {tmpname}") from e
# We rewind to zero the memory file descriptor
lseek(memfd, 0, SEEK_SET)
set_inheritable(memfd, True)
generic_input_params.extend(['-f', 's32le', '-ar', f'{sample_rate:d}', '-ac',
f'{nb_channels:d}', '-i', f'/proc/self/fd/{memfd:d}'])
generic_codec_params.extend([f'-c:a:{audio_id:d}', codec, f'-b:a:{audio_id:d}',
f'{bit_rate:d}'])
audio_id=audio_id+1
elif stream['codec_type'] == 'subtitle':
logger.info("Extracting a subtitle stream: s:%d", subtitle_id)
codec = stream['codec_name']
generic_input_params.extend(['-i', './empty.idx'])
if 'tags' in stream and 'language' in stream['tags']:
generic_codec_params.extend([f'-metadata:s:s:{subtitle_id:d}',
f"language={stream['tags']['language']}"])
generic_codec_params.extend([f'-c:s:{subtitle_id:d}', 'copy'])
subtitle_id=subtitle_id+1
else:
logger.error("Unknown stream type: %s", stream['codec_type'])
# Create a new MKV movie with all streams (except videos) that have been extracted.
generic_encoder_params.extend(generic_input_params)
for index in range(audio_id+subtitle_id):
generic_encoder_params.extend(['-map', f'{index:d}'])
generic_encoder_params.extend(generic_codec_params)
mkv_filename = f'{files_prefix}.mkv'
try:
mkv_output = open(mkv_filename,'wb+')
except OSError as e:
raise TemporaryFileError(f"Impossible to create file: {mkv_filename}") from e
mkvoutfd = mkv_output.fileno()
set_inheritable(mkvoutfd, True)
generic_encoder_params.extend(['-f', 'matroska', f'/proc/self/fd/{mkvoutfd:d}'])
logger.info('Encoding all streams (except video) into a MKV file: %s', mkv_filename)
logger.debug('Executing: %s', generic_encoder_params)
with Popen(generic_encoder_params, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
raise ExternalToolError(f"Encoding failed with status code: {status}")
temporaries.add(mkv_output)
h264_filename = f'{files_prefix}.h264'
try:
h264_output = open(h264_filename,'wb+')
except OSError as e:
raise TemporaryFileError(f"Impossible to create file {h264_filename}") from e
h264outfd = h264_output.fileno()
set_inheritable(h264outfd, True)
video_encoder_params.extend(video_input_params)
video_encoder_params.extend(video_codec_params)
video_encoder_params.extend([ '-x264opts', f'keyint=1:sps-id={1:d}','-bsf:v',
'h264_mp4toannexb,dump_extra=freq=keyframe,h264_metadata=\
overscan_appropriate_flag=1:sample_aspect_ratio=1:video_format=\
0:chroma_sample_loc_type=0','-f', 'h264',
f'/proc/self/fd/{h264outfd:d}'])
logger.info('Encoding video into a H264 file: %s', h264_filename)
logger.debug('Executing: %s', video_encoder_params)
with Popen(video_encoder_params, stdout=PIPE, close_fds=False) as ffmpeg:
status = ffmpeg.wait()
if status != 0:
raise ExternalToolError(f"Encoding failed with status code: {status:d}")
temporaries.add(h264_output)
h264_ts_filename = f'{files_prefix}-ts.txt'
try:
h264_ts_output = open(h264_ts_filename,'w+', encoding='utf8')
except OSError as e:
raise TemporaryFileError(f"Impossible to create file: {h264_ts_filename}") from e
h264_ts_output.write('# timestamp format v2\n')
ts = 0
for _ in range(nb_frames):
ts = ts+ceil(1000/framerate)
h264_ts_output.write(f'{ts:d}\n')
h264_ts_output.flush()
h264_ts_output.seek(0)
temporaries.add(h264_ts_output)
for memfd in memfds:
close(memfd)
return h264_output, h264_ts_output, mkv_output
# Nothing to be done. We are already at a i-frame boundary.
return None, None, None