Decodificação de Áudio com FFmpeg 5.x usando Java

Para desenvolvedores que trabalham com processamento de áudio e vídeo em Java sem profundo conhecimento em C/C++, a biblioteca org.bytedeco:ffmpeg-platform oferece uma enterface prática para as APIs do FFmpeg. Este guia demonstra como implementar a decodificação de áudio de arquivos MP4.

  1. Implementação do Código

O exemplo seguinte extrai e decodifica o fluxo de áudio de um arquivo MP4, aplicando resampling para o formato S16:

import org.bytedeco.ffmpeg.global.avcodec.*;
import org.bytedeco.ffmpeg.global.avutil.*;
import org.bytedeco.ffmpeg.global.swresample.*;
import org.bytedeco.ffmpeg.swresample.SwrContext;
import org.bytedeco.ffmpeg.avcodec.AVCodecContext;
import org.bytedeco.ffmpeg.avformat.AVFormatContext;
import org.bytedeco.ffmpeg.avutil.AVFrame;
import org.bytedeco.ffmpeg.avutil.AVPacket;
import org.bytedeco.javacpp.*;

import java.io.FileOutputStream;
import java.io.IOException;
import java.io.OutputStream;
import java.util.Objects;

public class AudioDecoder {

    public static void main(String[] arguments) throws IOException {
        processAudio("input.mp4", "output.pcm");
    }

    public static void processAudio(String sourceFile, String destinationFile) throws IOException {
        AVFormatContext formatCtx = new AVFormatContext(null);
        AVCodecContext decoderCtx = null;
        SwrContext resampleCtx = null;
        AVFrame audioFrame = null;
        AVPacket dataPacket = null;
        PointerPointer<BytePointer> outputBuffer = new PointerPointer<>(1);
        IntPointer outputLineSize = new IntPointer(1);

        AVChannelLayout targetChannelLayout = new AVChannelLayout();
        targetChannelLayout.nb_channels(2);
        targetChannelLayout.order(AV_CHANNEL_ORDER_NATIVE);
        targetChannelLayout.u_mask(AV_CH_LAYOUT_STEREO);
        
        int targetSampleRate = 44100;
        int targetChannels = 0;
        int targetSampleFormat = AV_SAMPLE_FMT_S16;
        long targetSamples = 0;
        long maxTargetSamples = 0;

        try (OutputStream fileStream = new FileOutputStream(destinationFile)) {
            int result = avformat_open_input(formatCtx, sourceFile, null, null);
            if (result < 0) {
                throw new IOException(result + ": erro ao abrir arquivo de entrada");
            }

            result = avformat_find_stream_info(formatCtx, (AVDictionary) null);
            if (result < 0) {
                throw new IOException(result + ": erro ao obter informações do stream");
            }

            int totalStreams = formatCtx.nb_streams();
            int audioStreamIndex = -1;
            
            for (int i = 0; i < totalStreams; i++) {
                if (formatCtx.streams(i).codecpar().codec_type() == AVMEDIA_TYPE_AUDIO) {
                    audioStreamIndex = i;
                    break;
                }
            }
            
            if (audioStreamIndex == -1) {
                throw new IOException("stream de áudio não encontrado");
            }

            AVCodec decoder = avcodec_find_decoder(formatCtx.streams(audioStreamIndex).codecpar().codec_id());
            if (decoder == null) {
                throw new IOException("codec de áudio não encontrado");
            }

            decoderCtx = avcodec_alloc_context3(decoder);
            if (decoderCtx == null) {
                throw new IOException("falha ao alocar contexto do codec");
            }

            result = avcodec_parameters_to_context(decoderCtx, formatCtx.streams(audioStreamIndex).codecpar());
            if (result < 0) {
                throw new IOException(result + ": erro ao transferir parâmetros do codec");
            }

            result = avcodec_open2(decoderCtx, decoder, (AVDictionary) null);
            if (result < 0) {
                throw new IOException(result + ": erro ao inicializar codec");
            }

            resampleCtx = swr_alloc();
            if (resampleCtx == null) {
                throw new IOException("falha ao alocar contexto de resampling");
            }

            av_opt_set_chlayout(resampleCtx, "in_chlayout", decoderCtx.ch_layout(), 0);
            av_opt_set_int(resampleCtx, "in_sample_rate", decoderCtx.sample_rate(), 0);
            av_opt_set_sample_fmt(resampleCtx, "in_sample_fmt", decoderCtx.sample_fmt(), 0);

            av_opt_set_chlayout(resampleCtx, "out_chlayout", targetChannelLayout, 0);
            av_opt_set_int(resampleCtx, "out_sample_rate", targetSampleRate, 0);
            av_opt_set_sample_fmt(resampleCtx, "out_sample_fmt", targetSampleFormat, 0);

            result = swr_init(resampleCtx);
            if (result < 0) {
                throw new IOException(result + ": erro ao inicializar resampling");
            }

            audioFrame = av_frame_alloc();
            if (audioFrame == null) {
                throw new IOException("falha ao alocar frame de áudio");
            }

            dataPacket = av_packet_alloc();
            if (dataPacket == null) {
                throw new IOException("falha ao alocar packet");
            }

            targetSamples = av_rescale_rnd(decoderCtx.frame_size(), targetSampleRate, decoderCtx.sample_rate(), AV_ROUND_UP);
            maxTargetSamples = targetSamples;
            targetChannels = targetChannelLayout.nb_channels();

            result = av_samples_alloc_array_and_samples(outputBuffer, outputLineSize, targetChannels, 
                    (int) targetSamples, targetSampleFormat, 0);
            if (result < 0) {
                throw new IOException(result + ": erro ao alocar buffer de saída");
            }

            int outputBufferSize;
            byte[] audioData;

            while (true) {
                result = av_read_frame(formatCtx, dataPacket);
                if (result == AVERROR_EAGAIN() || result == AVERROR_EOF) {
                    break;
                } else if (result < 0) {
                    throw new IOException(result + ": erro ao ler frame");
                }

                if (dataPacket.stream_index() != audioStreamIndex) {
                    continue;
                }

                result = avcodec_send_packet(decoderCtx, dataPacket);
                if (result < 0) {
                    throw new IOException(result + ": erro ao enviar packet para decodificação");
                }

                while (true) {
                    result = avcodec_receive_frame(decoderCtx, audioFrame);
                    if (result == AVERROR_EAGAIN() || result == AVERROR_EOF) {
                        break;
                    } else if (result < 0) {
                        throw new IOException(result + ": erro ao receber frame decodificado");
                    }

                    targetSamples = avutil.av_rescale_rnd(
                            swresample.swr_get_delay(resampleCtx, decoderCtx.sample_rate()) + audioFrame.nb_samples(),
                            decoderCtx.sample_rate(), decoderCtx.sample_rate(), AV_ROUND_UP);

                    if (targetSamples > maxTargetSamples) {
                        av_freep(outputBuffer.get());
                        result = av_samples_alloc(outputBuffer, outputLineSize, targetChannels, 
                                (int) targetSamples, targetSampleFormat, 1);
                        if (result < 0) {
                            break;
                        }
                        maxTargetSamples = targetSamples;
                    }

                    result = swr_convert(resampleCtx, outputBuffer, (int) targetSamples, 
                            audioFrame.data(), decoderCtx.frame_size());
                    if (result < 0) {
                        throw new IOException(result + ": erro durante resampling");
                    }

                    outputBufferSize = av_samples_get_buffer_size(outputLineSize, targetChannels, 
                            result, targetSampleFormat, 1);
                    if (outputBufferSize < 0) {
                        throw new IOException(result + ": erro ao calcular tamanho do buffer");
                    }

                    audioData = new byte[outputBufferSize];
                    outputBuffer.get(BytePointer.class, 0).get(audioData);
                    fileStream.write(audioData);
                    System.out.printf("Amostras processadas: %d, Tamanho do buffer: %d bytes\n", result, outputBufferSize);
                }
            }

            String formatSpec = "s16le";
            System.out.printf(
                    "Resampling concluído. Reproduza o arquivo com:\n"
                            + "ffplay -f %s -channel_layout %s -channels %d -ar %d %s\n",
                    formatSpec, AV_CH_LAYOUT_STEREO, targetChannels, targetSampleRate, destinationFile);
        } finally {
            outputBuffer.close();
            outputLineSize.close();
            if (dataPacket != null) {
                av_packet_free(dataPacket);
            }
            if (audioFrame != null) {
                av_frame_free(audioFrame);
            }
            if (decoderCtx != null) {
                avcodec_free_context(decoderCtx);
            }
            if (resampleCtx != null) {
                swr_free(resampleCtx);
            }
            avformat_close_input(formatCtx);
        }
    }
}

  1. Reprodução do Áudio Decodificado

O arquivo PCM gerado pode ser reproduzido diretamente com o FFplay usando o seguinte comando:

ffplay -f s16le -channel_layout 3 -channels 2 -ar 44100 output.pcm

Este comando configura o FFplay para interpretar os dados como áudio PCM de 16 bits litle-endian com dois canais estéreo a 44.1kHz.

Tags: FFmpeg java áudio Decodificação Resampling

Publicado em 9-18 07:05