Para desenvolvedores que trabalham com processamento de áudio e vídeo em Java sem profundo conhecimento em C/C++, a biblioteca org.bytedeco:ffmpeg-platform oferece uma enterface prática para as APIs do FFmpeg. Este guia demonstra como implementar a decodificação de áudio de arquivos MP4.
- Implementação do Código
O exemplo seguinte extrai e decodifica o fluxo de áudio de um arquivo MP4, aplicando resampling para o formato S16:
import org.bytedeco.ffmpeg.global.avcodec.*;
import org.bytedeco.ffmpeg.global.avutil.*;
import org.bytedeco.ffmpeg.global.swresample.*;
import org.bytedeco.ffmpeg.swresample.SwrContext;
import org.bytedeco.ffmpeg.avcodec.AVCodecContext;
import org.bytedeco.ffmpeg.avformat.AVFormatContext;
import org.bytedeco.ffmpeg.avutil.AVFrame;
import org.bytedeco.ffmpeg.avutil.AVPacket;
import org.bytedeco.javacpp.*;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.OutputStream;
import java.util.Objects;
public class AudioDecoder {
public static void main(String[] arguments) throws IOException {
processAudio("input.mp4", "output.pcm");
}
public static void processAudio(String sourceFile, String destinationFile) throws IOException {
AVFormatContext formatCtx = new AVFormatContext(null);
AVCodecContext decoderCtx = null;
SwrContext resampleCtx = null;
AVFrame audioFrame = null;
AVPacket dataPacket = null;
PointerPointer<BytePointer> outputBuffer = new PointerPointer<>(1);
IntPointer outputLineSize = new IntPointer(1);
AVChannelLayout targetChannelLayout = new AVChannelLayout();
targetChannelLayout.nb_channels(2);
targetChannelLayout.order(AV_CHANNEL_ORDER_NATIVE);
targetChannelLayout.u_mask(AV_CH_LAYOUT_STEREO);
int targetSampleRate = 44100;
int targetChannels = 0;
int targetSampleFormat = AV_SAMPLE_FMT_S16;
long targetSamples = 0;
long maxTargetSamples = 0;
try (OutputStream fileStream = new FileOutputStream(destinationFile)) {
int result = avformat_open_input(formatCtx, sourceFile, null, null);
if (result < 0) {
throw new IOException(result + ": erro ao abrir arquivo de entrada");
}
result = avformat_find_stream_info(formatCtx, (AVDictionary) null);
if (result < 0) {
throw new IOException(result + ": erro ao obter informações do stream");
}
int totalStreams = formatCtx.nb_streams();
int audioStreamIndex = -1;
for (int i = 0; i < totalStreams; i++) {
if (formatCtx.streams(i).codecpar().codec_type() == AVMEDIA_TYPE_AUDIO) {
audioStreamIndex = i;
break;
}
}
if (audioStreamIndex == -1) {
throw new IOException("stream de áudio não encontrado");
}
AVCodec decoder = avcodec_find_decoder(formatCtx.streams(audioStreamIndex).codecpar().codec_id());
if (decoder == null) {
throw new IOException("codec de áudio não encontrado");
}
decoderCtx = avcodec_alloc_context3(decoder);
if (decoderCtx == null) {
throw new IOException("falha ao alocar contexto do codec");
}
result = avcodec_parameters_to_context(decoderCtx, formatCtx.streams(audioStreamIndex).codecpar());
if (result < 0) {
throw new IOException(result + ": erro ao transferir parâmetros do codec");
}
result = avcodec_open2(decoderCtx, decoder, (AVDictionary) null);
if (result < 0) {
throw new IOException(result + ": erro ao inicializar codec");
}
resampleCtx = swr_alloc();
if (resampleCtx == null) {
throw new IOException("falha ao alocar contexto de resampling");
}
av_opt_set_chlayout(resampleCtx, "in_chlayout", decoderCtx.ch_layout(), 0);
av_opt_set_int(resampleCtx, "in_sample_rate", decoderCtx.sample_rate(), 0);
av_opt_set_sample_fmt(resampleCtx, "in_sample_fmt", decoderCtx.sample_fmt(), 0);
av_opt_set_chlayout(resampleCtx, "out_chlayout", targetChannelLayout, 0);
av_opt_set_int(resampleCtx, "out_sample_rate", targetSampleRate, 0);
av_opt_set_sample_fmt(resampleCtx, "out_sample_fmt", targetSampleFormat, 0);
result = swr_init(resampleCtx);
if (result < 0) {
throw new IOException(result + ": erro ao inicializar resampling");
}
audioFrame = av_frame_alloc();
if (audioFrame == null) {
throw new IOException("falha ao alocar frame de áudio");
}
dataPacket = av_packet_alloc();
if (dataPacket == null) {
throw new IOException("falha ao alocar packet");
}
targetSamples = av_rescale_rnd(decoderCtx.frame_size(), targetSampleRate, decoderCtx.sample_rate(), AV_ROUND_UP);
maxTargetSamples = targetSamples;
targetChannels = targetChannelLayout.nb_channels();
result = av_samples_alloc_array_and_samples(outputBuffer, outputLineSize, targetChannels,
(int) targetSamples, targetSampleFormat, 0);
if (result < 0) {
throw new IOException(result + ": erro ao alocar buffer de saída");
}
int outputBufferSize;
byte[] audioData;
while (true) {
result = av_read_frame(formatCtx, dataPacket);
if (result == AVERROR_EAGAIN() || result == AVERROR_EOF) {
break;
} else if (result < 0) {
throw new IOException(result + ": erro ao ler frame");
}
if (dataPacket.stream_index() != audioStreamIndex) {
continue;
}
result = avcodec_send_packet(decoderCtx, dataPacket);
if (result < 0) {
throw new IOException(result + ": erro ao enviar packet para decodificação");
}
while (true) {
result = avcodec_receive_frame(decoderCtx, audioFrame);
if (result == AVERROR_EAGAIN() || result == AVERROR_EOF) {
break;
} else if (result < 0) {
throw new IOException(result + ": erro ao receber frame decodificado");
}
targetSamples = avutil.av_rescale_rnd(
swresample.swr_get_delay(resampleCtx, decoderCtx.sample_rate()) + audioFrame.nb_samples(),
decoderCtx.sample_rate(), decoderCtx.sample_rate(), AV_ROUND_UP);
if (targetSamples > maxTargetSamples) {
av_freep(outputBuffer.get());
result = av_samples_alloc(outputBuffer, outputLineSize, targetChannels,
(int) targetSamples, targetSampleFormat, 1);
if (result < 0) {
break;
}
maxTargetSamples = targetSamples;
}
result = swr_convert(resampleCtx, outputBuffer, (int) targetSamples,
audioFrame.data(), decoderCtx.frame_size());
if (result < 0) {
throw new IOException(result + ": erro durante resampling");
}
outputBufferSize = av_samples_get_buffer_size(outputLineSize, targetChannels,
result, targetSampleFormat, 1);
if (outputBufferSize < 0) {
throw new IOException(result + ": erro ao calcular tamanho do buffer");
}
audioData = new byte[outputBufferSize];
outputBuffer.get(BytePointer.class, 0).get(audioData);
fileStream.write(audioData);
System.out.printf("Amostras processadas: %d, Tamanho do buffer: %d bytes\n", result, outputBufferSize);
}
}
String formatSpec = "s16le";
System.out.printf(
"Resampling concluído. Reproduza o arquivo com:\n"
+ "ffplay -f %s -channel_layout %s -channels %d -ar %d %s\n",
formatSpec, AV_CH_LAYOUT_STEREO, targetChannels, targetSampleRate, destinationFile);
} finally {
outputBuffer.close();
outputLineSize.close();
if (dataPacket != null) {
av_packet_free(dataPacket);
}
if (audioFrame != null) {
av_frame_free(audioFrame);
}
if (decoderCtx != null) {
avcodec_free_context(decoderCtx);
}
if (resampleCtx != null) {
swr_free(resampleCtx);
}
avformat_close_input(formatCtx);
}
}
}
- Reprodução do Áudio Decodificado
O arquivo PCM gerado pode ser reproduzido diretamente com o FFplay usando o seguinte comando:
ffplay -f s16le -channel_layout 3 -channels 2 -ar 44100 output.pcm
Este comando configura o FFplay para interpretar os dados como áudio PCM de 16 bits litle-endian com dois canais estéreo a 44.1kHz.