FFmpeg
Loading...
Searching...
No Matches
enc.c
Go to the documentation of this file.
1/*
2 * Opus encoder
3 * Copyright (c) 2017 Rostislav Pehlivanov <atomnuker@gmail.com>
4 *
5 * This file is part of FFmpeg.
6 *
7 * FFmpeg is free software; you can redistribute it and/or
8 * modify it under the terms of the GNU Lesser General Public
9 * License as published by the Free Software Foundation; either
10 * version 2.1 of the License, or (at your option) any later version.
11 *
12 * FFmpeg is distributed in the hope that it will be useful,
13 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
15 * Lesser General Public License for more details.
16 *
17 * You should have received a copy of the GNU Lesser General Public
18 * License along with FFmpeg; if not, write to the Free Software
19 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
20 */
21
22#include <float.h>
23
24#include "enc.h"
25#include "pvq.h"
26#include "enc_psy.h"
27#include "tab.h"
28
30#include "libavutil/float_dsp.h"
31#include "libavutil/mem.h"
33#include "libavutil/opt.h"
34
38#include "libavcodec/encode.h"
39
67
69{
70 uint8_t *bs = avctx->extradata;
71
72 bytestream_put_buffer(&bs, "OpusHead", 8);
73 bytestream_put_byte (&bs, 0x1);
74 bytestream_put_byte (&bs, avctx->ch_layout.nb_channels);
75 bytestream_put_le16 (&bs, avctx->initial_padding);
76 bytestream_put_le32 (&bs, avctx->sample_rate);
77 bytestream_put_le16 (&bs, 0x0);
78 bytestream_put_byte (&bs, 0x0); /* Default layout */
79}
80
81static int opus_gen_toc(OpusEncContext *s, uint8_t *toc, int *size, int *fsize_needed)
82{
83 int tmp = 0x0, extended_toc = 0;
84 static const int toc_cfg[][OPUS_MODE_NB][OPUS_BANDWITH_NB] = {
85 /* Silk Hybrid Celt Layer */
86 /* NB MB WB SWB FB NB MB WB SWB FB NB MB WB SWB FB Bandwidth */
87 { { 0, 0, 0, 0, 0 }, { 0, 0, 0, 0, 0 }, { 17, 0, 21, 25, 29 } }, /* 2.5 ms */
88 { { 0, 0, 0, 0, 0 }, { 0, 0, 0, 0, 0 }, { 18, 0, 22, 26, 30 } }, /* 5 ms */
89 { { 1, 5, 9, 0, 0 }, { 0, 0, 0, 13, 15 }, { 19, 0, 23, 27, 31 } }, /* 10 ms */
90 { { 2, 6, 10, 0, 0 }, { 0, 0, 0, 14, 16 }, { 20, 0, 24, 28, 32 } }, /* 20 ms */
91 { { 3, 7, 11, 0, 0 }, { 0, 0, 0, 0, 0 }, { 0, 0, 0, 0, 0 } }, /* 40 ms */
92 { { 4, 8, 12, 0, 0 }, { 0, 0, 0, 0, 0 }, { 0, 0, 0, 0, 0 } }, /* 60 ms */
93 };
94 int cfg = toc_cfg[s->packet.framesize][s->packet.mode][s->packet.bandwidth];
95 *fsize_needed = 0;
96 if (!cfg)
97 return 1;
98 if (s->packet.frames == 2) { /* 2 packets */
99 if (s->frame[0].framebits == s->frame[1].framebits) { /* same size */
100 tmp = 0x1;
101 } else { /* different size */
102 tmp = 0x2;
103 *fsize_needed = 1; /* put frame sizes in the packet */
104 }
105 } else if (s->packet.frames > 2) {
106 tmp = 0x3;
107 extended_toc = 1;
108 }
109 tmp |= (s->channels > 1) << 2; /* Stereo or mono */
110 tmp |= (cfg - 1) << 3; /* codec configuration */
111 *toc++ = tmp;
112 if (extended_toc) {
113 for (int i = 0; i < (s->packet.frames - 1); i++)
114 *fsize_needed |= (s->frame[i].framebits != s->frame[i + 1].framebits);
115 tmp = (*fsize_needed) << 7; /* vbr flag */
116 tmp |= (0) << 6; /* padding flag */
117 tmp |= s->packet.frames;
118 *toc++ = tmp;
119 }
120 *size = 1 + extended_toc;
121 return 0;
122}
123
125{
126 AVFrame *cur = NULL;
127 const int subframesize = s->avctx->frame_size;
128 int subframes = OPUS_BLOCK_SIZE(s->packet.framesize) / subframesize;
129
130 cur = ff_bufqueue_get(&s->bufqueue);
131
132 for (int ch = 0; ch < f->channels; ch++) {
133 CeltBlock *b = &f->block[ch];
134 const float *input = (const float *)cur->extended_data[ch];
135 /* The MDCT overlap is the trailing CELT_OVERLAP samples of the
136 * previous packet's last frame, as they were when that frame was
137 * encoded. Because the encoder advertises AV_CODEC_CAP_SMALL_LAST_FRAME,
138 * that frame may have been shorter than frame_size, in which case it
139 * was zero padded, so the overlap has to be zero padded the same way. */
140 const int start = subframesize - CELT_OVERLAP;
141 const int n = av_clip(cur->nb_samples - start, 0, CELT_OVERLAP);
142 if (n > 0)
143 memcpy(b->overlap, input + start, n * sizeof(float));
144 memset(b->overlap + n, 0, (CELT_OVERLAP - n) * sizeof(float));
145 }
146
147 av_frame_free(&cur);
148
149 for (int sf = 0; sf < subframes; sf++) {
150 if (sf != (subframes - 1))
151 cur = ff_bufqueue_get(&s->bufqueue);
152 else
153 cur = ff_bufqueue_peek(&s->bufqueue, 0);
154
155 for (int ch = 0; ch < f->channels; ch++) {
156 CeltBlock *b = &f->block[ch];
157 const float *input = (const float *)cur->extended_data[ch];
158 float *dst = &b->samples[sf * subframesize];
159 const int n = FFMIN(cur->nb_samples, subframesize);
160 memcpy(dst, input, n * sizeof(float));
161 memset(dst + n, 0, (subframesize - n) * sizeof(float));
162 }
163
164 /* Last frame isn't popped off and freed yet - we need it for overlap */
165 if (sf != (subframes - 1))
166 av_frame_free(&cur);
167 }
168}
169
170/* Apply the pre emphasis filter */
172{
173 const int frame_len = OPUS_BLOCK_SIZE(s->packet.framesize);
174 const float c = ff_opus_deemph_weights[0];
175
176 for (int ch = 0; ch < f->channels; ch++) {
177 CeltBlock *b = &f->block[ch];
178 float m = b->emph_coeff;
179
180 /* Filter the overlap (the trailing CELT_OVERLAP samples of the previous frame) */
181 for (int i = 0; i < CELT_OVERLAP; i++) {
182 float sample = b->overlap[i];
183 b->overlap[i] = sample - m;
184 m = sample * c;
185 }
186
187 /* Filter the samples. The trailing CELT_OVERLAP samples are filtered
188 * again as the next frame's overlap, so the filter state saved for
189 * the next frame is the one from right before them. */
190 for (int i = 0; i < frame_len; i++) {
191 float sample = b->samples[i];
192 if (i == frame_len - CELT_OVERLAP)
193 b->emph_coeff = m;
194 b->samples[i] = sample - m;
195 m = sample * c;
196 }
197 }
198}
199
200/* Create the window and do the mdct */
202{
203 float *win = s->scratch, *temp = s->scratch + 1920;
204
205 if (f->transient) {
206 for (int ch = 0; ch < f->channels; ch++) {
207 CeltBlock *b = &f->block[ch];
208 float *src1 = b->overlap;
209 for (int t = 0; t < f->blocks; t++) {
210 float *src2 = &b->samples[CELT_OVERLAP*t];
211 s->dsp->vector_fmul(win, src1, ff_celt_window, 128);
212 s->dsp->vector_fmul_reverse(&win[CELT_OVERLAP], src2,
214 src1 = src2;
215 s->tx_fn[0](s->tx[0], b->coeffs + t, win, sizeof(float)*f->blocks);
216 }
217 }
218 } else {
219 int blk_len = OPUS_BLOCK_SIZE(f->size), wlen = OPUS_BLOCK_SIZE(f->size + 1);
220 int rwin = blk_len - CELT_OVERLAP, lap_dst = (wlen - blk_len - CELT_OVERLAP) >> 1;
221 memset(win, 0, wlen*sizeof(float));
222 for (int ch = 0; ch < f->channels; ch++) {
223 CeltBlock *b = &f->block[ch];
224
225 /* Overlap */
226 s->dsp->vector_fmul(temp, b->overlap, ff_celt_window, 128);
227 memcpy(win + lap_dst, temp, CELT_OVERLAP*sizeof(float));
228
229 /* Samples, flat top window */
230 memcpy(&win[lap_dst + CELT_OVERLAP], b->samples, rwin*sizeof(float));
231
232 /* Samples, windowed */
233 s->dsp->vector_fmul_reverse(temp, b->samples + rwin,
235 memcpy(win + lap_dst + blk_len, temp, CELT_OVERLAP*sizeof(float));
236
237 s->tx_fn[f->size](s->tx[f->size], b->coeffs, win, sizeof(float));
238 }
239 }
240
241 for (int ch = 0; ch < f->channels; ch++) {
242 CeltBlock *block = &f->block[ch];
243 for (int i = 0; i < CELT_MAX_BANDS; i++) {
244 float ener = 0.0f;
245 int band_offset = ff_celt_freq_bands[i] << f->size;
246 int band_size = ff_celt_freq_range[i] << f->size;
247 float *coeffs = &block->coeffs[band_offset];
248
249 for (int j = 0; j < band_size; j++)
250 ener += coeffs[j]*coeffs[j];
251
252 block->lin_energy[i] = sqrtf(ener) + FLT_EPSILON;
253 ener = 1.0f/block->lin_energy[i];
254
255 for (int j = 0; j < band_size; j++)
256 coeffs[j] *= ener;
257
258 block->energy[i] = log2f(block->lin_energy[i]) - ff_celt_mean_energy[i];
259
260 /* CELT_ENERGY_SILENCE is what the decoder uses and its not -infinity */
261 block->energy[i] = FFMAX(block->energy[i], CELT_ENERGY_SILENCE);
262 }
263 }
264}
265
267{
268 int tf_select = 0, diff = 0, tf_changed = 0, tf_select_needed;
269 int bits = f->transient ? 2 : 4;
270
271 tf_select_needed = ((f->size && (opus_rc_tell(rc) + bits + 1) <= f->framebits));
272
273 for (int i = f->start_band; i < f->end_band; i++) {
274 if ((opus_rc_tell(rc) + bits + tf_select_needed) <= f->framebits) {
275 const int tbit = (diff ^ 1) == f->tf_change[i];
276 ff_opus_rc_enc_log(rc, tbit, bits);
277 diff ^= tbit;
278 tf_changed |= diff;
279 }
280 bits = f->transient ? 4 : 5;
281 }
282
283 if (tf_select_needed && ff_celt_tf_select[f->size][f->transient][0][tf_changed] !=
284 ff_celt_tf_select[f->size][f->transient][1][tf_changed]) {
285 ff_opus_rc_enc_log(rc, f->tf_select, 1);
286 tf_select = f->tf_select;
287 }
288
289 for (int i = f->start_band; i < f->end_band; i++)
290 f->tf_change[i] = ff_celt_tf_select[f->size][f->transient][tf_select][f->tf_change[i]];
291}
292
294{
295 float gain = f->pf_gain;
296 int txval, octave = f->pf_octave, period = f->pf_period, tapset = f->pf_tapset;
297
298 ff_opus_rc_enc_log(rc, f->pfilter, 1);
299 if (!f->pfilter)
300 return;
301
302 /* Octave */
303 txval = FFMIN(octave, 6);
304 ff_opus_rc_enc_uint(rc, txval, 6);
305 octave = txval;
306 /* Period */
307 txval = av_clip(period - (16 << octave) + 1, 0, (1 << (4 + octave)) - 1);
308 ff_opus_rc_put_raw(rc, period, 4 + octave);
309 period = txval + (16 << octave) - 1;
310 /* Gain */
311 txval = FFMIN(((int)(gain / 0.09375f)) - 1, 7);
312 ff_opus_rc_put_raw(rc, txval, 3);
313 gain = 0.09375f * (txval + 1);
314 /* Tapset */
315 if ((opus_rc_tell(rc) + 2) <= f->framebits)
317 else
318 tapset = 0;
319 /* Finally create the coeffs */
320 for (int i = 0; i < 2; i++) {
321 CeltBlock *block = &f->block[i];
322
323 block->pf_period_new = FFMAX(period, CELT_POSTFILTER_MINPERIOD);
324 block->pf_gains_new[0] = gain * ff_celt_postfilter_taps[tapset][0];
325 block->pf_gains_new[1] = gain * ff_celt_postfilter_taps[tapset][1];
326 block->pf_gains_new[2] = gain * ff_celt_postfilter_taps[tapset][2];
327 }
328}
329
331 float last_energy[][CELT_MAX_BANDS], int intra)
332{
333 float alpha, beta, prev[2] = { 0, 0 };
334 const uint8_t *pmod = ff_celt_coarse_energy_dist[f->size][intra];
335
336 /* Inter is really just differential coding */
337 if (opus_rc_tell(rc) + 3 <= f->framebits)
338 ff_opus_rc_enc_log(rc, intra, 3);
339 else
340 intra = 0;
341
342 if (intra) {
343 alpha = 0.0f;
344 beta = 1.0f - (4915.0f/32768.0f);
345 } else {
346 alpha = ff_celt_alpha_coef[f->size];
347 beta = ff_celt_beta_coef[f->size];
348 }
349
350 for (int i = f->start_band; i < f->end_band; i++) {
351 for (int ch = 0; ch < f->channels; ch++) {
352 CeltBlock *block = &f->block[ch];
353 const int left = f->framebits - opus_rc_tell(rc);
354 const float last = FFMAX(-9.0f, last_energy[ch][i]);
355 float diff = block->energy[i] - prev[ch] - last*alpha;
356 int q_en = lrintf(diff);
357 if (left >= 15) {
358 ff_opus_rc_enc_laplace(rc, &q_en, pmod[i << 1] << 7, pmod[(i << 1) + 1] << 6);
359 } else if (left >= 2) {
360 q_en = av_clip(q_en, -1, 1);
361 ff_opus_rc_enc_cdf(rc, 2*q_en + 3*(q_en < 0), ff_celt_model_energy_small);
362 } else if (left >= 1) {
363 q_en = av_clip(q_en, -1, 0);
364 ff_opus_rc_enc_log(rc, (q_en & 1), 1);
365 } else q_en = -1;
366
367 block->error_energy[i] = q_en - diff;
368 prev[ch] += beta * q_en;
369 }
370 }
371}
372
374 float last_energy[][CELT_MAX_BANDS])
375{
376 uint32_t inter, intra;
378
379 exp_quant_coarse(rc, f, last_energy, 1);
380 intra = OPUS_RC_CHECKPOINT_BITS(rc);
381
383
384 exp_quant_coarse(rc, f, last_energy, 0);
385 inter = OPUS_RC_CHECKPOINT_BITS(rc);
386
387 if (inter > intra) { /* Unlikely */
389 exp_quant_coarse(rc, f, last_energy, 1);
390 }
391}
392
394{
395 for (int i = f->start_band; i < f->end_band; i++) {
396 if (!f->fine_bits[i])
397 continue;
398 for (int ch = 0; ch < f->channels; ch++) {
399 CeltBlock *block = &f->block[ch];
400 int quant, lim = (1 << f->fine_bits[i]);
401 float offset, diff = 0.5f - block->error_energy[i];
402 quant = av_clip(floor(diff*lim), 0, lim - 1);
403 ff_opus_rc_put_raw(rc, quant, f->fine_bits[i]);
404 offset = 0.5f - ((quant + 0.5f) * (1 << (14 - f->fine_bits[i])) / 16384.0f);
405 block->error_energy[i] -= offset;
406 }
407 }
408}
409
411{
412 for (int priority = 0; priority < 2; priority++) {
413 for (int i = f->start_band; i < f->end_band && (f->framebits - opus_rc_tell(rc)) >= f->channels; i++) {
414 if (f->fine_priority[i] != priority || f->fine_bits[i] >= CELT_MAX_FINE_BITS)
415 continue;
416 for (int ch = 0; ch < f->channels; ch++) {
417 CeltBlock *block = &f->block[ch];
418 const float err = block->error_energy[i];
419 const float offset = 0.5f * (1 << (14 - f->fine_bits[i] - 1)) / 16384.0f;
420 const int sign = FFABS(err + offset) < FFABS(err - offset);
421 ff_opus_rc_put_raw(rc, sign, 1);
422 block->error_energy[i] -= offset*(1 - 2*sign);
423 }
424 }
425 }
426}
427
429 CeltFrame *f, int index)
430{
432
434
436
437 if (f->silence) {
438 if (f->framebits >= 16)
439 ff_opus_rc_enc_log(rc, 1, 15); /* Silence (if using explicit signalling) */
440 for (int ch = 0; ch < s->channels; ch++) {
441 /* The decoder sets all band energies to CELT_ENERGY_SILENCE on
442 * a silence frame, and predicts the next frame's from them. */
443 for (int i = 0; i < CELT_MAX_BANDS; i++)
444 s->last_quantized_energy[ch][i] = CELT_ENERGY_SILENCE;
445 /* The frame is all zeros, so this is the pre emphasis filter state
446 * at the point the next frame's overlap starts */
447 f->block[ch].emph_coeff = 0.0f;
448 }
449 return;
450 }
451
452 /* Filters */
454 if (f->pfilter) {
455 ff_opus_rc_enc_log(rc, 0, 15);
457 }
458
459 /* Transform */
461
462 /* Need to handle transient/non-transient switches at any point during analysis */
463 while (ff_opus_psy_celt_frame_process(&s->psyctx, f, index))
465
467
468 /* Silence */
469 ff_opus_rc_enc_log(rc, 0, 15);
470
471 /* Pitch filter */
472 if (!f->start_band && opus_rc_tell(rc) + 16 <= f->framebits)
474
475 /* Transient flag */
476 if (f->size && opus_rc_tell(rc) + 3 <= f->framebits)
477 ff_opus_rc_enc_log(rc, f->transient, 3);
478
479 /* Main encoding */
480 celt_quant_coarse (f, rc, s->last_quantized_energy);
481 celt_enc_tf (f, rc);
482 ff_celt_bitalloc (f, rc, 1);
483 celt_quant_fine (f, rc);
485
486 /* Anticollapse bit */
487 if (f->anticollapse_needed)
488 ff_opus_rc_put_raw(rc, f->anticollapse, 1);
489
490 /* Final per-band energy adjustments from leftover bits */
491 celt_quant_final(s, rc, f);
492
493 for (int ch = 0; ch < f->channels; ch++) {
494 CeltBlock *block = &f->block[ch];
495 for (int i = 0; i < CELT_MAX_BANDS; i++)
496 s->last_quantized_energy[ch][i] = block->energy[i] + block->error_energy[i];
497 }
498}
499
500static inline int write_opuslacing(uint8_t *dst, int v)
501{
502 dst[0] = FFMIN(v - FFALIGN(v - 255, 4), v);
503 dst[1] = v - dst[0] >> 2;
504 return 1 + (v >= 252);
505}
506
508{
509 int offset, fsize_needed;
510
511 /* Write toc */
512 opus_gen_toc(s, avpkt->data, &offset, &fsize_needed);
513
514 /* Frame sizes if needed */
515 if (fsize_needed) {
516 for (int i = 0; i < s->packet.frames - 1; i++) {
518 s->frame[i].framebits >> 3);
519 }
520 }
521
522 /* Packets */
523 for (int i = 0; i < s->packet.frames; i++) {
524 ff_opus_rc_enc_end(&s->rc[i], avpkt->data + offset,
525 s->frame[i].framebits >> 3);
526 offset += s->frame[i].framebits >> 3;
527 }
528
529 avpkt->size = offset;
530}
531
532/* Used as overlap for the first frame and padding for the last encoded packet */
534{
536 int ret;
537 if (!f)
538 return NULL;
539 f->format = s->avctx->sample_fmt;
540 f->nb_samples = s->avctx->frame_size;
541 ret = av_channel_layout_copy(&f->ch_layout, &s->avctx->ch_layout);
542 if (ret < 0) {
544 return NULL;
545 }
546 if (av_frame_get_buffer(f, 4)) {
548 return NULL;
549 }
550 for (int i = 0; i < s->channels; i++) {
551 size_t bps = av_get_bytes_per_sample(f->format);
552 memset(f->extended_data[i], 0, bps*f->nb_samples);
553 }
554 return f;
555}
556
557static int opus_encode_frame(AVCodecContext *avctx, AVPacket *avpkt,
558 const AVFrame *frame, int *got_packet_ptr)
559{
560 OpusEncContext *s = avctx->priv_data;
561 int ret, frame_size, discard_padding, alloc_size = 0;
562
563 if (frame) { /* Add new frame to queue */
564 if ((ret = ff_af_queue_add(&s->afq, frame)) < 0)
565 return ret;
566 ff_bufqueue_add(avctx, &s->bufqueue, av_frame_clone(frame));
567 } else {
568 ff_opus_psy_signal_eof(&s->psyctx);
569 if (!s->afq.remaining_samples || !avctx->frame_num)
570 return 0; /* We've been flushed and there's nothing left to encode */
571 }
572
573 /* Run the psychoacoustic system */
574 if (ff_opus_psy_process(&s->psyctx, &s->packet))
575 return 0;
576
577 frame_size = OPUS_BLOCK_SIZE(s->packet.framesize);
578
579 if (!frame) {
580 /* This can go negative, that's not a problem, we only pad if positive */
581 int pad_empty = s->packet.frames*(frame_size/s->avctx->frame_size) - s->bufqueue.available + 1;
582 /* Pad with empty 2.5 ms frames to whatever framesize was decided,
583 * this should only happen at the very last flush frame. The frames
584 * allocated here will be freed (because they have no other references)
585 * after they get used by celt_frame_setup_input() */
586 for (int i = 0; i < pad_empty; i++) {
587 AVFrame *empty = spawn_empty_frame(s);
588 if (!empty)
589 return AVERROR(ENOMEM);
590 ff_bufqueue_add(avctx, &s->bufqueue, empty);
591 }
592 }
593
594 for (int i = 0; i < s->packet.frames; i++) {
595 celt_encode_frame(s, &s->rc[i], &s->frame[i], i);
596 alloc_size += s->frame[i].framebits >> 3;
597 }
598
599 /* Worst case toc + the frame lengths if needed */
600 alloc_size += 2 + s->packet.frames*2;
601
602 if ((ret = ff_alloc_packet(avctx, avpkt, alloc_size)) < 0)
603 return ret;
604
605 /* Assemble packet */
606 opus_packet_assembler(s, avpkt);
607
608 /* Update the psychoacoustic system */
609 ff_opus_psy_postencode_update(&s->psyctx, s->frame);
610
611 /* Remove samples from queue and skip if needed */
612 ret = ff_af_queue_remove(&s->afq, s->packet.frames*frame_size, avpkt);
613 if (ret < 0)
614 return ret;
615
616 discard_padding = s->packet.frames*frame_size - ff_samples_from_time_base(avctx, avpkt->duration);
617 if (discard_padding > 0) {
618 uint8_t *side = av_packet_new_side_data(avpkt, AV_PKT_DATA_SKIP_SAMPLES, 10);
619 if (!side)
620 return AVERROR(ENOMEM);
621 AV_WL32(&side[4], discard_padding);
622 }
623
624 *got_packet_ptr = 1;
625
626 return 0;
627}
628
630{
631 OpusEncContext *s = avctx->priv_data;
632
633 for (int i = 0; i < CELT_BLOCK_NB; i++)
634 av_tx_uninit(&s->tx[i]);
635
636 ff_celt_pvq_uninit(&s->pvq);
637 av_freep(&s->dsp);
638 av_freep(&s->frame);
639 av_freep(&s->rc);
640 ff_af_queue_close(&s->afq);
641 ff_opus_psy_end(&s->psyctx);
642 ff_bufqueue_discard_all(&s->bufqueue);
643
644 return 0;
645}
646
648{
649 int ret, max_frames;
650 OpusEncContext *s = avctx->priv_data;
651
652 s->avctx = avctx;
653 s->channels = avctx->ch_layout.nb_channels;
654
655 int max_delay_samples = (s->options.max_delay_ms * s->avctx->sample_rate) / 1000;
657 /* Initial padding will change if SILK is ever supported */
658 avctx->initial_padding = 120;
659
660 if (!avctx->bit_rate) {
661 int coupled = ff_opus_default_coupled_streams[s->channels - 1];
662 avctx->bit_rate = coupled*(96000) + (s->channels - coupled*2)*(48000);
663 } else if (avctx->bit_rate < 6000 || avctx->bit_rate > 255000 * s->channels) {
664 int64_t clipped_rate = av_clip(avctx->bit_rate, 6000, 255000 * s->channels);
665 av_log(avctx, AV_LOG_ERROR, "Unsupported bitrate %"PRId64" kbps, clipping to %"PRId64" kbps\n",
666 avctx->bit_rate/1000, clipped_rate/1000);
667 avctx->bit_rate = clipped_rate;
668 }
669
670 /* Extradata */
671 avctx->extradata_size = 19;
673 if (!avctx->extradata)
674 return AVERROR(ENOMEM);
676
677 ff_af_queue_init(avctx, &s->afq);
678
679 if ((ret = ff_celt_pvq_init(&s->pvq, 1)) < 0)
680 return ret;
681
683 return AVERROR(ENOMEM);
684
685 /* I have no idea why a base scaling factor of 68 works, could be the twiddles */
686 for (int i = 0; i < CELT_BLOCK_NB; i++) {
687 const float scale = 68 << (CELT_BLOCK_NB - 1 - i);
688 if ((ret = av_tx_init(&s->tx[i], &s->tx_fn[i], AV_TX_FLOAT_MDCT, 0, 15 << (i + 3), &scale, 0)))
689 return AVERROR(ENOMEM);
690 }
691
692 /* Zero out previous energy (matters for inter first frame) */
693 for (int ch = 0; ch < s->channels; ch++)
694 memset(s->last_quantized_energy[ch], 0.0f, sizeof(float)*CELT_MAX_BANDS);
695
696 /* Allocate an empty frame to use as overlap for the first frame of audio */
697 ff_bufqueue_add(avctx, &s->bufqueue, spawn_empty_frame(s));
698 if (!ff_bufqueue_peek(&s->bufqueue, 0))
699 return AVERROR(ENOMEM);
700
701 if ((ret = ff_opus_psy_init(&s->psyctx, s->avctx, &s->bufqueue, &s->options)))
702 return ret;
703
704 /* Frame structs and range coder buffers */
705 max_frames = ceilf(FFMIN(s->options.max_delay_ms, 120.0f)/2.5f);
706 s->frame = av_malloc(max_frames*sizeof(CeltFrame));
707 if (!s->frame)
708 return AVERROR(ENOMEM);
709 s->rc = av_malloc(max_frames*sizeof(OpusRangeCoder));
710 if (!s->rc)
711 return AVERROR(ENOMEM);
712
713 for (int i = 0; i < max_frames; i++) {
714 s->frame[i].dsp = s->dsp;
715 s->frame[i].avctx = s->avctx;
716 s->frame[i].seed = 0;
717 s->frame[i].pvq = s->pvq;
718 s->frame[i].apply_phase_inv = s->options.apply_phase_inv;
719 s->frame[i].block[0].emph_coeff = s->frame[i].block[1].emph_coeff = 0.0f;
720 }
721
722 return 0;
723}
724
725#define OPUSENC_FLAGS AV_OPT_FLAG_ENCODING_PARAM | AV_OPT_FLAG_AUDIO_PARAM
726static const AVOption opusenc_options[] = {
727 { "opus_delay", "Maximum delay in milliseconds", offsetof(OpusEncContext, options.max_delay_ms), AV_OPT_TYPE_FLOAT, { .dbl = OPUS_MAX_LOOKAHEAD }, 2.5f, OPUS_MAX_LOOKAHEAD, OPUSENC_FLAGS, .unit = "max_delay_ms" },
728 { "apply_phase_inv", "Apply intensity stereo phase inversion", offsetof(OpusEncContext, options.apply_phase_inv), AV_OPT_TYPE_BOOL, { .i64 = 1 }, 0, 1, OPUSENC_FLAGS, .unit = "apply_phase_inv" },
729 { NULL },
730};
731
732static const AVClass opusenc_class = {
733 .class_name = "Opus encoder",
734 .item_name = av_default_item_name,
735 .option = opusenc_options,
736 .version = LIBAVUTIL_VERSION_INT,
737};
738
740 { "b", "0" },
741 { "compression_level", "10" },
742 { NULL },
743};
744
746 .p.name = "opus",
747 CODEC_LONG_NAME("Opus"),
748 .p.type = AVMEDIA_TYPE_AUDIO,
749 .p.id = AV_CODEC_ID_OPUS,
750 .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_DELAY |
752 .defaults = opusenc_defaults,
753 .p.priv_class = &opusenc_class,
754 .priv_data_size = sizeof(OpusEncContext),
757 .close = opus_encode_end,
758 .caps_internal = FF_CODEC_CAP_INIT_CLEANUP,
759 CODEC_SAMPLERATES(48000),
762};
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t intptr_t int int16_t * dst
Definition dsp.h:97
static float win(SuperEqualizerContext *s, float n, int N)
const FFCodec ff_opus_encoder
Definition enc.c:745
#define log2f(x)
Definition math.h:27
av_cold void ff_af_queue_close(AudioFrameQueue *afq)
Close AudioFrameQueue.
av_cold void ff_af_queue_init(AVCodecContext *avctx, AudioFrameQueue *afq)
Initialize AudioFrameQueue.
int ff_af_queue_remove(AudioFrameQueue *afq, int nb_samples, AVPacket *pkt)
Remove frame(s) from the queue.
int ff_af_queue_add(AudioFrameQueue *afq, const AVFrame *f)
Add a frame to the queue.
static int BS_FUNC left(const BSCTX *bc)
Return the number of the bits left in a buffer.
static void ff_bufqueue_add(void *log, struct FFBufQueue *queue, AVFrame *buf)
Add a buffer to the queue.
Definition bufferqueue.h:71
static void ff_bufqueue_discard_all(struct FFBufQueue *queue)
Unref and remove all buffers from the queue.
static AVFrame * ff_bufqueue_peek(struct FFBufQueue *queue, unsigned index)
Get a buffer from the queue without altering it.
Definition bufferqueue.h:87
static AVFrame * ff_bufqueue_get(struct FFBufQueue *queue)
Get the first buffer from the queue and remove it.
Definition bufferqueue.h:98
static av_always_inline void bytestream_put_buffer(uint8_t **b, const uint8_t *src, unsigned int size)
Definition bytestream.h:372
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define f(width, name)
Definition cbs_vp8.c:236
#define s(width, name)
Definition cbs_vp9.c:198
void ff_celt_quant_bands(CeltFrame *f, OpusRangeCoder *rc)
Definition celt.c:28
void ff_celt_bitalloc(CeltFrame *f, OpusRangeCoder *rc, int encode)
Definition celt.c:137
#define CELT_OVERLAP
Definition celt.h:40
#define CELT_POSTFILTER_MINPERIOD
Definition celt.h:52
#define CELT_ENERGY_SILENCE
Definition celt.h:53
#define CELT_MAX_BANDS
Definition celt.h:43
#define CELT_MAX_FINE_BITS
Definition celt.h:48
@ CELT_BLOCK_NB
Definition celt.h:68
@ CELT_BLOCK_960
Definition celt.h:66
Public libavutil channel layout APIs header.
#define CODEC_CH_LAYOUTS(...)
#define CODEC_SAMPLERATES(...)
#define FF_CODEC_ENCODE_CB(func)
#define CODEC_LONG_NAME(str)
#define FF_CODEC_CAP_INIT_CLEANUP
The codec allows calling the close function for deallocation even if the init function returned a fai...
#define CODEC_SAMPLEFMTS(...)
#define av_clip
Definition common.h:100
#define FFABS(a)
Absolute value, Note, INT_MIN / INT64_MIN result in undefined behavior as they are not representable ...
Definition common.h:74
#define NULL
Definition coverity.c:32
long long int64_t
Definition coverity.c:34
static __device__ float sqrtf(float a)
static __device__ float floor(float a)
static __device__ float ceilf(float a)
static int16_t block[64]
Definition dct.c:125
static AVFrame * frame
int(* init)(AVBSFContext *ctx)
Definition dts2pts.c:608
static void celt_quant_coarse(CeltFrame *f, OpusRangeCoder *rc, float last_energy[][CELT_MAX_BANDS])
Definition enc.c:373
static const AVClass opusenc_class
Definition enc.c:732
static void celt_frame_setup_input(OpusEncContext *s, CeltFrame *f)
Definition enc.c:124
static const AVOption opusenc_options[]
Definition enc.c:726
static av_cold int opus_encode_init(AVCodecContext *avctx)
Definition enc.c:647
static int opus_gen_toc(OpusEncContext *s, uint8_t *toc, int *size, int *fsize_needed)
Definition enc.c:81
static int opus_encode_frame(AVCodecContext *avctx, AVPacket *avpkt, const AVFrame *frame, int *got_packet_ptr)
Definition enc.c:557
static void celt_enc_tf(CeltFrame *f, OpusRangeCoder *rc)
Definition enc.c:266
static void celt_frame_mdct(OpusEncContext *s, CeltFrame *f)
Definition enc.c:201
static void opus_packet_assembler(OpusEncContext *s, AVPacket *avpkt)
Definition enc.c:507
static void exp_quant_coarse(OpusRangeCoder *rc, CeltFrame *f, float last_energy[][CELT_MAX_BANDS], int intra)
Definition enc.c:330
static void celt_apply_preemph_filter(OpusEncContext *s, CeltFrame *f)
Definition enc.c:171
static void celt_encode_frame(OpusEncContext *s, OpusRangeCoder *rc, CeltFrame *f, int index)
Definition enc.c:428
static void celt_quant_final(OpusEncContext *s, OpusRangeCoder *rc, CeltFrame *f)
Definition enc.c:410
#define OPUSENC_FLAGS
Definition enc.c:725
static void opus_write_extradata(AVCodecContext *avctx)
Definition enc.c:68
static void celt_enc_quant_pfilter(OpusRangeCoder *rc, CeltFrame *f)
Definition enc.c:293
static av_cold int opus_encode_end(AVCodecContext *avctx)
Definition enc.c:629
static AVFrame * spawn_empty_frame(OpusEncContext *s)
Definition enc.c:533
static int write_opuslacing(uint8_t *dst, int v)
Definition enc.c:500
static void celt_quant_fine(CeltFrame *f, OpusRangeCoder *rc)
Definition enc.c:393
static const FFCodecDefault opusenc_defaults[]
Definition enc.c:739
#define OPUS_SAMPLES_TO_BLOCK_SIZE(x)
Definition enc.h:41
#define OPUS_MAX_CHANNELS
Definition enc.h:34
#define OPUS_BLOCK_SIZE(x)
Definition enc.h:39
#define OPUS_MAX_LOOKAHEAD
Definition enc.h:32
void ff_opus_psy_postencode_update(OpusPsyContext *s, CeltFrame *f)
Definition enc_psy.c:518
av_cold int ff_opus_psy_end(OpusPsyContext *s)
Definition enc_psy.c:638
int ff_opus_psy_celt_frame_process(OpusPsyContext *s, CeltFrame *f, int index)
Definition enc_psy.c:496
int ff_opus_psy_process(OpusPsyContext *s, OpusPacketInfo *p)
Definition enc_psy.c:254
av_cold int ff_opus_psy_init(OpusPsyContext *s, AVCodecContext *avctx, struct FFBufQueue *bufqueue, OpusEncOptions *options)
Definition enc_psy.c:557
void ff_opus_psy_signal_eof(OpusPsyContext *s)
Definition enc_psy.c:633
void ff_opus_psy_celt_frame_init(OpusPsyContext *s, CeltFrame *f, int index)
Definition enc_psy.c:288
int ff_alloc_packet(AVCodecContext *avctx, AVPacket *avpkt, int64_t size)
Check AVPacket size and allocate data.
Definition encode.c:62
static av_always_inline int64_t ff_samples_from_time_base(const AVCodecContext *avctx, int64_t duration)
Rescale from time base to AVCodecContext.sample_rate.
Definition encode.h:108
static CheckasmConfig cfg
Definition checkasm.c:74
static const uint8_t bits[8]
Definition fastaudio.c:100
#define sample
static const uint8_t frame_size[4]
Definition g723_1.h:222
@ AV_OPT_TYPE_FLOAT
Underlying C type is float.
Definition opt.h:270
@ AV_OPT_TYPE_BOOL
Underlying C type is int.
Definition opt.h:326
#define AV_CODEC_FLAG_BITEXACT
Use only bitexact stuff (except (I)DCT).
Definition avcodec.h:322
#define AV_CODEC_CAP_DELAY
Encoder or decoder requires flushing with NULL input at the end in order to give the complete and cor...
Definition codec.h:79
#define AV_CODEC_CAP_DR1
Codec uses get_buffer() or get_encode_buffer() for allocating buffers and supports custom allocators.
Definition codec.h:49
#define AV_CODEC_CAP_SMALL_LAST_FRAME
Codec can be fed a final frame with a smaller size.
Definition codec.h:84
#define AV_CODEC_CAP_EXPERIMENTAL
Codec is experimental and is thus avoided in favor of non experimental encoders.
Definition codec.h:90
@ AV_CODEC_ID_OPUS
Definition codec_id.h:516
#define AV_INPUT_BUFFER_PADDING_SIZE
Required number of additionally allocated bytes at the end of the input bitstream for decoding.
Definition defs.h:40
@ AV_PKT_DATA_SKIP_SAMPLES
Recommends skipping the specified number of samples.
Definition packet.h:153
uint8_t * av_packet_new_side_data(AVPacket *pkt, enum AVPacketSideDataType type, size_t size)
Allocate new information of a packet.
Definition packet.c:231
#define AV_CHANNEL_LAYOUT_STEREO
#define AV_CHANNEL_LAYOUT_MONO
int av_channel_layout_copy(AVChannelLayout *dst, const AVChannelLayout *src)
Make a copy of a channel layout.
#define AVERROR(e)
Definition error.h:45
int av_frame_get_buffer(AVFrame *frame, int align)
Allocate new buffer(s) for audio or video data.
Definition frame.c:206
void av_frame_free(AVFrame **frame)
Free the frame and any dynamically allocated objects in it, e.g.
Definition frame.c:64
AVFrame * av_frame_alloc(void)
Allocate an AVFrame and set its fields to default values.
Definition frame.c:52
AVFrame * av_frame_clone(const AVFrame *src)
Create a new frame that references the same data as src.
Definition frame.c:483
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition log.h:210
const char * av_default_item_name(void *ptr)
Return the context name.
Definition log.c:241
@ AVMEDIA_TYPE_AUDIO
Definition avutil.h:201
int av_get_bytes_per_sample(enum AVSampleFormat sample_fmt)
Return number of bytes per sample.
Definition samplefmt.c:109
@ AV_SAMPLE_FMT_FLTP
float, planar
Definition samplefmt.h:66
#define LIBAVUTIL_VERSION_INT
Definition version.h:85
int index
Definition gxfenc.c:90
const pixel * src2
static const int16_t alpha[]
Definition ilbcdata.h:55
#define b
Definition input.c:43
static void scale(int *out, const int *in, const int w, const int h, const int shift)
Definition intra.c:278
#define AV_WL32(p, v)
unsigned offset
Definition libaomenc.c:763
#define av_cold
Definition attributes.h:117
av_cold AVFloatDSPContext * avpriv_float_dsp_alloc(int bit_exact)
Allocate a float DSP context.
Definition float_dsp.c:135
#define lrintf(x)
Definition libm_mips.h:74
#define FFMIN(a, b)
Definition macros.h:49
#define FFMAX(a, b)
Definition macros.h:47
#define FFALIGN(x, a)
Definition macros.h:78
Memory handling functions.
#define DECLARE_ALIGNED(n, t, v)
Declare a variable that is aligned in memory.
unsigned bps
Definition movenc.c:2129
#define av_malloc(s)
Definition ops_static.c:52
AVOptions.
@ OPUS_BANDWITH_NB
Definition opus.h:56
@ OPUS_MODE_NB
Definition opus.h:46
int av_cold ff_celt_pvq_init(CeltPVQ **pvq, int encode)
Definition pvq.c:907
void av_cold ff_celt_pvq_uninit(CeltPVQ **pvq)
Definition pvq.c:927
void ff_opus_rc_enc_end(OpusRangeCoder *rc, uint8_t *dst, int size)
Definition rc.c:360
void ff_opus_rc_enc_uint(OpusRangeCoder *rc, uint32_t val, uint32_t size)
CELT: write a uniformly distributed integer.
Definition rc.c:204
void ff_opus_rc_put_raw(OpusRangeCoder *rc, uint32_t val, uint32_t count)
CELT: write 0 - 31 bits to the rawbits buffer.
Definition rc.c:161
void ff_opus_rc_enc_cdf(OpusRangeCoder *rc, int val, const uint16_t *cdf)
Definition rc.c:109
void ff_opus_rc_enc_log(OpusRangeCoder *rc, int val, uint32_t bits)
Definition rc.c:131
void ff_opus_rc_enc_laplace(OpusRangeCoder *rc, int *value, uint32_t symbol, int decay)
Definition rc.c:314
void ff_opus_rc_enc_init(OpusRangeCoder *rc)
Definition rc.c:410
#define OPUS_RC_CHECKPOINT_ROLLBACK(rc)
Definition rc.h:124
static av_always_inline uint32_t opus_rc_tell(const OpusRangeCoder *rc)
CELT: estimate bits of entropy that have thus far been consumed for the current CELT frame,...
Definition rc.h:62
#define OPUS_RC_CHECKPOINT_SPAWN(rc)
Definition rc.h:117
#define OPUS_RC_CHECKPOINT_BITS(rc)
Definition rc.h:121
int nb_channels
Number of channels in this layout.
Describe the class of an AVClass context structure.
Definition log.h:76
main external API structure.
Definition avcodec.h:443
AVChannelLayout ch_layout
Audio channel layout.
Definition avcodec.h:1055
int64_t frame_num
Frame counter, set by libavcodec.
Definition avcodec.h:1888
int64_t bit_rate
the average bitrate
Definition avcodec.h:493
int initial_padding
Audio only.
Definition avcodec.h:1114
int sample_rate
samples per second
Definition avcodec.h:1040
int flags
AV_CODEC_FLAG_*.
Definition avcodec.h:500
uint8_t * extradata
Out-of-band global headers that may be used by some codecs.
Definition avcodec.h:526
int extradata_size
Definition avcodec.h:527
int frame_size
Number of samples per channel in an audio frame.
Definition avcodec.h:1068
void * priv_data
Definition avcodec.h:470
This structure describes decoded (raw) audio or video data.
Definition frame.h:479
int nb_samples
number of audio samples (per channel) described by this frame
Definition frame.h:559
uint8_t ** extended_data
pointers to the data planes/channels.
Definition frame.h:540
AVOption.
Definition opt.h:428
This structure stores compressed data.
Definition packet.h:580
int size
Definition packet.h:604
int64_t duration
Duration of this packet in AVStream->time_base units, 0 if unknown.
Definition packet.h:621
uint8_t * data
Definition packet.h:603
Definition pvq.h:37
Structure holding the queue.
Definition bufferqueue.h:49
struct FFBufQueue bufqueue
Definition enc.c:50
av_tx_fn tx_fn[CELT_BLOCK_NB]
Definition enc.c:48
float last_quantized_energy[OPUS_MAX_CHANNELS][CELT_MAX_BANDS]
Definition enc.c:63
float scratch[2048]
Definition enc.c:65
int channels
Definition enc.c:57
AVTXContext * tx[CELT_BLOCK_NB]
Definition enc.c:47
OpusPsyContext psyctx
Definition enc.c:43
int enc_id_bits
Definition enc.c:53
OpusRangeCoder * rc
Definition enc.c:60
AVClass * av_class
Definition enc.c:41
CeltFrame * frame
Definition enc.c:59
OpusEncOptions options
Definition enc.c:42
AVCodecContext * avctx
Definition enc.c:44
AVFloatDSPContext * dsp
Definition enc.c:46
uint8_t enc_id[64]
Definition enc.c:52
CeltPVQ * pvq
Definition enc.c:49
OpusPacketInfo packet
Definition enc.c:55
AudioFrameQueue afq
Definition enc.c:45
const uint8_t ff_celt_freq_range[]
Definition tab.c:836
const uint8_t ff_celt_freq_bands[]
Definition tab.c:832
const float ff_celt_window_padded[136]
Definition tab.c:1168
const uint16_t ff_celt_model_tapset[]
Definition tab.c:824
const uint8_t ff_opus_default_coupled_streams[]
Definition tab.c:27
const int8_t ff_celt_tf_select[4][2][2][2]
Definition tab.c:846
const uint8_t ff_celt_coarse_energy_dist[4][2][42]
Definition tab.c:872
const float ff_celt_alpha_coef[]
Definition tab.c:864
const float ff_opus_deemph_weights[]
Definition tab.c:1233
const float ff_celt_beta_coef[]
Definition tab.c:868
const float ff_celt_mean_energy[]
Definition tab.c:856
const float ff_celt_postfilter_taps[3][3]
Definition tab.c:1162
#define ff_celt_model_energy_small
Definition tab.h:130
#define ff_celt_window
Definition tab.h:165
#define av_freep(p)
#define av_log(a,...)
static uint8_t tmp[40]
Definition aes_ctr.c:52
#define src1
Definition h264pred.c:141
int size
av_cold void av_tx_uninit(AVTXContext **ctx)
Frees a context and sets *ctx to NULL, does nothing when *ctx == NULL.
Definition tx.c:295
av_cold int av_tx_init(AVTXContext **ctx, av_tx_fn *tx, enum AVTXType type, int inv, int len, const void *scale, uint64_t flags)
Initialize a transform context with the given configuration (i)MDCTs with an odd length are currently...
Definition tx.c:903
@ AV_TX_FLOAT_MDCT
Standard MDCT with a sample data type of float, double or int32_t, respectively.
Definition tx.h:68
void(* av_tx_fn)(AVTXContext *s, void *out, void *in, ptrdiff_t stride)
Function pointer to a function to perform the transform.
Definition tx.h:151
else temp
Definition vf_mcdeint.c:275
static av_always_inline int diff(const struct color_info *a, const struct color_info *b, const int trans_thresh)
static const uint8_t quant[64]
Definition vmixdec.c:71
static double c[64]