encint.h 34 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846
  1. /********************************************************************
  2. * *
  3. * THIS FILE IS PART OF THE OggTheora SOFTWARE CODEC SOURCE CODE. *
  4. * USE, DISTRIBUTION AND REPRODUCTION OF THIS LIBRARY SOURCE IS *
  5. * GOVERNED BY A BSD-STYLE SOURCE LICENSE INCLUDED WITH THIS SOURCE *
  6. * IN 'COPYING'. PLEASE READ THESE TERMS BEFORE DISTRIBUTING. *
  7. * *
  8. * THE Theora SOURCE CODE IS COPYRIGHT (C) 2002-2009 *
  9. * by the Xiph.Org Foundation http://www.xiph.org/ *
  10. * *
  11. ********************************************************************
  12. function:
  13. last mod: $Id$
  14. ********************************************************************/
  15. #if !defined(_encint_H)
  16. # define _encint_H (1)
  17. # include "theora/theoraenc.h"
  18. # include "state.h"
  19. # include "mathops.h"
  20. # include "enquant.h"
  21. # include "huffenc.h"
  22. /*# define OC_COLLECT_METRICS*/
  23. typedef oc_mv oc_mv2[2];
  24. typedef struct oc_enc_opt_vtable oc_enc_opt_vtable;
  25. typedef struct oc_enc_opt_data oc_enc_opt_data;
  26. typedef struct oc_mb_enc_info oc_mb_enc_info;
  27. typedef struct oc_mode_scheme_chooser oc_mode_scheme_chooser;
  28. typedef struct oc_fr_state oc_fr_state;
  29. typedef struct oc_qii_state oc_qii_state;
  30. typedef struct oc_enc_pipeline_state oc_enc_pipeline_state;
  31. typedef struct oc_mode_rd oc_mode_rd;
  32. typedef struct oc_iir_filter oc_iir_filter;
  33. typedef struct oc_frame_metrics oc_frame_metrics;
  34. typedef struct oc_rc_state oc_rc_state;
  35. typedef struct th_enc_ctx oc_enc_ctx;
  36. typedef struct oc_token_checkpoint oc_token_checkpoint;
  37. /*Encoder-specific accelerated functions.*/
  38. # if defined(OC_X86_ASM)
  39. # if defined(_MSC_VER)
  40. # include "x86_vc/x86enc.h"
  41. # else
  42. # include "x86/x86enc.h"
  43. # endif
  44. # endif
  45. # if defined(OC_ARM_ASM)
  46. # include "arm/armenc.h"
  47. # endif
  48. # if !defined(oc_enc_accel_init)
  49. # define oc_enc_accel_init oc_enc_accel_init_c
  50. # endif
  51. # if defined(OC_ENC_USE_VTABLE)
  52. # if !defined(oc_enc_frag_sub)
  53. # define oc_enc_frag_sub(_enc,_diff,_src,_ref,_ystride) \
  54. ((*(_enc)->opt_vtable.frag_sub)(_diff,_src,_ref,_ystride))
  55. # endif
  56. # if !defined(oc_enc_frag_sub_128)
  57. # define oc_enc_frag_sub_128(_enc,_diff,_src,_ystride) \
  58. ((*(_enc)->opt_vtable.frag_sub_128)(_diff,_src,_ystride))
  59. # endif
  60. # if !defined(oc_enc_frag_sad)
  61. # define oc_enc_frag_sad(_enc,_src,_ref,_ystride) \
  62. ((*(_enc)->opt_vtable.frag_sad)(_src,_ref,_ystride))
  63. # endif
  64. # if !defined(oc_enc_frag_sad_thresh)
  65. # define oc_enc_frag_sad_thresh(_enc,_src,_ref,_ystride,_thresh) \
  66. ((*(_enc)->opt_vtable.frag_sad_thresh)(_src,_ref,_ystride,_thresh))
  67. # endif
  68. # if !defined(oc_enc_frag_sad2_thresh)
  69. # define oc_enc_frag_sad2_thresh(_enc,_src,_ref1,_ref2,_ystride,_thresh) \
  70. ((*(_enc)->opt_vtable.frag_sad2_thresh)(_src,_ref1,_ref2,_ystride,_thresh))
  71. # endif
  72. # if !defined(oc_enc_frag_intra_sad)
  73. # define oc_enc_frag_intra_sad(_enc,_src,_ystride) \
  74. ((*(_enc)->opt_vtable.frag_intra_sad)(_src,_ystride))
  75. # endif
  76. # if !defined(oc_enc_frag_satd)
  77. # define oc_enc_frag_satd(_enc,_dc,_src,_ref,_ystride) \
  78. ((*(_enc)->opt_vtable.frag_satd)(_dc,_src,_ref,_ystride))
  79. # endif
  80. # if !defined(oc_enc_frag_satd2)
  81. # define oc_enc_frag_satd2(_enc,_dc,_src,_ref1,_ref2,_ystride) \
  82. ((*(_enc)->opt_vtable.frag_satd2)(_dc,_src,_ref1,_ref2,_ystride))
  83. # endif
  84. # if !defined(oc_enc_frag_intra_satd)
  85. # define oc_enc_frag_intra_satd(_enc,_dc,_src,_ystride) \
  86. ((*(_enc)->opt_vtable.frag_intra_satd)(_dc,_src,_ystride))
  87. # endif
  88. # if !defined(oc_enc_frag_ssd)
  89. # define oc_enc_frag_ssd(_enc,_src,_ref,_ystride) \
  90. ((*(_enc)->opt_vtable.frag_ssd)(_src,_ref,_ystride))
  91. # endif
  92. # if !defined(oc_enc_frag_border_ssd)
  93. # define oc_enc_frag_border_ssd(_enc,_src,_ref,_ystride,_mask) \
  94. ((*(_enc)->opt_vtable.frag_border_ssd)(_src,_ref,_ystride,_mask))
  95. # endif
  96. # if !defined(oc_enc_frag_copy2)
  97. # define oc_enc_frag_copy2(_enc,_dst,_src1,_src2,_ystride) \
  98. ((*(_enc)->opt_vtable.frag_copy2)(_dst,_src1,_src2,_ystride))
  99. # endif
  100. # if !defined(oc_enc_enquant_table_init)
  101. # define oc_enc_enquant_table_init(_enc,_enquant,_dequant) \
  102. ((*(_enc)->opt_vtable.enquant_table_init)(_enquant,_dequant))
  103. # endif
  104. # if !defined(oc_enc_enquant_table_fixup)
  105. # define oc_enc_enquant_table_fixup(_enc,_enquant,_nqis) \
  106. ((*(_enc)->opt_vtable.enquant_table_fixup)(_enquant,_nqis))
  107. # endif
  108. # if !defined(oc_enc_quantize)
  109. # define oc_enc_quantize(_enc,_qdct,_dct,_dequant,_enquant) \
  110. ((*(_enc)->opt_vtable.quantize)(_qdct,_dct,_dequant,_enquant))
  111. # endif
  112. # if !defined(oc_enc_frag_recon_intra)
  113. # define oc_enc_frag_recon_intra(_enc,_dst,_ystride,_residue) \
  114. ((*(_enc)->opt_vtable.frag_recon_intra)(_dst,_ystride,_residue))
  115. # endif
  116. # if !defined(oc_enc_frag_recon_inter)
  117. # define oc_enc_frag_recon_inter(_enc,_dst,_src,_ystride,_residue) \
  118. ((*(_enc)->opt_vtable.frag_recon_inter)(_dst,_src,_ystride,_residue))
  119. # endif
  120. # if !defined(oc_enc_fdct8x8)
  121. # define oc_enc_fdct8x8(_enc,_y,_x) \
  122. ((*(_enc)->opt_vtable.fdct8x8)(_y,_x))
  123. # endif
  124. # else
  125. # if !defined(oc_enc_frag_sub)
  126. # define oc_enc_frag_sub(_enc,_diff,_src,_ref,_ystride) \
  127. oc_enc_frag_sub_c(_diff,_src,_ref,_ystride)
  128. # endif
  129. # if !defined(oc_enc_frag_sub_128)
  130. # define oc_enc_frag_sub_128(_enc,_diff,_src,_ystride) \
  131. oc_enc_frag_sub_128_c(_diff,_src,_ystride)
  132. # endif
  133. # if !defined(oc_enc_frag_sad)
  134. # define oc_enc_frag_sad(_enc,_src,_ref,_ystride) \
  135. oc_enc_frag_sad_c(_src,_ref,_ystride)
  136. # endif
  137. # if !defined(oc_enc_frag_sad_thresh)
  138. # define oc_enc_frag_sad_thresh(_enc,_src,_ref,_ystride,_thresh) \
  139. oc_enc_frag_sad_thresh_c(_src,_ref,_ystride,_thresh)
  140. # endif
  141. # if !defined(oc_enc_frag_sad2_thresh)
  142. # define oc_enc_frag_sad2_thresh(_enc,_src,_ref1,_ref2,_ystride,_thresh) \
  143. oc_enc_frag_sad2_thresh_c(_src,_ref1,_ref2,_ystride,_thresh)
  144. # endif
  145. # if !defined(oc_enc_frag_intra_sad)
  146. # define oc_enc_frag_intra_sad(_enc,_src,_ystride) \
  147. oc_enc_frag_intra_sad_c(_src,_ystride)
  148. # endif
  149. # if !defined(oc_enc_frag_satd)
  150. # define oc_enc_frag_satd(_enc,_dc,_src,_ref,_ystride) \
  151. oc_enc_frag_satd_c(_dc,_src,_ref,_ystride)
  152. # endif
  153. # if !defined(oc_enc_frag_satd2)
  154. # define oc_enc_frag_satd2(_enc,_dc,_src,_ref1,_ref2,_ystride) \
  155. oc_enc_frag_satd2_c(_dc,_src,_ref1,_ref2,_ystride)
  156. # endif
  157. # if !defined(oc_enc_frag_intra_satd)
  158. # define oc_enc_frag_intra_satd(_enc,_dc,_src,_ystride) \
  159. oc_enc_frag_intra_satd_c(_dc,_src,_ystride)
  160. # endif
  161. # if !defined(oc_enc_frag_ssd)
  162. # define oc_enc_frag_ssd(_enc,_src,_ref,_ystride) \
  163. oc_enc_frag_ssd_c(_src,_ref,_ystride)
  164. # endif
  165. # if !defined(oc_enc_frag_border_ssd)
  166. # define oc_enc_frag_border_ssd(_enc,_src,_ref,_ystride,_mask) \
  167. oc_enc_frag_border_ssd_c(_src,_ref,_ystride,_mask)
  168. # endif
  169. # if !defined(oc_enc_frag_copy2)
  170. # define oc_enc_frag_copy2(_enc,_dst,_src1,_src2,_ystride) \
  171. oc_enc_frag_copy2_c(_dst,_src1,_src2,_ystride)
  172. # endif
  173. # if !defined(oc_enc_enquant_table_init)
  174. # define oc_enc_enquant_table_init(_enc,_enquant,_dequant) \
  175. oc_enc_enquant_table_init_c(_enquant,_dequant)
  176. # endif
  177. # if !defined(oc_enc_enquant_table_fixup)
  178. # define oc_enc_enquant_table_fixup(_enc,_enquant,_nqis) \
  179. oc_enc_enquant_table_fixup_c(_enquant,_nqis)
  180. # endif
  181. # if !defined(oc_enc_quantize)
  182. # define oc_enc_quantize(_enc,_qdct,_dct,_dequant,_enquant) \
  183. oc_enc_quantize_c(_qdct,_dct,_dequant,_enquant)
  184. # endif
  185. # if !defined(oc_enc_frag_recon_intra)
  186. # define oc_enc_frag_recon_intra(_enc,_dst,_ystride,_residue) \
  187. oc_frag_recon_intra_c(_dst,_ystride,_residue)
  188. # endif
  189. # if !defined(oc_enc_frag_recon_inter)
  190. # define oc_enc_frag_recon_inter(_enc,_dst,_src,_ystride,_residue) \
  191. oc_frag_recon_inter_c(_dst,_src,_ystride,_residue)
  192. # endif
  193. # if !defined(oc_enc_fdct8x8)
  194. # define oc_enc_fdct8x8(_enc,_y,_x) oc_enc_fdct8x8_c(_y,_x)
  195. # endif
  196. # endif
  197. /*Constants for the packet-out state machine specific to the encoder.*/
  198. /*Next packet to emit: Data packet, but none are ready yet.*/
  199. #define OC_PACKET_EMPTY (0)
  200. /*Next packet to emit: Data packet, and one is ready.*/
  201. #define OC_PACKET_READY (1)
  202. /*All features enabled.*/
  203. #define OC_SP_LEVEL_SLOW (0)
  204. /*Enable early skip.*/
  205. #define OC_SP_LEVEL_EARLY_SKIP (1)
  206. /*Use analysis shortcuts, single quantizer, and faster tokenization.*/
  207. #define OC_SP_LEVEL_FAST_ANALYSIS (2)
  208. /*Use SAD instead of SATD*/
  209. #define OC_SP_LEVEL_NOSATD (3)
  210. /*Disable motion compensation.*/
  211. #define OC_SP_LEVEL_NOMC (4)
  212. /*Maximum valid speed level.*/
  213. #define OC_SP_LEVEL_MAX (4)
  214. /*The number of extra bits of precision at which to store rate metrics.*/
  215. # define OC_BIT_SCALE (6)
  216. /*The number of extra bits of precision at which to store RMSE metrics.
  217. This must be at least half OC_BIT_SCALE (rounded up).*/
  218. # define OC_RMSE_SCALE (5)
  219. /*The number of quantizer bins to partition statistics into.*/
  220. # define OC_LOGQ_BINS (8)
  221. /*The number of SAD/SATD bins to partition statistics into.*/
  222. # define OC_COMP_BINS (24)
  223. /*The number of bits of precision to drop from SAD and SATD scores
  224. to assign them to a bin.*/
  225. # define OC_SAD_SHIFT (6)
  226. # define OC_SATD_SHIFT (9)
  227. /*Masking is applied by scaling the D used in R-D optimization (via rd_scale)
  228. or the lambda parameter (via rd_iscale).
  229. These are only equivalent within a single block; when more than one block is
  230. being considered, the former is the interpretation used.*/
  231. /*This must be at least 4 for OC_RD_SKIP_SCALE() to work below.*/
  232. # define OC_RD_SCALE_BITS (12-OC_BIT_SCALE)
  233. # define OC_RD_ISCALE_BITS (11)
  234. /*This macro is applied to _ssd values with just 4 bits of headroom
  235. ((15-OC_RMSE_SCALE)*2+OC_BIT_SCALE+2); since we want to allow rd_scales as
  236. large as 16, and need additional fractional bits, our only recourse that
  237. doesn't lose precision on blocks with very small SSDs is to use a wider
  238. multiply.*/
  239. # if LONG_MAX>2147483647
  240. # define OC_RD_SCALE(_ssd,_rd_scale) \
  241. ((unsigned)((unsigned long)(_ssd)*(_rd_scale) \
  242. +((1<<OC_RD_SCALE_BITS)>>1)>>OC_RD_SCALE_BITS))
  243. # else
  244. # define OC_RD_SCALE(_ssd,_rd_scale) \
  245. (((_ssd)>>OC_RD_SCALE_BITS)*(_rd_scale) \
  246. +(((_ssd)&(1<<OC_RD_SCALE_BITS)-1)*(_rd_scale) \
  247. +((1<<OC_RD_SCALE_BITS)>>1)>>OC_RD_SCALE_BITS))
  248. # endif
  249. # define OC_RD_SKIP_SCALE(_ssd,_rd_scale) \
  250. ((_ssd)*(_rd_scale)+((1<<OC_RD_SCALE_BITS-4)>>1)>>OC_RD_SCALE_BITS-4)
  251. # define OC_RD_ISCALE(_lambda,_rd_iscale) \
  252. ((_lambda)*(_rd_iscale)+((1<<OC_RD_ISCALE_BITS)>>1)>>OC_RD_ISCALE_BITS)
  253. /*The bits used for each of the MB mode codebooks.*/
  254. extern const unsigned char OC_MODE_BITS[2][OC_NMODES];
  255. /*The bits used for each of the MV codebooks.*/
  256. extern const unsigned char OC_MV_BITS[2][64];
  257. /*The minimum value that can be stored in a SB run for each codeword.
  258. The last entry is the upper bound on the length of a single SB run.*/
  259. extern const ogg_uint16_t OC_SB_RUN_VAL_MIN[8];
  260. /*The bits used for each SB run codeword.*/
  261. extern const unsigned char OC_SB_RUN_CODE_NBITS[7];
  262. /*The bits used for each block run length (starting with 1).*/
  263. extern const unsigned char OC_BLOCK_RUN_CODE_NBITS[30];
  264. /*Encoder specific functions with accelerated variants.*/
  265. struct oc_enc_opt_vtable{
  266. void (*frag_sub)(ogg_int16_t _diff[64],const unsigned char *_src,
  267. const unsigned char *_ref,int _ystride);
  268. void (*frag_sub_128)(ogg_int16_t _diff[64],
  269. const unsigned char *_src,int _ystride);
  270. unsigned (*frag_sad)(const unsigned char *_src,
  271. const unsigned char *_ref,int _ystride);
  272. unsigned (*frag_sad_thresh)(const unsigned char *_src,
  273. const unsigned char *_ref,int _ystride,unsigned _thresh);
  274. unsigned (*frag_sad2_thresh)(const unsigned char *_src,
  275. const unsigned char *_ref1,const unsigned char *_ref2,int _ystride,
  276. unsigned _thresh);
  277. unsigned (*frag_intra_sad)(const unsigned char *_src,int _ystride);
  278. unsigned (*frag_satd)(int *_dc,const unsigned char *_src,
  279. const unsigned char *_ref,int _ystride);
  280. unsigned (*frag_satd2)(int *_dc,const unsigned char *_src,
  281. const unsigned char *_ref1,const unsigned char *_ref2,int _ystride);
  282. unsigned (*frag_intra_satd)(int *_dc,const unsigned char *_src,int _ystride);
  283. unsigned (*frag_ssd)(const unsigned char *_src,
  284. const unsigned char *_ref,int _ystride);
  285. unsigned (*frag_border_ssd)(const unsigned char *_src,
  286. const unsigned char *_ref,int _ystride,ogg_int64_t _mask);
  287. void (*frag_copy2)(unsigned char *_dst,
  288. const unsigned char *_src1,const unsigned char *_src2,int _ystride);
  289. void (*enquant_table_init)(void *_enquant,
  290. const ogg_uint16_t _dequant[64]);
  291. void (*enquant_table_fixup)(void *_enquant[3][3][2],int _nqis);
  292. int (*quantize)(ogg_int16_t _qdct[64],const ogg_int16_t _dct[64],
  293. const ogg_uint16_t _dequant[64],const void *_enquant);
  294. void (*frag_recon_intra)(unsigned char *_dst,int _ystride,
  295. const ogg_int16_t _residue[64]);
  296. void (*frag_recon_inter)(unsigned char *_dst,
  297. const unsigned char *_src,int _ystride,const ogg_int16_t _residue[64]);
  298. void (*fdct8x8)(ogg_int16_t _y[64],const ogg_int16_t _x[64]);
  299. };
  300. /*Encoder specific data that varies according to which variants of the above
  301. functions are used.*/
  302. struct oc_enc_opt_data{
  303. /*The size of a single quantizer table.
  304. This must be a multiple of enquant_table_alignment.*/
  305. size_t enquant_table_size;
  306. /*The alignment required for the quantizer tables.
  307. This must be a positive power of two.*/
  308. int enquant_table_alignment;
  309. };
  310. void oc_enc_accel_init(oc_enc_ctx *_enc);
  311. /*Encoder-specific macroblock information.*/
  312. struct oc_mb_enc_info{
  313. /*Neighboring macro blocks that have MVs available from the current frame.*/
  314. unsigned cneighbors[4];
  315. /*Neighboring macro blocks to use for MVs from the previous frame.*/
  316. unsigned pneighbors[4];
  317. /*The number of current-frame neighbors.*/
  318. unsigned char ncneighbors;
  319. /*The number of previous-frame neighbors.*/
  320. unsigned char npneighbors;
  321. /*Flags indicating which MB modes have been refined.*/
  322. unsigned char refined;
  323. /*Motion vectors for a macro block for the current frame and the
  324. previous two frames.
  325. Each is a set of 2 vectors against OC_FRAME_GOLD and OC_FRAME_PREV, which
  326. can be used to estimate constant velocity and constant acceleration
  327. predictors.
  328. Uninitialized MVs are (0,0).*/
  329. oc_mv2 analysis_mv[3];
  330. /*Current unrefined analysis MVs.*/
  331. oc_mv unref_mv[2];
  332. /*Unrefined block MVs.*/
  333. oc_mv block_mv[4];
  334. /*Refined block MVs.*/
  335. oc_mv ref_mv[4];
  336. /*Minimum motion estimation error from the analysis stage.*/
  337. ogg_uint16_t error[2];
  338. /*MB error for half-pel refinement for each frame type.*/
  339. unsigned satd[2];
  340. /*Block error for half-pel refinement.*/
  341. unsigned block_satd[4];
  342. };
  343. /*State machine to estimate the opportunity cost of coding a MB mode.*/
  344. struct oc_mode_scheme_chooser{
  345. /*Pointers to the a list containing the index of each mode in the mode
  346. alphabet used by each scheme.
  347. The first entry points to the dynamic scheme0_ranks, while the remaining 7
  348. point to the constant entries stored in OC_MODE_SCHEMES.*/
  349. const unsigned char *mode_ranks[8];
  350. /*The ranks for each mode when coded with scheme 0.
  351. These are optimized so that the more frequent modes have lower ranks.*/
  352. unsigned char scheme0_ranks[OC_NMODES];
  353. /*The list of modes, sorted in descending order of frequency, that
  354. corresponds to the ranks above.*/
  355. unsigned char scheme0_list[OC_NMODES];
  356. /*The number of times each mode has been chosen so far.*/
  357. unsigned mode_counts[OC_NMODES];
  358. /*The list of mode coding schemes, sorted in ascending order of bit cost.*/
  359. unsigned char scheme_list[8];
  360. /*The number of bits used by each mode coding scheme.*/
  361. ptrdiff_t scheme_bits[8];
  362. };
  363. void oc_mode_scheme_chooser_init(oc_mode_scheme_chooser *_chooser);
  364. /*State to track coded block flags and their bit cost.
  365. We use opportunity cost to measure the bits required to code or skip the next
  366. block, using the cheaper of the cost to code it fully or partially, so long
  367. as both are possible.*/
  368. struct oc_fr_state{
  369. /*The number of bits required for the coded block flags so far this frame.*/
  370. ptrdiff_t bits;
  371. /*The length of the current run for the partial super block flag, not
  372. including the current super block.*/
  373. unsigned sb_partial_count:16;
  374. /*The length of the current run for the full super block flag, not
  375. including the current super block.*/
  376. unsigned sb_full_count:16;
  377. /*The length of the coded block flag run when the current super block
  378. started.*/
  379. unsigned b_coded_count_prev:6;
  380. /*The coded block flag when the current super block started.*/
  381. signed int b_coded_prev:2;
  382. /*The length of the current coded block flag run.*/
  383. unsigned b_coded_count:6;
  384. /*The current coded block flag.*/
  385. signed int b_coded:2;
  386. /*The number of blocks processed in the current super block.*/
  387. unsigned b_count:5;
  388. /*Whether or not it is cheaper to code the current super block partially,
  389. even if it could still be coded fully.*/
  390. unsigned sb_prefer_partial:1;
  391. /*Whether the last super block was coded partially.*/
  392. signed int sb_partial:2;
  393. /*The number of bits required for the flags for the current super block.*/
  394. unsigned sb_bits:6;
  395. /*Whether the last non-partial super block was coded fully.*/
  396. signed int sb_full:2;
  397. };
  398. struct oc_qii_state{
  399. ptrdiff_t bits;
  400. unsigned qi01_count:14;
  401. signed int qi01:2;
  402. unsigned qi12_count:14;
  403. signed int qi12:2;
  404. };
  405. /*Temporary encoder state for the analysis pipeline.*/
  406. struct oc_enc_pipeline_state{
  407. /*DCT coefficient storage.
  408. This is kept off the stack because a) gcc can't align things on the stack
  409. reliably on ARM, and b) it avoids (unintentional) data hazards between
  410. ARM and NEON code.*/
  411. OC_ALIGN16(ogg_int16_t dct_data[64*3]);
  412. OC_ALIGN16(signed char bounding_values[256]);
  413. oc_fr_state fr[3];
  414. oc_qii_state qs[3];
  415. /*Skip SSD storage for the current MCU in each plane.*/
  416. unsigned *skip_ssd[3];
  417. /*Coded/uncoded fragment lists for each plane for the current MCU.*/
  418. ptrdiff_t *coded_fragis[3];
  419. ptrdiff_t *uncoded_fragis[3];
  420. ptrdiff_t ncoded_fragis[3];
  421. ptrdiff_t nuncoded_fragis[3];
  422. /*The starting fragment for the current MCU in each plane.*/
  423. ptrdiff_t froffset[3];
  424. /*The starting row for the current MCU in each plane.*/
  425. int fragy0[3];
  426. /*The ending row for the current MCU in each plane.*/
  427. int fragy_end[3];
  428. /*The starting superblock for the current MCU in each plane.*/
  429. unsigned sbi0[3];
  430. /*The ending superblock for the current MCU in each plane.*/
  431. unsigned sbi_end[3];
  432. /*The number of tokens for zzi=1 for each color plane.*/
  433. int ndct_tokens1[3];
  434. /*The outstanding eob_run count for zzi=1 for each color plane.*/
  435. int eob_run1[3];
  436. /*Whether or not the loop filter is enabled.*/
  437. int loop_filter;
  438. };
  439. /*Statistics used to estimate R-D cost of a block in a given coding mode.
  440. See modedec.h for more details.*/
  441. struct oc_mode_rd{
  442. /*The expected bits used by the DCT tokens, shifted by OC_BIT_SCALE.*/
  443. ogg_int16_t rate;
  444. /*The expected square root of the sum of squared errors, shifted by
  445. OC_RMSE_SCALE.*/
  446. ogg_int16_t rmse;
  447. };
  448. # if defined(OC_COLLECT_METRICS)
  449. # include "collect.h"
  450. # endif
  451. /*A 2nd order low-pass Bessel follower.
  452. We use this for rate control because it has fast reaction time, but is
  453. critically damped.*/
  454. struct oc_iir_filter{
  455. ogg_int32_t c[2];
  456. ogg_int64_t g;
  457. ogg_int32_t x[2];
  458. ogg_int32_t y[2];
  459. };
  460. /*The 2-pass metrics associated with a single frame.*/
  461. struct oc_frame_metrics{
  462. /*The log base 2 of the scale factor for this frame in Q24 format.*/
  463. ogg_int32_t log_scale;
  464. /*The number of application-requested duplicates of this frame.*/
  465. unsigned dup_count:31;
  466. /*The frame type from pass 1.*/
  467. unsigned frame_type:1;
  468. /*The frame activity average from pass 1.*/
  469. unsigned activity_avg;
  470. };
  471. /*Rate control state information.*/
  472. struct oc_rc_state{
  473. /*The target average bits per frame.*/
  474. ogg_int64_t bits_per_frame;
  475. /*The current buffer fullness (bits available to be used).*/
  476. ogg_int64_t fullness;
  477. /*The target buffer fullness.
  478. This is where we'd like to be by the last keyframe the appears in the next
  479. buf_delay frames.*/
  480. ogg_int64_t target;
  481. /*The maximum buffer fullness (total size of the buffer).*/
  482. ogg_int64_t max;
  483. /*The log of the number of pixels in a frame in Q57 format.*/
  484. ogg_int64_t log_npixels;
  485. /*The exponent used in the rate model in Q8 format.*/
  486. unsigned exp[2];
  487. /*The number of frames to distribute the buffer usage over.*/
  488. int buf_delay;
  489. /*The total drop count from the previous frame.
  490. This includes duplicates explicitly requested via the
  491. TH_ENCCTL_SET_DUP_COUNT API as well as frames we chose to drop ourselves.*/
  492. ogg_uint32_t prev_drop_count;
  493. /*The log of an estimated scale factor used to obtain the real framerate, for
  494. VFR sources or, e.g., 12 fps content doubled to 24 fps, etc.*/
  495. ogg_int64_t log_drop_scale;
  496. /*The log of estimated scale factor for the rate model in Q57 format.*/
  497. ogg_int64_t log_scale[2];
  498. /*The log of the target quantizer level in Q57 format.*/
  499. ogg_int64_t log_qtarget;
  500. /*Will we drop frames to meet bitrate target?*/
  501. unsigned char drop_frames;
  502. /*Do we respect the maximum buffer fullness?*/
  503. unsigned char cap_overflow;
  504. /*Can the reservoir go negative?*/
  505. unsigned char cap_underflow;
  506. /*Second-order lowpass filters to track scale and VFR.*/
  507. oc_iir_filter scalefilter[2];
  508. int inter_count;
  509. int inter_delay;
  510. int inter_delay_target;
  511. oc_iir_filter vfrfilter;
  512. /*Two-pass mode state.
  513. 0 => 1-pass encoding.
  514. 1 => 1st pass of 2-pass encoding.
  515. 2 => 2nd pass of 2-pass encoding.*/
  516. int twopass;
  517. /*Buffer for current frame metrics.*/
  518. unsigned char twopass_buffer[48];
  519. /*The number of bytes in the frame metrics buffer.
  520. When 2-pass encoding is enabled, this is set to 0 after each frame is
  521. submitted, and must be non-zero before the next frame will be accepted.*/
  522. int twopass_buffer_bytes;
  523. int twopass_buffer_fill;
  524. /*Whether or not to force the next frame to be a keyframe.*/
  525. unsigned char twopass_force_kf;
  526. /*The metrics for the previous frame.*/
  527. oc_frame_metrics prev_metrics;
  528. /*The metrics for the current frame.*/
  529. oc_frame_metrics cur_metrics;
  530. /*The buffered metrics for future frames.*/
  531. oc_frame_metrics *frame_metrics;
  532. int nframe_metrics;
  533. int cframe_metrics;
  534. /*The index of the current frame in the circular metric buffer.*/
  535. int frame_metrics_head;
  536. /*The frame count of each type (keyframes, delta frames, and dup frames);
  537. 32 bits limits us to 2.268 years at 60 fps.*/
  538. ogg_uint32_t frames_total[3];
  539. /*The number of frames of each type yet to be processed.*/
  540. ogg_uint32_t frames_left[3];
  541. /*The sum of the scale values for each frame type.*/
  542. ogg_int64_t scale_sum[2];
  543. /*The start of the window over which the current scale sums are taken.*/
  544. int scale_window0;
  545. /*The end of the window over which the current scale sums are taken.*/
  546. int scale_window_end;
  547. /*The frame count of each type in the current 2-pass window; this does not
  548. include dup frames.*/
  549. int nframes[3];
  550. /*The total accumulated estimation bias.*/
  551. ogg_int64_t rate_bias;
  552. };
  553. void oc_rc_state_init(oc_rc_state *_rc,oc_enc_ctx *_enc);
  554. void oc_rc_state_clear(oc_rc_state *_rc);
  555. void oc_enc_rc_resize(oc_enc_ctx *_enc);
  556. int oc_enc_select_qi(oc_enc_ctx *_enc,int _qti,int _clamp);
  557. void oc_enc_calc_lambda(oc_enc_ctx *_enc,int _frame_type);
  558. int oc_enc_update_rc_state(oc_enc_ctx *_enc,
  559. long _bits,int _qti,int _qi,int _trial,int _droppable);
  560. int oc_enc_rc_2pass_out(oc_enc_ctx *_enc,unsigned char **_buf);
  561. int oc_enc_rc_2pass_in(oc_enc_ctx *_enc,unsigned char *_buf,size_t _bytes);
  562. /*The internal encoder state.*/
  563. struct th_enc_ctx{
  564. /*Shared encoder/decoder state.*/
  565. oc_theora_state state;
  566. /*Buffer in which to assemble packets.*/
  567. oggpack_buffer opb;
  568. /*Encoder-specific macroblock information.*/
  569. oc_mb_enc_info *mb_info;
  570. /*DC coefficients after prediction.*/
  571. ogg_int16_t *frag_dc;
  572. /*The list of coded macro blocks, in coded order.*/
  573. unsigned *coded_mbis;
  574. /*The number of coded macro blocks.*/
  575. size_t ncoded_mbis;
  576. /*Whether or not packets are ready to be emitted.
  577. This takes on negative values while there are remaining header packets to
  578. be emitted, reaches 0 when the codec is ready for input, and becomes
  579. positive when a frame has been processed and data packets are ready.*/
  580. int packet_state;
  581. /*The maximum distance between keyframes.*/
  582. ogg_uint32_t keyframe_frequency_force;
  583. /*The number of duplicates to produce for the next frame.*/
  584. ogg_uint32_t dup_count;
  585. /*The number of duplicates remaining to be emitted for the current frame.*/
  586. ogg_uint32_t nqueued_dups;
  587. /*The number of duplicates emitted for the last frame.*/
  588. ogg_uint32_t prev_dup_count;
  589. /*The current speed level.*/
  590. int sp_level;
  591. /*Whether or not VP3 compatibility mode has been enabled.*/
  592. unsigned char vp3_compatible;
  593. /*Whether or not any INTER frames have been coded.*/
  594. unsigned char coded_inter_frame;
  595. /*Whether or not previous frame was dropped.*/
  596. unsigned char prevframe_dropped;
  597. /*Stores most recently chosen Huffman tables for each frame type, DC and AC
  598. coefficients, and luma and chroma tokens.
  599. The actual Huffman table used for a given coefficient depends not only on
  600. the choice made here, but also its index in the zig-zag ordering.*/
  601. unsigned char huff_idxs[2][2][2];
  602. /*Current count of bits used by each MV coding mode.*/
  603. size_t mv_bits[2];
  604. /*The mode scheme chooser for estimating mode coding costs.*/
  605. oc_mode_scheme_chooser chooser;
  606. /*Temporary encoder state for the analysis pipeline.*/
  607. oc_enc_pipeline_state pipe;
  608. /*The number of vertical super blocks in an MCU.*/
  609. int mcu_nvsbs;
  610. /*The SSD error for skipping each fragment in the current MCU.*/
  611. unsigned *mcu_skip_ssd;
  612. /*The masking scale factors for chroma blocks in the current MCU.*/
  613. ogg_uint16_t *mcu_rd_scale;
  614. ogg_uint16_t *mcu_rd_iscale;
  615. /*The DCT token lists for each coefficient and each plane.*/
  616. unsigned char **dct_tokens[3];
  617. /*The extra bits associated with each DCT token.*/
  618. ogg_uint16_t **extra_bits[3];
  619. /*The number of DCT tokens for each coefficient for each plane.*/
  620. ptrdiff_t ndct_tokens[3][64];
  621. /*Pending EOB runs for each coefficient for each plane.*/
  622. ogg_uint16_t eob_run[3][64];
  623. /*The offset of the first DCT token for each coefficient for each plane.*/
  624. unsigned char dct_token_offs[3][64];
  625. /*The last DC coefficient for each plane and reference frame.*/
  626. int dc_pred_last[3][4];
  627. #if defined(OC_COLLECT_METRICS)
  628. /*Fragment SAD statistics for MB mode estimation metrics.*/
  629. unsigned *frag_sad;
  630. /*Fragment SATD statistics for MB mode estimation metrics.*/
  631. unsigned *frag_satd;
  632. /*Fragment SSD statistics for MB mode estimation metrics.*/
  633. unsigned *frag_ssd;
  634. #endif
  635. /*The R-D optimization parameter.*/
  636. int lambda;
  637. /*The average block "activity" of the previous frame.*/
  638. unsigned activity_avg;
  639. /*The average MB luma of the previous frame.*/
  640. unsigned luma_avg;
  641. /*The huffman tables in use.*/
  642. th_huff_code huff_codes[TH_NHUFFMAN_TABLES][TH_NDCT_TOKENS];
  643. /*The quantization parameters in use.*/
  644. th_quant_info qinfo;
  645. /*The original DC coefficients saved off from the dequatization tables.*/
  646. ogg_uint16_t dequant_dc[64][3][2];
  647. /*Condensed dequantization tables.*/
  648. const ogg_uint16_t *dequant[3][3][2];
  649. /*Condensed quantization tables.*/
  650. void *enquant[3][3][2];
  651. /*The full set of quantization tables.*/
  652. void *enquant_tables[64][3][2];
  653. /*Storage for the quantization tables.*/
  654. unsigned char *enquant_table_data;
  655. /*An "average" quantizer for each frame type (INTRA or INTER) and qi value.
  656. This is used to parameterize the rate control decisions.
  657. They are kept in the log domain to simplify later processing.
  658. These are DCT domain quantizers, and so are scaled by an additional factor
  659. of 4 from the pixel domain.*/
  660. ogg_int64_t log_qavg[2][64];
  661. /*The "average" quantizer futher partitioned by color plane.
  662. This is used to parameterize mode decision.
  663. These are DCT domain quantizers, and so are scaled by an additional factor
  664. of 4 from the pixel domain.*/
  665. ogg_int16_t log_plq[64][3][2];
  666. /*The R-D scale factors to apply to chroma blocks for a given frame type
  667. (INTRA or INTER) and qi value.
  668. The first is the "D" modifier (rd_scale), while the second is the "lambda"
  669. modifier (rd_iscale).*/
  670. ogg_uint16_t chroma_rd_scale[2][64][2];
  671. /*The interpolated mode decision R-D lookup tables for the current
  672. quantizers, color plane, and quantization type.*/
  673. oc_mode_rd mode_rd[3][3][2][OC_COMP_BINS];
  674. /*The buffer state used to drive rate control.*/
  675. oc_rc_state rc;
  676. # if defined(OC_ENC_USE_VTABLE)
  677. /*Table for encoder acceleration functions.*/
  678. oc_enc_opt_vtable opt_vtable;
  679. # endif
  680. /*Table for encoder data used by accelerated functions.*/
  681. oc_enc_opt_data opt_data;
  682. };
  683. void oc_enc_analyze_intra(oc_enc_ctx *_enc,int _recode);
  684. int oc_enc_analyze_inter(oc_enc_ctx *_enc,int _allow_keyframe,int _recode);
  685. /*Perform fullpel motion search for a single MB against both reference frames.*/
  686. void oc_mcenc_search(oc_enc_ctx *_enc,int _mbi);
  687. /*Refine a MB MV for one frame.*/
  688. void oc_mcenc_refine1mv(oc_enc_ctx *_enc,int _mbi,int _frame);
  689. /*Refine the block MVs.*/
  690. void oc_mcenc_refine4mv(oc_enc_ctx *_enc,int _mbi);
  691. /*Used to rollback a tokenlog transaction when we retroactively decide to skip
  692. a fragment.
  693. A checkpoint is taken right before each token is added.*/
  694. struct oc_token_checkpoint{
  695. /*The color plane the token was added to.*/
  696. unsigned char pli;
  697. /*The zig-zag index the token was added to.*/
  698. unsigned char zzi;
  699. /*The outstanding EOB run count before the token was added.*/
  700. ogg_uint16_t eob_run;
  701. /*The token count before the token was added.*/
  702. ptrdiff_t ndct_tokens;
  703. };
  704. void oc_enc_tokenize_start(oc_enc_ctx *_enc);
  705. int oc_enc_tokenize_ac(oc_enc_ctx *_enc,int _pli,ptrdiff_t _fragi,
  706. ogg_int16_t *_qdct_out,const ogg_int16_t *_qdct_in,
  707. const ogg_uint16_t *_dequant,const ogg_int16_t *_dct,
  708. int _zzi,oc_token_checkpoint **_stack,int _lambda,int _acmin);
  709. int oc_enc_tokenize_ac_fast(oc_enc_ctx *_enc,int _pli,ptrdiff_t _fragi,
  710. ogg_int16_t *_qdct_out,const ogg_int16_t *_qdct_in,
  711. const ogg_uint16_t *_dequant,const ogg_int16_t *_dct,
  712. int _zzi,oc_token_checkpoint **_stack,int _lambda,int _acmin);
  713. void oc_enc_tokenlog_rollback(oc_enc_ctx *_enc,
  714. const oc_token_checkpoint *_stack,int _n);
  715. void oc_enc_pred_dc_frag_rows(oc_enc_ctx *_enc,
  716. int _pli,int _fragy0,int _frag_yend);
  717. void oc_enc_tokenize_dc_frag_list(oc_enc_ctx *_enc,int _pli,
  718. const ptrdiff_t *_coded_fragis,ptrdiff_t _ncoded_fragis,
  719. int _prev_ndct_tokens1,int _prev_eob_run1);
  720. void oc_enc_tokenize_finish(oc_enc_ctx *_enc);
  721. /*Utility routine to encode one of the header packets.*/
  722. int oc_state_flushheader(oc_theora_state *_state,int *_packet_state,
  723. oggpack_buffer *_opb,const th_quant_info *_qinfo,
  724. const th_huff_code _codes[TH_NHUFFMAN_TABLES][TH_NDCT_TOKENS],
  725. const char *_vendor,th_comment *_tc,ogg_packet *_op);
  726. /*Default pure-C implementations of encoder-specific accelerated functions.*/
  727. void oc_enc_accel_init_c(oc_enc_ctx *_enc);
  728. void oc_enc_frag_sub_c(ogg_int16_t _diff[64],
  729. const unsigned char *_src,const unsigned char *_ref,int _ystride);
  730. void oc_enc_frag_sub_128_c(ogg_int16_t _diff[64],
  731. const unsigned char *_src,int _ystride);
  732. unsigned oc_enc_frag_sad_c(const unsigned char *_src,
  733. const unsigned char *_ref,int _ystride);
  734. unsigned oc_enc_frag_sad_thresh_c(const unsigned char *_src,
  735. const unsigned char *_ref,int _ystride,unsigned _thresh);
  736. unsigned oc_enc_frag_sad2_thresh_c(const unsigned char *_src,
  737. const unsigned char *_ref1,const unsigned char *_ref2,int _ystride,
  738. unsigned _thresh);
  739. unsigned oc_enc_frag_intra_sad_c(const unsigned char *_src, int _ystride);
  740. unsigned oc_enc_frag_satd_c(int *_dc,const unsigned char *_src,
  741. const unsigned char *_ref,int _ystride);
  742. unsigned oc_enc_frag_satd2_c(int *_dc,const unsigned char *_src,
  743. const unsigned char *_ref1,const unsigned char *_ref2,int _ystride);
  744. unsigned oc_enc_frag_intra_satd_c(int *_dc,
  745. const unsigned char *_src,int _ystride);
  746. unsigned oc_enc_frag_ssd_c(const unsigned char *_src,
  747. const unsigned char *_ref,int _ystride);
  748. unsigned oc_enc_frag_border_ssd_c(const unsigned char *_src,
  749. const unsigned char *_ref,int _ystride,ogg_int64_t _mask);
  750. void oc_enc_frag_copy2_c(unsigned char *_dst,
  751. const unsigned char *_src1,const unsigned char *_src2,int _ystride);
  752. void oc_enc_enquant_table_init_c(void *_enquant,
  753. const ogg_uint16_t _dequant[64]);
  754. void oc_enc_enquant_table_fixup_c(void *_enquant[3][3][2],int _nqis);
  755. int oc_enc_quantize_c(ogg_int16_t _qdct[64],const ogg_int16_t _dct[64],
  756. const ogg_uint16_t _dequant[64],const void *_enquant);
  757. void oc_enc_fdct8x8_c(ogg_int16_t _y[64],const ogg_int16_t _x[64]);
  758. #endif