mirror of
https://github.com/opencv/opencv.git
synced 2026-07-22 11:53:04 +04:00
Compare commits
2249 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 725e440d27 | |||
| 121034876d | |||
| 9627ab9462 | |||
| 499035c6ea | |||
| 71765858dc | |||
| 9a2a34f94e | |||
| 838c34eee1 | |||
| 253429d3f1 | |||
| 38f7cd7173 | |||
| eab7faf536 | |||
| 83391ac59d | |||
| a8a93a57e7 | |||
| f637629c5a | |||
| 93aa94e71e | |||
| 86b46a27cf | |||
| 1bc3077890 | |||
| fc27a343e9 | |||
| 692d6168b3 | |||
| de9787a6ac | |||
| d2bf2be8e6 | |||
| b7292bc899 | |||
| dbd4a0e5e6 | |||
| b361209d52 | |||
| 4abe6dc48d | |||
| 26f36f2ff9 | |||
| b42c11de82 | |||
| a494c75bfe | |||
| bc8c912c7a | |||
| 5247237bf0 | |||
| 8e6aae0d7a | |||
| 8681686d8f | |||
| 9012e6dd9b | |||
| 139bd30247 | |||
| 4930516652 | |||
| ad568edd7f | |||
| 62b3a20da5 | |||
| 1f41d06f9a | |||
| 1339c7f30c | |||
| 734fb18c4d | |||
| 71c6339af0 | |||
| 34a0897f90 | |||
| a32100d9ba | |||
| b5400902a7 | |||
| d35fbe6bfc | |||
| 645930387c | |||
| 44dfe62af0 | |||
| 4acb267cf4 | |||
| 8676d19dc3 | |||
| 6b4f3e5fab | |||
| dafc4e456d | |||
| 2a884cc179 | |||
| b774753922 | |||
| 5e03305da5 | |||
| 3ff1ec99ac | |||
| bc8d494617 | |||
| 35e771daab | |||
| 1ab259df9a | |||
| 0bd54a60e9 | |||
| 63b6b24cd0 | |||
| 8e49560709 | |||
| b8f57c9a96 | |||
| c6a15e1835 | |||
| 05f4416ba0 | |||
| 91ac790249 | |||
| 3cfe737581 | |||
| a2b3acfc6e | |||
| f4b23de9dd | |||
| a08c98cdfb | |||
| 279e2be56b | |||
| cdbb893b27 | |||
| 41d172f22d | |||
| 3f7ec99166 | |||
| 1102b7eff8 | |||
| da43778c1f | |||
| 9aa5ab7557 | |||
| d44c58a1fb | |||
| 7463e9b8bb | |||
| 9f201a8ebe | |||
| 91998d6424 | |||
| 420db56ffd | |||
| 07ed5e5346 | |||
| 4824ce300f | |||
| 5855eba9f3 | |||
| cdaf4c7321 | |||
| eace6adb6d | |||
| 6db9fcbc30 | |||
| b4b35cff15 | |||
| 47fb79bd8c | |||
| 6b50410336 | |||
| 6e3700593f | |||
| 6a8c5a1d27 | |||
| 50da209dc4 | |||
| b7b08fa0c3 | |||
| b1288dad40 | |||
| ac6fb17784 | |||
| 3f22f4727c | |||
| 0153e796cc | |||
| 4891818114 | |||
| db4a557187 | |||
| 04c3a534af | |||
| a32143003d | |||
| 8bd17163c7 | |||
| 81aaca8c04 | |||
| 189e1b228d | |||
| 52709c7771 | |||
| 7dbb125a34 | |||
| aff375808d | |||
| 3f5f09e730 | |||
| 332ff4bf1c | |||
| 253a4c113e | |||
| 1788c93aea | |||
| c2ecbc76ce | |||
| 4203c903f8 | |||
| 727feda935 | |||
| 39087fecdc | |||
| c725771e11 | |||
| 103212f209 | |||
| be326ff752 | |||
| ebaee3ea21 | |||
| 423bc515e5 | |||
| 941d89e06d | |||
| 281b790618 | |||
| f5d7c5f103 | |||
| 93c4bca04d | |||
| 726f0adde3 | |||
| 24d7eb0ca5 | |||
| 8ba44e7d55 | |||
| 49f539cb46 | |||
| 0a650b573b | |||
| 7e3c53b9d3 | |||
| ab912329b6 | |||
| 6ad216576d | |||
| c5a4df30c6 | |||
| cb8f1dca3b | |||
| d1ff87d94d | |||
| b16f76eede | |||
| 4792837f2e | |||
| 416830fb59 | |||
| 07b1bc2e88 | |||
| d16b3b2487 | |||
| 74d0b4cc78 | |||
| 8832a9dbd5 | |||
| c55613ccf7 | |||
| 6c399aa62a | |||
| 5862b50217 | |||
| 5696629b13 | |||
| 3a15152be5 | |||
| f1055a7e91 | |||
| d4dc6b509b | |||
| 1192779d05 | |||
| 71a1150c95 | |||
| e14ca39fd7 | |||
| 421ba8730a | |||
| 0d56524b72 | |||
| 9fded9ca53 | |||
| 441624a5fb | |||
| 25ac77e010 | |||
| 77d887898d | |||
| a32f2cd24a | |||
| eb68de9268 | |||
| ed3810f7a5 | |||
| 1c3e287d32 | |||
| 5044af69d1 | |||
| bc6544c0bc | |||
| a462f49b99 | |||
| 7622fbf895 | |||
| 3f371fe2dd | |||
| 6ca205a029 | |||
| aba2167d9c | |||
| 5db4f1f7df | |||
| a0a8d2160d | |||
| 2a5da50902 | |||
| 51758b2dd6 | |||
| 90b144cf0a | |||
| 529bd0425e | |||
| 6c36cd5d6e | |||
| c102720af4 | |||
| c63a6c472d | |||
| 5b3f721efa | |||
| e5bea2bde4 | |||
| e9d64e0a8c | |||
| f0df78b7e7 | |||
| e93d976d00 | |||
| 64aad34cb4 | |||
| 7592d58f0c | |||
| 08906ddd4b | |||
| d21761c0fd | |||
| 277ead95d9 | |||
| 5d14cc68b7 | |||
| ce80e0dc57 | |||
| 4c74e6d89d | |||
| 45d04175d4 | |||
| 687c9b7b29 | |||
| d9f66413ee | |||
| 1b1bbe4262 | |||
| 3edf7c25d4 | |||
| 54531f8e3b | |||
| b5a68f235a | |||
| 5bf64e7dfe | |||
| 51b897b672 | |||
| da4ac6b7ef | |||
| 1ba0984203 | |||
| f5e852cdf0 | |||
| ef2677b0a6 | |||
| 2aad039b4f | |||
| 8a90948a1c | |||
| 63bff33e85 | |||
| 21133a2091 | |||
| 39f995e319 | |||
| 91624e87ab | |||
| 11d492b0b9 | |||
| 17f2b56291 | |||
| d1d8ac57f3 | |||
| b418eb112c | |||
| ee9137f176 | |||
| 903bf0147e | |||
| 540aa13300 | |||
| bbbf1168dc | |||
| 18fbb72f7d | |||
| 0b5fd4f6bb | |||
| 87360c2ae5 | |||
| 17b98dd005 | |||
| 1293e99ad8 | |||
| 028d4d83d3 | |||
| be670e168a | |||
| b619477be9 | |||
| 778faddbd8 | |||
| f644634aa6 | |||
| 547f4c2c5a | |||
| 5462a6be6e | |||
| a60496f9df | |||
| acd8d3a617 | |||
| 0c92305a4b | |||
| 1ecfff0fa0 | |||
| d933034ad4 | |||
| a6fadfe1c2 | |||
| aae317c017 | |||
| 37683aca56 | |||
| 22f8fb4d5c | |||
| 7d1589d407 | |||
| b35018d955 | |||
| c29a3aa0a0 | |||
| eae5fd81e5 | |||
| 23edec83fb | |||
| e4cd430710 | |||
| 35b2cff295 | |||
| 8e5f37f87c | |||
| cee8c86b6e | |||
| c0a84dcc85 | |||
| e80b443cd9 | |||
| e309a06b47 | |||
| 86dad3aebe | |||
| 5a9fe9dedb | |||
| c7d24c0fb3 | |||
| 5d292826b2 | |||
| f378f02954 | |||
| ba575fd4ad | |||
| dd14cf6a9c | |||
| 2c81cb2f76 | |||
| 6eb34716b8 | |||
| 871bc98933 | |||
| ec7fc5adca | |||
| c0c3f4cbac | |||
| 02143cd0e2 | |||
| 1c5dcbcac8 | |||
| d24d8f2abe | |||
| 584ea43b2f | |||
| 292f62c5cc | |||
| 762481411d | |||
| d816442e4d | |||
| ea5ca16036 | |||
| 2991717191 | |||
| 0fa43e3aac | |||
| 085fb78e85 | |||
| a565aa6db9 | |||
| 2763f988da | |||
| 82ac7ea236 | |||
| f89dee4f3e | |||
| 70779d4e66 | |||
| c0ecf08ca0 | |||
| 35f43cc429 | |||
| cfafd0493c | |||
| 9119692bb8 | |||
| 8b0aa6a64d | |||
| ec26541771 | |||
| 1c825dd509 | |||
| d9eff7daeb | |||
| 3419e64dcf | |||
| 1e2ceca4df | |||
| 347246901e | |||
| 68bd156a71 | |||
| 9821fae59d | |||
| 43b2bb2c25 | |||
| a3ebafbdeb | |||
| 5cd07006f6 | |||
| 3d350a002e | |||
| f18b8cd569 | |||
| 5389d9e0d4 | |||
| 07c795408d | |||
| 5f50e7bafe | |||
| e92716a1b6 | |||
| bbcc1752b1 | |||
| 40ae06091d | |||
| d480e2e51b | |||
| 839321642e | |||
| 9f88a65873 | |||
| b77330bafc | |||
| 4a4b2a0ce6 | |||
| 0ae126d3b8 | |||
| af9332dfaf | |||
| 4103567776 | |||
| 8f0edf6a1c | |||
| ab5279f4ad | |||
| 1113c9ab10 | |||
| bf5d7c0c10 | |||
| fef8d4c990 | |||
| 5a0c85b3ef | |||
| 9fd877acc9 | |||
| ed3b56d763 | |||
| 2f79b1b087 | |||
| 7208f63221 | |||
| 572d4f4491 | |||
| 0cd4396180 | |||
| aafb7567c1 | |||
| 468431cd8f | |||
| cf5db9b94f | |||
| 589b6c15f0 | |||
| 2af0813634 | |||
| 15d2a5faf8 | |||
| b1d28d5b4a | |||
| a6fbd8287c | |||
| b0b77b3047 | |||
| 96844b0ca5 | |||
| 7fc6146973 | |||
| 6cf0910842 | |||
| 692c536ac5 | |||
| 1646a21197 | |||
| 4557971481 | |||
| 8baf46c0a8 | |||
| 0c0083ba72 | |||
| d18362c726 | |||
| b403d37267 | |||
| 2d189e24ee | |||
| c8c29b0f1a | |||
| 784dd55d88 | |||
| adb916ce82 | |||
| 15cfafb360 | |||
| d43cb4fe7c | |||
| 38c9c20a35 | |||
| 935acc500f | |||
| c189f31f23 | |||
| cbf43a54fb | |||
| 2918071a3e | |||
| 1829eba584 | |||
| 3a64607d94 | |||
| 64649a1207 | |||
| 4dfbbabf8b | |||
| 72f1ce6550 | |||
| c34c4b50d0 | |||
| 4521d66103 | |||
| 0c238c9e72 | |||
| 3d9f27b877 | |||
| 50e66a2e53 | |||
| a682f02f59 | |||
| df24bd295d | |||
| 04ebedb6f0 | |||
| 71ae1cac5f | |||
| e0b21dccf1 | |||
| bfeeb0ad70 | |||
| 2273af0166 | |||
| 604192337f | |||
| ccfc34b13f | |||
| 3f4abcb228 | |||
| f2ccce23f3 | |||
| 7d96ef2671 | |||
| 2882725927 | |||
| 062cee2933 | |||
| fcf9f117b0 | |||
| 0f1cb5ad7c | |||
| 0ab4872032 | |||
| a6274647a4 | |||
| 65f71ce2eb | |||
| 6aefb8e86f | |||
| 4aef9b1c93 | |||
| 0ac67ee8d9 | |||
| 66b3155a48 | |||
| b3adffe437 | |||
| ec92f3fefa | |||
| fb85a83108 | |||
| 41ef2bfb16 | |||
| 1ae801554d | |||
| 65998d8076 | |||
| 45a7b71cac | |||
| e1e9261450 | |||
| 819e5b19b3 | |||
| a3cb2020bc | |||
| a6017ac550 | |||
| 7faeb82585 | |||
| 48e50a7674 | |||
| dc9d775f88 | |||
| 66cbb7b911 | |||
| 9ca3a3139a | |||
| 6fb465cb4e | |||
| 8ae03fcd6e | |||
| 122250b554 | |||
| eb8883160d | |||
| 8af4fe9ed3 | |||
| adb5971494 | |||
| 377fdf5ae5 | |||
| e1272d73fe | |||
| e3e61078a5 | |||
| 13823f117b | |||
| 2e15582799 | |||
| 65bdb3a544 | |||
| ac91c45b0f | |||
| 6481cfd048 | |||
| 0f6b01a531 | |||
| 364b3f181d | |||
| 04849f26b2 | |||
| 26a7647e0e | |||
| 8eb036fb3c | |||
| fce8349c99 | |||
| 5ea912e778 | |||
| c2c8da2517 | |||
| 1be40554af | |||
| d717de5719 | |||
| bcc19a622d | |||
| fb3fc5322c | |||
| 2eff70fbf4 | |||
| d885242b61 | |||
| 46d988e2cb | |||
| 7a5122121b | |||
| 7fc14504b1 | |||
| cb1a1e9a51 | |||
| 619e038de9 | |||
| 4154bd0667 | |||
| 5408949951 | |||
| 866191478f | |||
| c9060b053d | |||
| 9dc844a6e1 | |||
| e2a9cff3dc | |||
| c3b83b8354 | |||
| e24382691f | |||
| 9074b3e980 | |||
| 65d66a05dd | |||
| 7de7891c4d | |||
| 4bec43cf79 | |||
| 6cc0107693 | |||
| ca7f964104 | |||
| b26fc6f31b | |||
| 5fa624f35e | |||
| 7955469f7c | |||
| ce62fc8371 | |||
| 2e40b7f113 | |||
| 2ac62bccec | |||
| 22eb91a7e0 | |||
| 448e3a7e58 | |||
| bf54a370e5 | |||
| 337452b4c0 | |||
| 5bb098ba5d | |||
| 4159842195 | |||
| b69b1eae8f | |||
| c4a6e1fd4d | |||
| 3456d28cc2 | |||
| f376bfe95d | |||
| a72074b33f | |||
| 5185544864 | |||
| 7834a465d5 | |||
| 1f0bfc8d83 | |||
| a122f0f248 | |||
| a31fb88fd0 | |||
| 70fb1cd603 | |||
| d2c48b898c | |||
| 2d837efba7 | |||
| 1852d0b9b8 | |||
| 837e41f9a7 | |||
| 1fd45a1b85 | |||
| 2cd7e17b65 | |||
| 8eb4f72531 | |||
| 2619099fe5 | |||
| 2959286eb5 | |||
| a4b191a7e0 | |||
| 3929e26276 | |||
| a1d5565e65 | |||
| 9638e34ab0 | |||
| ed9d4c0b2b | |||
| bb64db98d8 | |||
| 88115811a9 | |||
| 67fa8a2f47 | |||
| 560eba91e5 | |||
| 42561e1233 | |||
| 7e2c8cc9f4 | |||
| eea05f5078 | |||
| 7eaec9dd22 | |||
| e1305e8d05 | |||
| 2dd3408caa | |||
| 3651831e7e | |||
| a1d752bfc0 | |||
| d10832074e | |||
| a001ab3a44 | |||
| ef570e4e13 | |||
| 7831aae5dd | |||
| 1336da9450 | |||
| 6432f02996 | |||
| 1fb8d60fd2 | |||
| 5e92bf8e41 | |||
| 7e5d012f75 | |||
| 925ff6241f | |||
| e14c3cff85 | |||
| 984e42b0bc | |||
| 0702685e7e | |||
| 7deb8f568f | |||
| d8434e6875 | |||
| b9a1039566 | |||
| 03130548ea | |||
| 11c5a6bb4d | |||
| 8dc332721f | |||
| 189f647264 | |||
| 56baf4ed87 | |||
| 7ffb103758 | |||
| 9fcf015214 | |||
| 27bfab4e6e | |||
| ffccfa2ee9 | |||
| d3dcef4b8b | |||
| 3cd36ebc7b | |||
| 2ebdc04787 | |||
| cc8add9f66 | |||
| c2c539e3cc | |||
| 0cdff46725 | |||
| d0d115321d | |||
| 99683e958a | |||
| f572ae3474 | |||
| e65ad44b32 | |||
| 3d8614cb47 | |||
| d09cc0f30c | |||
| 80c82e10aa | |||
| 2fb652ce09 | |||
| 1bab920cf5 | |||
| bb71cb200e | |||
| 48afe1586a | |||
| e7e814fa8c | |||
| 496eed950f | |||
| f0d29cd33c | |||
| 75bb6aa9a1 | |||
| 44b2f9637a | |||
| 0643165519 | |||
| cde18648dc | |||
| 04e01e2b31 | |||
| 348b7383f9 | |||
| b2b7193374 | |||
| 47a59d72c9 | |||
| 163770f99a | |||
| 4c1e064a2b | |||
| 88555948d0 | |||
| 1893b37e23 | |||
| c3cbd302cb | |||
| 2ffa7ac0da | |||
| 885a446b10 | |||
| 8bfe620fcc | |||
| 129319b0bc | |||
| 7ce83f2a95 | |||
| 4b05765174 | |||
| 3390da6beb | |||
| 50707a2741 | |||
| 0614c40b42 | |||
| facd07a732 | |||
| 26ac81df3e | |||
| 1ec10ca565 | |||
| 26a39c733b | |||
| 15f0e2e7cb | |||
| d4640f4647 | |||
| 4c9364a803 | |||
| e951edeed3 | |||
| e7a787aa41 | |||
| 500e24d6cd | |||
| 759cbd7486 | |||
| 57545653b1 | |||
| 3c5377ca1b | |||
| a89868928b | |||
| 5bf3991f55 | |||
| 08696d92ea | |||
| 2c2466fb25 | |||
| fc3e393516 | |||
| ebaf8cc06c | |||
| 0862d69a6e | |||
| b8106e4ba4 | |||
| 6ce2f1316f | |||
| 294348939b | |||
| e2db95184f | |||
| f8597fc150 | |||
| 85b0b0cd77 | |||
| 2bd72af2ef | |||
| f729202272 | |||
| 47f30a04d2 | |||
| a42b35598e | |||
| db53f4533e | |||
| 71e33265f5 | |||
| a016f6022c | |||
| 3e3b53f815 | |||
| e2bfe0ce76 | |||
| 98c33c605d | |||
| b3269b08a1 | |||
| e59cff47d4 | |||
| 47912431e6 | |||
| 0ef803950b | |||
| 3c23a44786 | |||
| 701d0d9905 | |||
| d4e1d89901 | |||
| 14aa9eaadd | |||
| 1feabf4275 | |||
| 65c173f2a3 | |||
| 697acf7f6a | |||
| ed69bcae2d | |||
| 728545468c | |||
| 33b1e76e48 | |||
| 1d134025a5 | |||
| 96a9cfab36 | |||
| c9e10e1d0b | |||
| 60846b2b7a | |||
| 6fe7886e2b | |||
| 199c2d2fd0 | |||
| bd54ba911d | |||
| a9354fc743 | |||
| b4b69ae484 | |||
| 1c7b71bf9e | |||
| 1377f0147e | |||
| d2b1e38207 | |||
| 9ffb67478f | |||
| 1dee848d3e | |||
| afe1c70f2d | |||
| 1b8fba8e26 | |||
| 45fbb67aba | |||
| 914005174f | |||
| 2a82467a6f | |||
| a233232b7d | |||
| ed4bf13960 | |||
| 3135063100 | |||
| 87ef6a9cc1 | |||
| ff31f90b7e | |||
| 2b2ba534e2 | |||
| 3635b3dee7 | |||
| fa613e393f | |||
| 7d3dbcb027 | |||
| db70676933 | |||
| b3b235ebc0 | |||
| 70492c2127 | |||
| f562264674 | |||
| 0a88f84847 | |||
| 139c443770 | |||
| a630ad73cb | |||
| 366e8217c2 | |||
| 01226cb8ac | |||
| 3891b72f33 | |||
| a80fcacd90 | |||
| c54ccaac31 | |||
| 79731cb0ff | |||
| 16b5fd4bf2 | |||
| fb9463c55f | |||
| 2920a8e0ec | |||
| 6a4c3b61e6 | |||
| 6360c3bf46 | |||
| e313603aac | |||
| 0f3de805f4 | |||
| 3bf22d0024 | |||
| 66567933d7 | |||
| 56e19e3494 | |||
| d2f5a10f5d | |||
| 32bb4fa950 | |||
| 8ae06b4648 | |||
| 652396678d | |||
| 59b870a87a | |||
| 0f067fd0a6 | |||
| b91f173680 | |||
| a311d1bdc0 | |||
| eb7add1828 | |||
| b152b8cbcd | |||
| 2336b0706d | |||
| 300b57dd70 | |||
| 0769bf416f | |||
| 90f2e1f8b5 | |||
| 2a4926f417 | |||
| 5e1c9099e8 | |||
| 397e9bc8d6 | |||
| 2366f2cb2e | |||
| 35f1a90df7 | |||
| 9faefa0c96 | |||
| b5adffd5c2 | |||
| 82010bf5c1 | |||
| 6234f01a6d | |||
| 44c2519d75 | |||
| ef94275eb6 | |||
| a6ca48a1c2 | |||
| 24a66a44bf | |||
| 2411b825b4 | |||
| dd7b9000ad | |||
| f1328c7395 | |||
| 9d06e58c3c | |||
| 711b136191 | |||
| a04f9e7a59 | |||
| db5b22e895 | |||
| 0d52c37e11 | |||
| 734e309b3e | |||
| 3efc645975 | |||
| 572812217b | |||
| 008507327c | |||
| f4a6c3e7ea | |||
| ff41fbc5c1 | |||
| cac864192c | |||
| b19683eb41 | |||
| 7a46d7efde | |||
| d09bd6f862 | |||
| 7676be5570 | |||
| ed9524e125 | |||
| 533bb035cf | |||
| dda2e9374b | |||
| 14754deb21 | |||
| b0dc474160 | |||
| adf89bbb33 | |||
| c103b63fe1 | |||
| a3d0882317 | |||
| 583bd1a6e2 | |||
| 53eda42da7 | |||
| c26c4a61c6 | |||
| ea2527c2d1 | |||
| 65fcf22670 | |||
| d0572538ff | |||
| 3577265508 | |||
| f28e191d70 | |||
| 38b6c44b4c | |||
| 1a24e316d5 | |||
| 9da9e8244b | |||
| 8d51ef0f35 | |||
| 9cfae823a7 | |||
| d6151eae23 | |||
| 4a0a3281c6 | |||
| 1230075011 | |||
| 7459954377 | |||
| cf38098ba8 | |||
| 8ca394efaf | |||
| f9c7931800 | |||
| 24547f40ff | |||
| 72debee125 | |||
| a3d6994afa | |||
| 08c270f65a | |||
| e585453c2e | |||
| 86f4524010 | |||
| 6d098cc230 | |||
| 434c96e625 | |||
| d6e9616256 | |||
| e9187ae38c | |||
| 978dc76653 | |||
| b409fd32ac | |||
| a2ad997e97 | |||
| e9428726ca | |||
| 400906b433 | |||
| 50d7c61c01 | |||
| d9bf522b27 | |||
| 93dc0679ec | |||
| 1ddc51cfd1 | |||
| 50bf0e10f5 | |||
| d1ccb7e47f | |||
| 6a22c5b2b5 | |||
| 7e845a3b87 | |||
| ea1c970190 | |||
| 3d207ccf11 | |||
| aaf7f5ae02 | |||
| 50e8ad285b | |||
| bb5462e327 | |||
| df2e7fa6eb | |||
| 11f36bdf9a | |||
| bc0d4585b3 | |||
| 70917f291c | |||
| 9f6f0c7424 | |||
| 4722a8df3c | |||
| 7be8a71c60 | |||
| ce859edba8 | |||
| 90820cd044 | |||
| c929f1b62f | |||
| ff88132620 | |||
| 89f8d4ae12 | |||
| 3f4ffe7844 | |||
| ed567a2dd1 | |||
| 22a00036e2 | |||
| 2aac0a5a26 | |||
| 07ad6a437e | |||
| d9a444ca1a | |||
| 766f58ed7e | |||
| 1099b4c881 | |||
| 7c0f49fc36 | |||
| 761e796201 | |||
| 8d0fbc6a1e | |||
| affcea8822 | |||
| 53f0be00a9 | |||
| 602caa9cd6 | |||
| f35ec8c955 | |||
| 1316491e50 | |||
| 2cc4309bf8 | |||
| 612d43f284 | |||
| 56824d1769 | |||
| 2238ac7d59 | |||
| 2b5b192cd7 | |||
| 64e1d23cba | |||
| 40d2904f3d | |||
| 9712cc479b | |||
| 0ef1fb0b67 | |||
| 2b67bc448d | |||
| a3f81b79ed | |||
| 8d0dae4cec | |||
| dda96264df | |||
| 4af8e5c5c4 | |||
| eff5605be5 | |||
| 2b53ab23c0 | |||
| 2010de9104 | |||
| 280a99b3b6 | |||
| 646025589b | |||
| 4502003e61 | |||
| 07abb6240e | |||
| dea0815199 | |||
| e4ed2d2e42 | |||
| c8228e5789 | |||
| 667e5e4f89 | |||
| ada703a73a | |||
| 2b8094f915 | |||
| 9e1e960116 | |||
| 82ae9ef541 | |||
| 1b07274e34 | |||
| 15ac54d5d6 | |||
| 1cdd8510fd | |||
| 039f3d01a0 | |||
| f63feea05e | |||
| fbac38d98b | |||
| ca2ab3387f | |||
| 64ded50bbf | |||
| d44ac61033 | |||
| 18ada77d8a | |||
| 6d0f1275c2 | |||
| a848eccfc6 | |||
| 2402fa4824 | |||
| 2bf4135afc | |||
| c677f132cd | |||
| ec4015d73c | |||
| 8c42dbf71c | |||
| 4ca679180d | |||
| 64da959619 | |||
| 88123e2512 | |||
| 87e98e8788 | |||
| 767857c516 | |||
| 2b60166e5c | |||
| 2e41db39f5 | |||
| a55fa8389e | |||
| d23142027f | |||
| 438fe3f9db | |||
| 5cc154147f | |||
| 9cd5a0a1e6 | |||
| 2958142e31 | |||
| 27c15bed60 | |||
| 2985739b8c | |||
| 2465d93330 | |||
| 556d211136 | |||
| 0ee2a58cdc | |||
| 7daf84fb44 | |||
| 9aff01c9a9 | |||
| a2b84e9897 | |||
| 66f3c2673c | |||
| 554d08c3a1 | |||
| 3a595ea5c4 | |||
| 03c9648f2e | |||
| e3a55af336 | |||
| be4a432bea | |||
| 0d32a24cba | |||
| e36948cfbc | |||
| 9aa647068b | |||
| 092030c1ca | |||
| 0bd261ded4 | |||
| 7ed557497d | |||
| d793ec2ffe | |||
| 271f7df343 | |||
| 08d44f588f | |||
| 84b4a5a495 | |||
| b2e20a82ba | |||
| 1b3a06a02a | |||
| 91a5e75151 | |||
| 4754b0e253 | |||
| 84b517f5a0 | |||
| 8ac88cf069 | |||
| 13a995cc1d | |||
| a93fb52632 | |||
| 4d927e73f1 | |||
| 38228709c8 | |||
| 38788a3161 | |||
| 46b6973c05 | |||
| af2b7708a8 | |||
| a233982931 | |||
| ed1439fbc6 | |||
| b379b67a32 | |||
| f3945fbddb | |||
| 8b44ee2ce1 | |||
| 7b582b71ba | |||
| 4e8c507276 | |||
| 9390c56831 | |||
| abebbf04b1 | |||
| 5e434073d4 | |||
| 3e4a566e46 | |||
| 6f5cf8c15f | |||
| b687bc807a | |||
| 5440fd6cb4 | |||
| b060151625 | |||
| be38d4ea93 | |||
| 56f21c4fd5 | |||
| 386df457a9 | |||
| e4abf6e723 | |||
| 1339ebaa84 | |||
| c9b90884da | |||
| 9b2b2c88df | |||
| e5bdab0355 | |||
| 4d46958c82 | |||
| 9dd8e4df7f | |||
| d98e07c3d3 | |||
| 78bc11465b | |||
| 8e8e4bbabc | |||
| 90671233c6 | |||
| 12bae51384 | |||
| 7b848e215f | |||
| 21872e2fb0 | |||
| 2a218b96c4 | |||
| e76cda6923 | |||
| 9d16612a9b | |||
| 63c865ace6 | |||
| 0a62de0d4f | |||
| 593996216f | |||
| 4eae23a2cc | |||
| f3699b5ac8 | |||
| 9ce0e51305 | |||
| 4c79318694 | |||
| f7ac724c5d | |||
| ee9fe1239a | |||
| b6b5c27cec | |||
| 632e07b749 | |||
| 48cd2d190f | |||
| 685797f403 | |||
| ef6f421f89 | |||
| e0ffd3e8a5 | |||
| 0d16b5fc38 | |||
| ac8a27cba9 | |||
| e5f2a8ebf2 | |||
| 54733eba6f | |||
| 93353aea70 | |||
| e16cb8b4a2 | |||
| 5bf3c1df24 | |||
| 419918076e | |||
| fb768c420d | |||
| eb067fee55 | |||
| 6390b50d6e | |||
| 5d8134ed32 | |||
| 852904e1a4 | |||
| a120adde63 | |||
| a80af177b6 | |||
| 8db7d435b9 | |||
| 901e0ddfe4 | |||
| ecb30409f6 | |||
| 44c2c77548 | |||
| 5be5efdacf | |||
| 057c3da82a | |||
| 8fa429caef | |||
| 87f526d82f | |||
| 1ae2320e09 | |||
| 54693cf7b1 | |||
| a082375d57 | |||
| 9c7adb7248 | |||
| ccebbbc0ac | |||
| ebb6915e58 | |||
| 5cc27fd3b5 | |||
| a332509e02 | |||
| 375fe81311 | |||
| d354ad1c34 | |||
| 119d8b3aca | |||
| cd9edba26f | |||
| 8f1c502d2b | |||
| 92312fbc0c | |||
| 2efcaa9e8e | |||
| aa53541235 | |||
| b863c25d21 | |||
| f77c3574af | |||
| f5105bac65 | |||
| 5a86592e93 | |||
| 2c83cfc14c | |||
| 863546e125 | |||
| 9198e30688 | |||
| 1890157faa | |||
| 0898f372b1 | |||
| b9e5256cf9 | |||
| 6b6d89b911 | |||
| d3e0193c8a | |||
| 31881209d6 | |||
| 5c7e893393 | |||
| 3c814ebf87 | |||
| 5cf1dca36e | |||
| 12338c1dc4 | |||
| 5b23752205 | |||
| 4e53f301d8 | |||
| 2b7803dbac | |||
| 446a278f3d | |||
| 0047d3f81a | |||
| 4e1f17d65b | |||
| f9b1dbe2ac | |||
| a251474144 | |||
| 8d88bb06b2 | |||
| 9f2ff3ed22 | |||
| 45cbf70265 | |||
| 59e16b88ae | |||
| 33f219dfe6 | |||
| 44db2eea70 | |||
| 12ab54648c | |||
| 19926e2979 | |||
| 1620a1e014 | |||
| 888546b6f5 | |||
| 3215db26aa | |||
| 76aff84788 | |||
| fc28ba3156 | |||
| effce0573b | |||
| e8db363431 | |||
| 9603b6877d | |||
| d5ecb5ff5a | |||
| dae73938e8 | |||
| 619b6dfae3 | |||
| 8f9c36b730 | |||
| 3eeec4faae | |||
| 972a4b95b6 | |||
| ffee1a4126 | |||
| 96e23c2ff6 | |||
| 08356007c9 | |||
| 5064b6f747 | |||
| 57d3002ee1 | |||
| a00a0dbfcd | |||
| b41d2c5c14 | |||
| 1da48beeec | |||
| b57ff73086 | |||
| 67978b5746 | |||
| 062f305d1a | |||
| 1f70d4e2a5 | |||
| aa5bc20c83 | |||
| 09af10f635 | |||
| 4d7953aa56 | |||
| 46e1560678 | |||
| 4d0148b417 | |||
| 1605d1d24d | |||
| 5190043e56 | |||
| f66a2ffa1e | |||
| d1e76a34a0 | |||
| 5b3d5f9f3c | |||
| 415a42f327 | |||
| 09f6e55338 | |||
| a351e05a17 | |||
| d37bcbdc92 | |||
| 437af37b13 | |||
| c62367612d | |||
| a92cba8484 | |||
| 5f7d922b10 | |||
| dc29632d4e | |||
| f4a7754cc0 | |||
| 6811d3d80b | |||
| 245f6273bd | |||
| 870c8d3c4e | |||
| d573472a86 | |||
| 441b6dbda0 | |||
| a7e6a1059c | |||
| 85719a0a5d | |||
| b5b52afd35 | |||
| dc35633aa4 | |||
| 9e3ba487fa | |||
| aee8425b68 | |||
| 94229bb262 | |||
| 1f6444fbea | |||
| 439ef6447d | |||
| 9cab808c5d | |||
| d41fcf7345 | |||
| ac51ba38f9 | |||
| ef85b24a78 | |||
| d88730685e | |||
| 16661847a7 | |||
| 7f782a1a24 | |||
| 9188ce68aa | |||
| 2b79a6ff8f | |||
| 70b0274c8e | |||
| 83ce1de8e7 | |||
| b1d484f827 | |||
| 123519165d | |||
| 30ff9c6775 | |||
| 906f5f7e96 | |||
| 9238316cf1 | |||
| 5e89b9a455 | |||
| e3a4ff33d2 | |||
| 5d9ea394ba | |||
| abb5c9fd92 | |||
| 266835cd2e | |||
| 92af03579c | |||
| 2e20f2b89f | |||
| e67593673f | |||
| d1b533d399 | |||
| 2647902fee | |||
| 574320ec3f | |||
| 3f2377017c | |||
| 2b5bb02817 | |||
| 302d14adef | |||
| b796ededae | |||
| f811ba8777 | |||
| eb7b45d26b | |||
| 17b2d92a3d | |||
| 22f0bcaf8f | |||
| e61277002c | |||
| ca5258f140 | |||
| 8f4473b3e3 | |||
| 51e8af9e5f | |||
| 82818e7324 | |||
| 6ae8103022 | |||
| eca2d92791 | |||
| 6778e19710 | |||
| 25f25275cd | |||
| 7e746ad2a6 | |||
| 6ffa2b01e1 | |||
| ed6ca0d7fa | |||
| 8e2a3c1d2f | |||
| 7f6dcc2745 | |||
| 76e34d6f2c | |||
| e613b17c05 | |||
| 5ba9a089e1 | |||
| f2d5d6d24e | |||
| a1ec4ea3a9 | |||
| 0fe7420638 | |||
| b304730225 | |||
| 4801526f35 | |||
| f1857030b5 | |||
| b5b3f4e1c6 | |||
| 172bb7887c | |||
| 60228d30d1 | |||
| 9592537840 | |||
| 8768a80242 | |||
| d5f73f89d8 | |||
| 19bbe6c67d | |||
| aebb65e983 | |||
| 5e327af327 | |||
| bc1f8813c2 | |||
| 80d9f624d0 | |||
| e9c46f38fc | |||
| ecfbaa267d | |||
| c25fd3349b | |||
| c3e27bcf87 | |||
| a5894f3e6b | |||
| 4db3a388dd | |||
| 6fd1a2ef52 | |||
| fe2a259eb1 | |||
| 1ba7c2f276 | |||
| 649f747f8a | |||
| 05dbaf7672 | |||
| 742a08229e | |||
| 5602d2c7bb | |||
| f3e0479a8f | |||
| a1143c4ea0 | |||
| 0314d1dd01 | |||
| 751b3f502d | |||
| bf5e09d5ab | |||
| 14e4a10312 | |||
| 53b89e1ee4 | |||
| 370f878a36 | |||
| 6e4d61c1fd | |||
| f6fe5c07f6 | |||
| c685293297 | |||
| 0e6a2c0491 | |||
| c2209ad5e4 | |||
| 5f249a3e67 | |||
| 76fb3652fc | |||
| 92925e846d | |||
| 43c04c29ce | |||
| 06b74c87da | |||
| 7f075b0b15 | |||
| c5a86c22a4 | |||
| 3621c6234c | |||
| 80c2fefc43 | |||
| a47952146a | |||
| a55950865e | |||
| ab181ac329 | |||
| db172d1053 | |||
| dad26339a9 | |||
| 3b0ed61826 | |||
| 9699e2b483 | |||
| 3048188b5b | |||
| 547bb3b053 | |||
| 7c2087cf05 | |||
| 217fea9667 | |||
| 523c3cd50b | |||
| 631126c77a | |||
| 655d381ef3 | |||
| c68fec7e97 | |||
| cdfa8a668b | |||
| 92651d228d | |||
| 90e3692603 | |||
| 3a8316ab93 | |||
| 8d3f0d0eab | |||
| ded0efb955 | |||
| cdd4354256 | |||
| bde955fc56 | |||
| abc9e98625 | |||
| d086ee14dd | |||
| f43fec7ee6 | |||
| 6385511e88 | |||
| ed4becf007 | |||
| f11ab4c569 | |||
| b12c2c76d1 | |||
| f5589445b9 | |||
| 032a61b197 | |||
| 011ed380aa | |||
| 88a18c8b6a | |||
| 51e65db715 | |||
| cc02fcd889 | |||
| 9f17c62533 | |||
| 36bd2a65e3 | |||
| 85d4e56bb1 | |||
| 381d9bafdf | |||
| ada16fd188 | |||
| c408157a4d | |||
| 6e299b582a | |||
| 9777fbacf6 | |||
| 643b4dd126 | |||
| 19ac54277b | |||
| c78a8dfd2d | |||
| b1a57c4cb2 | |||
| 54c180092d | |||
| be110d0464 | |||
| 6e733b65ae | |||
| f948a2d0c0 | |||
| 2aa47554fc | |||
| c80b270678 | |||
| dcc32e84b9 | |||
| 947d610309 | |||
| 3e01a387ba | |||
| 4dbba5ac98 | |||
| 374cc4183e | |||
| b9d9730b62 | |||
| 0a178a687a | |||
| c4c43c3d26 | |||
| 8fa86d4d34 | |||
| a22dd28e02 | |||
| 0e274fc4be | |||
| bc3da25a4a | |||
| 80492d663e | |||
| 172c539a5a | |||
| a079acc0d9 | |||
| 30df77fa4c | |||
| 71a22e45b0 | |||
| defd8a527b | |||
| 3ce29236f3 | |||
| 04ee99f1a3 | |||
| 91817bffe1 | |||
| c39b83667b | |||
| 1bd382c1d0 | |||
| 49bef1cdbf | |||
| 249c508126 | |||
| 04b27525fa | |||
| 175bcb1734 | |||
| fec2c7e715 | |||
| 07dca8cc03 | |||
| 60c093f086 | |||
| db4ab1c936 | |||
| f071207463 | |||
| 622b9d9276 | |||
| 05bfdeab7a | |||
| b4bb98ea60 | |||
| 792b7e0629 | |||
| a079c2eb7c | |||
| f7aa91e660 | |||
| 6d677bbd63 | |||
| cee3ec6dc4 | |||
| 3da17c42a4 | |||
| 2329cbc1a2 | |||
| b39ee54103 | |||
| 299f9837b7 | |||
| 6a889ed3e5 | |||
| f3ba88c87c | |||
| 498f1525b2 | |||
| e7c5120682 | |||
| 19cfaa2be7 | |||
| e97c7e042b | |||
| d8b1fc45aa | |||
| 35f27d0d3e | |||
| 330a11861e | |||
| 4827fe86bb | |||
| 72c55b56f2 | |||
| 295da7e5f3 | |||
| 4eab85d364 | |||
| 3cc83ce024 | |||
| 52c423a423 | |||
| 952f8e38cf | |||
| 542b3e8a64 | |||
| 4f79eb256d | |||
| 7620456486 | |||
| 052d209901 | |||
| e9c45ff406 | |||
| f1053d48a2 | |||
| ba12f52f6c | |||
| 456137fa24 | |||
| 692059e899 | |||
| 252ce0b581 | |||
| d24befa0bc | |||
| 5462710103 | |||
| d1b923bee9 | |||
| 81a6d1b0d4 | |||
| 04020f391a | |||
| c4ab0c09ea | |||
| 6e50e4b9ee | |||
| 1599f9f0c0 | |||
| 0c5b6e5556 | |||
| e3e04f5dae | |||
| 17bc8565f6 | |||
| c08954c18b | |||
| 7ba56a4b88 | |||
| 9b7f40ce4d | |||
| b70370f3fd | |||
| 4d3cf77ad5 | |||
| b2005ccaef | |||
| 659cf7249e | |||
| 62a010a25d | |||
| 6b4ea68bc7 | |||
| d6891c705e | |||
| 9bcb006ec7 | |||
| 02ac6ec81c | |||
| 5c91f5b71d | |||
| 16b674b984 | |||
| 65392d5e6b | |||
| 41d108ead6 | |||
| 0c9dbc61a1 | |||
| 6788fdb0d6 | |||
| 3a4b61579d | |||
| 51d7ba7446 | |||
| e608adea60 | |||
| 8dd6882222 | |||
| 5aef565fb6 | |||
| a4d6bcba09 | |||
| f09a577ab5 | |||
| 973e1acb67 | |||
| 676a724491 | |||
| 73318fd514 | |||
| cb08c15616 | |||
| a3287b85c5 | |||
| 0c0b1ec9ae | |||
| d2f87ca76c | |||
| 4935b14539 | |||
| 0835611d3a | |||
| 8b4fa2605e | |||
| c3910807c5 | |||
| c5b8b5687f | |||
| f61883b227 | |||
| 35ff9af6ce | |||
| dad2b9aac8 | |||
| 1613d30544 | |||
| b6df9debaf | |||
| b9d0dc60b0 | |||
| 37b1876807 | |||
| d206350738 | |||
| 2da1f9181a | |||
| bd396e1fd5 | |||
| f55c9ed1ba | |||
| 5da69c0b9a | |||
| a806e8cc58 | |||
| 369b260e12 | |||
| d9e7c1626a | |||
| 1a1a7bbbfd | |||
| 33e97e994d | |||
| 11e6848bb9 | |||
| 829410729c | |||
| 4995aecd62 | |||
| 0e2a3686c0 | |||
| 66b2140892 | |||
| 0d2857a242 | |||
| 17d99e6266 | |||
| ea7d4be3f8 | |||
| 05db8784ae | |||
| 68667d6057 | |||
| d58b5ef74b | |||
| f044037ec5 | |||
| a97f21ba4e | |||
| b594ed99b8 | |||
| 58b06222ff | |||
| 58dc397930 | |||
| 15073d63d9 | |||
| a6277370ca | |||
| 31b2d6be75 | |||
| 57ee14d62d | |||
| d470cfe86e | |||
| b55d8f46f4 | |||
| c15218e37a | |||
| 985aa0423d | |||
| c14a8dce93 | |||
| f5d45221ca | |||
| d668aa7c24 | |||
| e20fe421e7 | |||
| 2deb38d615 | |||
| b95d71af2b | |||
| ebe4ca6b60 | |||
| f159ed20c2 | |||
| cdbb042ce4 | |||
| 23bbe511fe | |||
| 444218e755 | |||
| 9cc60c9dd3 | |||
| cc1fbe0956 | |||
| 2c226d597d | |||
| c6ab32ffb9 | |||
| 97c6ec6d49 | |||
| 1fcf7ba5bc | |||
| 12c1e1d149 | |||
| fa222b0ea2 | |||
| 101be77d9d | |||
| a6dbef89c2 | |||
| 02f08879a4 | |||
| 61f1ee2d2d | |||
| ac4b592b4e | |||
| 93b6e80cd7 | |||
| f5de714451 | |||
| d4741eece1 | |||
| 585484cb06 | |||
| b696928a5b | |||
| 79d4e865fe | |||
| 8380879804 | |||
| 091461cece | |||
| d228c5459d | |||
| 3d5d3ea20c | |||
| de7f8eec04 | |||
| 4b6047e746 | |||
| c47673bf10 | |||
| 49432009fe | |||
| 2f6d2b08aa | |||
| 473f10877c | |||
| c7df82460c | |||
| b19697e3ac | |||
| 3e51448ef0 | |||
| 2b2e515a30 | |||
| 394e640909 | |||
| b48eb4e88b | |||
| b525480b25 | |||
| 45f18eaa52 | |||
| 4c92a2869b | |||
| 8041ab8a61 | |||
| 8251fc0811 | |||
| 827ff80c06 | |||
| 18ca998f67 | |||
| cb286a66be | |||
| 8444470e3a | |||
| f33828a1ca | |||
| 68e425f869 | |||
| da6344297a | |||
| 3cfca01372 | |||
| d934bb15b0 | |||
| 0c0a8392e5 | |||
| 98b6ce353c | |||
| 58043dac0f | |||
| d2c1f1131b | |||
| fa5c7a9e75 | |||
| ceb94d52a1 | |||
| a2716712ab | |||
| 1ac7baceff | |||
| 968d94d417 | |||
| 7842181b47 | |||
| 2ce47fda88 | |||
| ffd010767f | |||
| 635990a5b0 | |||
| 562f2375c5 | |||
| edf533c83e | |||
| c1d61c88e9 | |||
| 6360b846c6 | |||
| 57900d07f0 | |||
| 8e72e1ed88 | |||
| 10c547396d | |||
| 66dd871288 | |||
| d484939c02 | |||
| 85fd8729ce | |||
| b3e16c6423 | |||
| 30d6766db4 | |||
| 17234f82d0 | |||
| 7b57df02a7 | |||
| b65ae3b036 | |||
| bce76a7977 | |||
| 66f3e97457 | |||
| 0ee61d178f | |||
| 1ceba29f4b | |||
| 40c748a2ae | |||
| 0e9453a395 | |||
| 69d0bc8fd5 | |||
| 6a73e5a720 | |||
| e70ba29d95 | |||
| e5647cf70d | |||
| 9637cf0574 | |||
| 74cc63ba2f | |||
| 0b6e360602 | |||
| a49cda6523 | |||
| d612c72405 | |||
| eb152d7431 | |||
| 6bd143dd25 | |||
| 770d3eabc0 | |||
| 75e2ba5af3 | |||
| 1726bb6c0d | |||
| 5dfe65d53a | |||
| 83afd09f13 | |||
| d059cc7170 | |||
| 96488c1c74 | |||
| 84a81579ba | |||
| 244ba1a61a | |||
| 9dadc06e64 | |||
| 60f949e36f | |||
| af154e3053 | |||
| 9e6c290696 | |||
| 0a71063530 | |||
| 0b1ae11498 | |||
| 017a4e7c30 | |||
| 3891f18e71 | |||
| 6a2077cbd8 | |||
| aa11cc19e8 | |||
| 9267536fee | |||
| a6f5717567 | |||
| 0a58d6812a | |||
| 12507aab8a | |||
| d376fe9e17 | |||
| d21622bef4 | |||
| 7febec49b2 | |||
| 14d5098ca2 | |||
| bc8eac2439 | |||
| 4e65db80e8 | |||
| c0c71d6b3a | |||
| bac1c6d12f | |||
| ce68291d83 | |||
| 4176a0ade3 | |||
| ec10f2e72b | |||
| f36c268b9e | |||
| 824392a1c2 | |||
| 1f9a7b8fd3 | |||
| 9379e85e23 | |||
| 805c2832c1 | |||
| d33a048d89 | |||
| f77fdc0ce8 | |||
| 7da51787b9 | |||
| b1f422c1c5 | |||
| b3f966e2ca | |||
| 3f191e1b75 | |||
| 6d5fdfbf73 | |||
| 9a9e457dd6 | |||
| c316dbe2aa | |||
| b5fcb06a76 | |||
| b5a9a6793b | |||
| 0cf79155d4 | |||
| f8f9f3c438 | |||
| 1926e919be | |||
| e6c68eed51 | |||
| f9e747dbc6 | |||
| 0e86e292e4 | |||
| a590682764 | |||
| 31c40fa4cc | |||
| 1feb3838b5 | |||
| bd0732b1d0 | |||
| 0b5cbcefdd | |||
| e4a87f2f4f | |||
| 3c8cadf7ca | |||
| 7c0b26e8a0 | |||
| 982503e9a8 | |||
| 3d93675ff9 | |||
| 53d6c9b9c0 | |||
| a02e90d502 | |||
| 238dbffb48 | |||
| a9d7b6eab7 | |||
| 39c3334147 | |||
| 4223495e6c | |||
| 023e86d68f | |||
| ac57be91e1 | |||
| c2e65bafb5 | |||
| 74161b2122 | |||
| 39ee5c5a46 | |||
| 788f330d07 | |||
| af56151231 | |||
| 34d359fe03 | |||
| 4672dbda2a | |||
| 62252d157e | |||
| b1cf550123 | |||
| 9c84749e2c | |||
| cca4c47781 | |||
| 17bd9a1fa1 | |||
| 0321644bbd | |||
| 81e7988eb9 | |||
| 4985311d46 | |||
| 05348f3250 | |||
| 003609e565 | |||
| e0cfaee7aa | |||
| 9537a909f7 | |||
| e75387f029 | |||
| 8c2dd5fb9a | |||
| 27545dcc86 | |||
| 724e04e979 | |||
| dfc94c58f0 | |||
| 4b0f8d76f4 | |||
| fac895d7ba | |||
| d023f316ac | |||
| 04b40ff221 | |||
| 03a08435e2 | |||
| 4dd3ab8f32 | |||
| 8e0660ad54 | |||
| 1f42512199 | |||
| 1a29ea1038 | |||
| eab2b9dc09 | |||
| 94e92cd6c0 | |||
| 57cd6d2de1 | |||
| 822d468232 | |||
| bdaa6a1910 | |||
| 2221dcc9f2 | |||
| 2eb1ee967c | |||
| c49cfefe88 | |||
| c6a6f39d29 | |||
| a3d7811f24 | |||
| eb56ca3b0d | |||
| 755e0143fb | |||
| dfa48094dc | |||
| e585192eeb | |||
| 646924fce8 | |||
| 13c6eb42e9 | |||
| e5fb50476c | |||
| 073c590d0b | |||
| c63aa7f085 | |||
| c832e62db0 | |||
| b7a7119b1f | |||
| 19a880bb91 | |||
| 672399c751 | |||
| c54abde1bd | |||
| 95c1d2a887 | |||
| 3e6f27522b | |||
| 1b70f94282 | |||
| ebef84e9ea | |||
| 87d4970e8b | |||
| b1a772d194 | |||
| 4938765eb3 | |||
| cce78cc5e2 | |||
| f11f2bfb56 | |||
| 24f43e7ae9 | |||
| 9085b933d8 | |||
| d94d469c86 | |||
| 59502594f8 | |||
| d20c9bde7e | |||
| 62414e3073 | |||
| 505dde09de | |||
| 603d623eda | |||
| 87ebf2e50b | |||
| 37c3f0d8a0 | |||
| 48c985e775 | |||
| 7358fffb0f | |||
| de5b6386e0 | |||
| 1de2d5c2b6 | |||
| ef53a9229f | |||
| 259c39a63a | |||
| 4d87f6025e | |||
| 327b98eb13 | |||
| f977d10a19 | |||
| a0cf8c322d | |||
| 627be179c1 | |||
| 1e74f5850b | |||
| 6d83a73858 | |||
| 9b76872708 | |||
| 9b093c9a12 | |||
| 4d587c341b | |||
| f8f6cd6ef5 | |||
| cf08eac15e | |||
| 9a8552e8ae | |||
| 846317ef37 | |||
| 7e62789edf | |||
| 5865af7f6e | |||
| 852663f6d2 | |||
| c1148c4ea6 | |||
| 280dc77f8b | |||
| a9b30984a3 | |||
| 2f83c3b689 | |||
| b25ad12f1a | |||
| d95e43a6a1 | |||
| 27df987211 | |||
| 982745fb83 | |||
| 8fa8d471af | |||
| 2cc14bd0fb | |||
| 98ad72b096 | |||
| fdc8ed8d05 | |||
| 24fcb7f813 | |||
| 236c64a17d | |||
| 9e835e8edb | |||
| f96569da1e | |||
| 91ff45fbde | |||
| 1261f250c6 | |||
| 3673b45437 | |||
| 45aabc5d0d | |||
| f9bd83c854 | |||
| 499d8adb75 | |||
| 54386c82fd | |||
| 86a51015b1 | |||
| 38b9ec7a18 | |||
| 998406d20e | |||
| f7b4b750d8 | |||
| 2558ab3de7 | |||
| a4e2c56317 | |||
| 0c10ae1861 | |||
| b3cc828995 | |||
| ba8f9d8620 | |||
| 3c89a28a06 | |||
| 9c5d7716e2 | |||
| 46fd26e366 | |||
| 41a2eb5245 | |||
| c410d7a97d | |||
| 6fa63dcc0c | |||
| 3d7670a6ba | |||
| 96f25332ea | |||
| 3385d38648 | |||
| 1618c963e4 | |||
| 50462dcdc6 | |||
| 1e3be09b3b | |||
| 6e66a9222a | |||
| c3ac834526 | |||
| 9d1e8b1e1d | |||
| f605373a2b | |||
| 56b7622612 | |||
| 696a6ccd57 | |||
| 51b03b87e6 | |||
| aa7ba0bc1a | |||
| 07e4076585 | |||
| de1a459879 | |||
| 1aacb9bb15 | |||
| 9b4ecc96f6 | |||
| e3f4f874c5 | |||
| 6ace801418 | |||
| d31b93b513 | |||
| ac0fd6aa9a | |||
| c7e0888982 | |||
| 068f33cfdf | |||
| 4807cd8a6e | |||
| c703f1eed6 | |||
| 5ac5da3524 | |||
| 0c13d34ade | |||
| 35e824c287 | |||
| 1e0d290f2e | |||
| 0097a8d097 | |||
| 36cc43170d | |||
| 5578ad5e14 | |||
| d11f0a709d | |||
| 0a43b23275 | |||
| 7967683296 | |||
| 5b2c016834 | |||
| aaff125608 | |||
| 407adc7061 | |||
| d0e612dc36 | |||
| 5aa7435d25 | |||
| 7c23ec90a9 | |||
| 390957fec4 | |||
| 060a76dc3e | |||
| 6625810d2a | |||
| edc442afdb | |||
| 16b9514543 | |||
| 95c7f4a7f0 | |||
| ae6fabc6fe | |||
| 7eaadf616c | |||
| 8fed5fc5ae | |||
| f25951c412 | |||
| c11195d5e3 | |||
| 2ed5cba110 | |||
| 4c05a697fa | |||
| 1259a474ba | |||
| 3995deaf76 | |||
| 076587425e | |||
| da6aeaca46 | |||
| 8ee33ca551 | |||
| ea7f13922b | |||
| 38d0063c36 | |||
| 56d0d5986f | |||
| df83459721 | |||
| 54a9e00970 | |||
| 9bda96d39e | |||
| 053470d5a8 | |||
| 6f4160c014 | |||
| 65ef82a946 | |||
| b509a7060a | |||
| 5ad6ff239b | |||
| fdaa6ff9e3 | |||
| 2f31763335 | |||
| 350562919c | |||
| aa5c4945d6 | |||
| 6fbfc58602 | |||
| 77a5c43d50 | |||
| f28e4b86fb | |||
| 6801dd043d | |||
| b675e6ab77 | |||
| d6306f8ccb | |||
| a9817e9127 | |||
| bb5f33d13c | |||
| c08897cd10 | |||
| 384875f4fc | |||
| 9cfa84313c | |||
| f787c49b53 | |||
| fa90e14b06 | |||
| fe625a558e | |||
| 03b989251d | |||
| 95919051e0 | |||
| 46fb88c76f | |||
| 210bfaf8d6 | |||
| a50dec88d5 | |||
| 8dcec034ed | |||
| 9ef41f68fb | |||
| 05d733e707 | |||
| 4bbe28bdf0 | |||
| 0c01cf7c85 | |||
| 917cd13ce2 | |||
| cfb36443fb | |||
| 9d3826c676 | |||
| 4300bb2e1f | |||
| 6edc438789 | |||
| 955cf35d5f | |||
| 25cd7c7c50 | |||
| 0f8efb07c7 | |||
| 732cb6c45c | |||
| 4d63a89fa6 | |||
| 266a868ba9 | |||
| 8199967b31 | |||
| 9d61c18143 | |||
| 7c73e28a6d | |||
| 221bfa4c67 | |||
| aaca4987c9 | |||
| 1a8b7f7513 | |||
| 992b47b991 | |||
| 21d0f40751 | |||
| 5e7f06397f | |||
| adaace4ab8 | |||
| 739ff84732 | |||
| 2a177052de | |||
| 424eaba4c5 | |||
| e1cafa3834 | |||
| ba539eb9aa | |||
| 24de676a64 | |||
| 0c2741f7ad | |||
| f7c82baee9 | |||
| 2b34e0abdc | |||
| 6306bc3ddc | |||
| ea068dcc2c | |||
| 633fedaa96 | |||
| 4ff76cad2a | |||
| 65134c793b | |||
| 5af09e73f2 | |||
| d5f34cf34c | |||
| 6a2e559222 | |||
| cefa602601 | |||
| d773691848 | |||
| da07ad16c1 | |||
| f40707dc68 | |||
| 2d8ce500fa | |||
| ba0cea6826 | |||
| ddf1b04cce | |||
| ad5b3a4753 | |||
| 531ea5b3a2 | |||
| b468468e7e | |||
| d52e4e5df3 | |||
| 907743eee7 | |||
| 610e3dccb7 | |||
| 27392f832d | |||
| e9db7fea5b | |||
| bdd3930855 | |||
| cff0168f3a | |||
| 70d5c88026 | |||
| d83901e665 | |||
| 1e1984a586 | |||
| 2f180cea7f | |||
| f4d6a3ec4e | |||
| 7aa922ceac | |||
| 06dcc5a2c6 | |||
| 000f762fb9 | |||
| 3bc3eeb5da | |||
| 4e5699fa71 | |||
| acc576658a | |||
| a76274b549 | |||
| 803ff8ebb9 | |||
| b42152ffeb | |||
| 4015a5486c | |||
| 024b43ca06 | |||
| aae48e6fd7 | |||
| d29c7e7871 | |||
| 9448fe3db4 | |||
| 3817f3a89b | |||
| 755f4f324b | |||
| 863ab0e72e | |||
| 4ab0377c6e | |||
| 2062a7ca8f | |||
| b61a55eebf | |||
| c30078c5a3 | |||
| 39b91c97f0 | |||
| 8334ee18e6 | |||
| cc2592f582 | |||
| 96d35f7c54 | |||
| 98c5fc6ae2 | |||
| fbde0c6c96 | |||
| 602e7c83e2 | |||
| eb9218a86b | |||
| 9f2dcc3f13 | |||
| bc210b292b | |||
| 2113af9c52 | |||
| 6f417b57c1 | |||
| a7742d7d63 | |||
| 5179e37bd1 | |||
| a7b17bfaf0 | |||
| 4af1f31a3f | |||
| 0acd06d13c | |||
| fd22e98298 | |||
| 9e42e04b4a | |||
| 9103837228 | |||
| 3f3c5de851 | |||
| 167a12028d | |||
| 6af2faebd2 | |||
| 34b65be44a | |||
| fd16222613 | |||
| 1ba6cd8423 | |||
| b928ebdd53 | |||
| b42623ff9d | |||
| 59ae0e0013 | |||
| 4788de784b | |||
| 926535469d | |||
| c0f63eb21f | |||
| 5627a0cbdf | |||
| ed2a698392 | |||
| aaad1791d9 | |||
| ad6e82942b | |||
| 9a9954a036 | |||
| d60bb57d4b | |||
| 591708903b | |||
| f9d62fba7a | |||
| 85dde8a800 | |||
| 9d039c206b | |||
| cbff19ff1a | |||
| 4c3f9b2ef4 | |||
| 167bac23aa | |||
| 5d0cfa2527 | |||
| 0e523618a1 | |||
| 821fae0d94 | |||
| 3b26105f68 | |||
| 0f2f966a91 | |||
| d7d491d445 | |||
| 41effbe2da | |||
| 9b0d6862c4 | |||
| 890fcdf842 | |||
| 18dbac203f | |||
| 8d1f254dcc | |||
| e5841d3126 | |||
| 90df3af6cf | |||
| 11cc36d770 | |||
| 05f1939b02 | |||
| 050ea9762f | |||
| 0f24d4d2a1 | |||
| fc799191f4 | |||
| 8fad85edda | |||
| d70053aba5 | |||
| b699fe7a9d | |||
| 94c67faaea | |||
| b2ed5c3070 | |||
| 9fe49497bb | |||
| 5b8c10f2f8 | |||
| 24983f62e2 | |||
| f2057ce1ab | |||
| 6797fd65a5 | |||
| bf489feef1 | |||
| 947e06a860 | |||
| 6a3d925a47 | |||
| 04d5ba266f | |||
| 90be83ae99 | |||
| 5e80bd3cc9 | |||
| fb7ef76e74 | |||
| db4b1e613c | |||
| ee39081b11 | |||
| 7d842f5bcf | |||
| 4eac198270 | |||
| faac32418c | |||
| 42810621df | |||
| 61a5378aeb | |||
| 42d644ef91 | |||
| c95a56450d | |||
| dc5199feea | |||
| f88fdf6a1b | |||
| b68057d927 | |||
| e9a860d9cb | |||
| 8be86cbdfd | |||
| 5091e64a42 | |||
| 828304d587 | |||
| 9d584475f6 | |||
| bb60cb0bf9 | |||
| 25f908b320 | |||
| 9b7dca2fa1 | |||
| 55e1dfb778 | |||
| 735a79ae83 | |||
| ef2b400c61 | |||
| c2263db7bc | |||
| 53eca2ff5b | |||
| 7bbbda71df | |||
| 3a15a3821a | |||
| 9557b9f70f | |||
| 464441d8c3 | |||
| f30f1afd47 | |||
| b3db37b99d | |||
| 7a276f39fb | |||
| 415668ecf0 | |||
| 651967b95c | |||
| 8e0baf257c | |||
| c8268e65fd | |||
| 3cf4375387 | |||
| 438e2dc228 | |||
| c1adbe3189 | |||
| 7ee1816612 | |||
| b4084491e5 | |||
| cb97421edf | |||
| e461031d40 | |||
| c82e4596e4 | |||
| 8f4f834ce6 | |||
| 15ba3e123f | |||
| 2db243b8ed | |||
| b57b64b7a3 | |||
| 8e386ac71f | |||
| bc1af6227a | |||
| 2b84c97a78 | |||
| 3ddf84534b | |||
| 4c2dff88de | |||
| bd26104088 | |||
| 80238880e6 | |||
| f4abafb093 | |||
| 3e538355e2 | |||
| 5f80f43ff5 | |||
| 76e9da3fe9 | |||
| cafa04f842 | |||
| fef84f69ec | |||
| ba98cd97e5 | |||
| 81b897c291 | |||
| 1c4d70896a | |||
| 039bcd932a | |||
| eaa9228a4f | |||
| a8757df963 | |||
| b221143c0f | |||
| 2c796de92b | |||
| 6f2a64a511 | |||
| 995841624c | |||
| f5f675ef6c | |||
| bdc8e9118b | |||
| d9ed9a9a83 | |||
| b57faa41c2 | |||
| 1b5fe91624 | |||
| 1a305cadb3 | |||
| bbcd06f42f | |||
| 34183237ce | |||
| 286ec92967 | |||
| 43940f7ffc | |||
| b05631432b | |||
| dc92886ab6 | |||
| 993416d9cf | |||
| f4a79b0554 | |||
| 411fd2b761 | |||
| 327109f327 | |||
| f1f9121bc7 | |||
| 3b42e19505 | |||
| 709156ee65 | |||
| 472030907b | |||
| 61e30c15a9 | |||
| d332f33346 | |||
| 636db09d73 | |||
| 64c018507b | |||
| 3e513ee6ab | |||
| c34445a496 | |||
| c2ce3d927a | |||
| ff60abb575 | |||
| 7d91dfe339 | |||
| 15af65d4cf | |||
| 0e8431d17b | |||
| 81afeda537 | |||
| e10de25e86 | |||
| 826fdaf06c | |||
| a4d0a1483a | |||
| 746bd47ce5 | |||
| dcb4cabb26 | |||
| 59b4baee0c | |||
| 2610724ee0 | |||
| 61359a5bd0 | |||
| bb3bbd192b | |||
| de781b306f | |||
| 1212aef03b | |||
| 814550d2a6 | |||
| 478663b08c | |||
| fb9a00c36d | |||
| fb68fe8930 | |||
| 73ee01a7f4 | |||
| eb9b5fa9a5 | |||
| 8bd5405228 | |||
| cb51a155b2 | |||
| d3be58b6d7 | |||
| 8ecfbdb4ff | |||
| 63256a00ff | |||
| 450dc92452 | |||
| 9c16408e91 | |||
| d42a2b2d16 | |||
| 3d2f4fa164 | |||
| 3d394943e6 | |||
| cf96a9fd27 | |||
| 76e81dfbb0 | |||
| 684ba6fe14 | |||
| 830cb5cad7 | |||
| 3e25f32c9b | |||
| 81567a9d3e | |||
| 2b06208bbd | |||
| 5f637e5a02 | |||
| cc712a165d | |||
| 4a2adba8f4 | |||
| 84f78059d3 | |||
| b67c0e5f4a | |||
| 1e1ddd3279 | |||
| 70f69cb265 | |||
| ae4cc404c1 | |||
| 32df5faa25 | |||
| ac8e7d57dc | |||
| da4a531717 | |||
| fea12f7806 | |||
| 7499a15c92 | |||
| 115e471515 | |||
| c4df8989e9 | |||
| 0f11b1fc0d | |||
| 5f4e55bc68 | |||
| c621384707 | |||
| c0162a64d1 | |||
| 527d86a93d | |||
| 7d66f1e391 | |||
| 6265155ce4 | |||
| 70d1ff7155 | |||
| 4753206783 | |||
| 6376a3aef2 | |||
| 1ae16beb06 | |||
| e53a4ce64d | |||
| 72655a9eea | |||
| 125f890dc3 | |||
| 101d50703c | |||
| 901ed5545f | |||
| a604d44d06 | |||
| bb92eb5a93 | |||
| 699b4b9fc5 | |||
| 1fb3133ec5 | |||
| fea45c6911 | |||
| 1d7d18afba | |||
| aadbebf9d8 | |||
| 7e12af2448 | |||
| b335fe67b0 | |||
| 776d92e797 | |||
| dde029f105 | |||
| 7ffc02283b | |||
| 7a31a6edee | |||
| df05bc65c5 | |||
| 3a49ff9e72 | |||
| 302c2354a3 | |||
| 158b13e0ba | |||
| 71e2a17fdb | |||
| b1dc7ed873 | |||
| 1bced43e96 | |||
| 7de627c504 | |||
| 6a4bfc0863 | |||
| 170bf6d7af | |||
| 8550f7e04d | |||
| 9f3e83baf6 | |||
| 17ab61a40d | |||
| 1b844f8413 | |||
| baa22fa808 | |||
| f78cebfc98 | |||
| d2a9ca13f1 | |||
| 71bae7c23f | |||
| 1dacea3a20 | |||
| 083a7c8f0a | |||
| d4f104f17f | |||
| fd44ba296d | |||
| 4ed91ce7ed | |||
| b04d6a2d9b | |||
| 15e2f991dd | |||
| a08eac452e | |||
| 0c171c0749 | |||
| 7021ea6748 | |||
| dbd65a3b01 | |||
| 55aa1d4852 | |||
| 971dacaf41 | |||
| 159534313e | |||
| 2f68c43e39 | |||
| e2483aa072 | |||
| 6c53af8e41 | |||
| c4ca9a7bae | |||
| c69f37343c | |||
| aec62a7fc5 | |||
| bf26050f7e | |||
| 402bce1a31 | |||
| e68657cdb2 | |||
| e00cfe067d | |||
| 8868350888 | |||
| fcaeeac931 | |||
| a53582d706 | |||
| e655083e3c | |||
| 1b64851fa8 | |||
| 40a7c70969 | |||
| 2e143b8799 | |||
| 5105a937d1 | |||
| 3930c9a492 | |||
| 29fb4f98b1 | |||
| f0839d2703 | |||
| 405e820fe1 | |||
| 3386efddba | |||
| c41650db20 | |||
| a0ff55db7d | |||
| 896bffb543 | |||
| 0df6159149 | |||
| cfb77091ca | |||
| 6598e9d506 | |||
| 8c2cf89845 | |||
| aeb8dfc52d | |||
| 0649a2fbdb | |||
| ec32061f5f | |||
| f2da1a0a7a | |||
| fda1e6f148 | |||
| 2a48730166 | |||
| 0bdbc745c4 | |||
| 6f70b0524a | |||
| b9b19185bc | |||
| 63ba9970bd | |||
| 1f726e81f9 | |||
| a9a6801c6d | |||
| 222af8e7e4 | |||
| df71853075 | |||
| bfb10d74eb | |||
| ec8b7c933a | |||
| 68d15fc62e | |||
| f479935cda | |||
| 76860933f0 | |||
| 19a936fc03 | |||
| be17fce657 | |||
| 9a1d7736f8 | |||
| 9f295b2c91 | |||
| 6e8daaec0f | |||
| 3a8154051f | |||
| f3f46096d6 | |||
| ace37df941 | |||
| de7377485b | |||
| e23578acd9 | |||
| 770445ae2a | |||
| b9b65e9392 | |||
| ac9182f20d | |||
| 125b9f6057 | |||
| e9e9e3898a | |||
| 2cf1a13755 | |||
| 6465e393b6 | |||
| 6e6c0f31f7 | |||
| d0e3e638c3 | |||
| 61d7d67c39 |
@@ -34,11 +34,11 @@ This is a template helping you to create an issue which can be processed as quic
|
||||
- [ ] I report the issue, it's not a question
|
||||
<!--
|
||||
OpenCV team works with forum.opencv.org, Stack Overflow and other communities
|
||||
to discuss problems. Tickets with question without real issue statement will be
|
||||
to discuss problems. Tickets with questions without a real issue statement will be
|
||||
closed.
|
||||
-->
|
||||
- [ ] I checked the problem with documentation, FAQ, open issues,
|
||||
forum.opencv.org, Stack Overflow, etc and have not found solution
|
||||
forum.opencv.org, Stack Overflow, etc and have not found any solution
|
||||
<!--
|
||||
Places to check:
|
||||
* OpenCV documentation: https://docs.opencv.org
|
||||
@@ -47,11 +47,11 @@ This is a template helping you to create an issue which can be processed as quic
|
||||
* OpenCV issue tracker: https://github.com/opencv/opencv/issues?q=is%3Aissue
|
||||
* Stack Overflow branch: https://stackoverflow.com/questions/tagged/opencv
|
||||
-->
|
||||
- [ ] I updated to latest OpenCV version and the issue is still there
|
||||
- [ ] I updated to the latest OpenCV version and the issue is still there
|
||||
<!--
|
||||
master branch for OpenCV 4.x and 3.4 branch for OpenCV 3.x releases.
|
||||
OpenCV team supports only latest release for each branch.
|
||||
The ticket is closed, if the problem is not reproduced with modern version.
|
||||
OpenCV team supports only the latest release for each branch.
|
||||
The ticket is closed if the problem is not reproduced with the modern version.
|
||||
-->
|
||||
- [ ] There is reproducer code and related data files: videos, images, onnx, etc
|
||||
<!--
|
||||
@@ -61,9 +61,9 @@ This is a template helping you to create an issue which can be processed as quic
|
||||
to reduce attachment size
|
||||
* Use PNG for images, if you report some CV related bug, but not image reader
|
||||
issue
|
||||
* Attach the image as archive to the ticket, if you report some reader issue.
|
||||
* Attach the image as an archive to the ticket, if you report some reader issue.
|
||||
Image hosting services compress images and it breaks the repro code.
|
||||
* Provide ONNX file for some public model or ONNX file with with random weights,
|
||||
* Provide ONNX file for some public model or ONNX file with random weights,
|
||||
if you report ONNX parsing or handling issue. Architecture details diagram
|
||||
from netron tool can be very useful too. See https://lutzroeder.github.io/netron/
|
||||
-->
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
name: Bug Report
|
||||
description: Create a report to help us reproduce and fix the bug
|
||||
labels: ["bug"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Thank you for contributing! Before reporting a bug, please have a look at the [FAQ](https://github.com/opencv/opencv/wiki/FAQ), make sure the issue has no duplicate and hasn't been already addressed by searching through [the existing and past issues](https://github.com/opencv/opencv/issues?page=1&q=is%3Aissue+sort%3Acreated-desc).
|
||||
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: System Information
|
||||
description: |
|
||||
Please provide the following system information to help us diagnose the bug. For example:
|
||||
|
||||
// example for c++ user
|
||||
OpenCV version: 4.6.0
|
||||
Operating System / Platform: Ubuntu 20.04
|
||||
Compiler & compiler version: GCC 9.3.0
|
||||
|
||||
// example for python user
|
||||
OpenCV python version: 4.6.0.66
|
||||
Operating System / Platform: Ubuntu 20.04
|
||||
Python version: 3.9.6
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Detailed description
|
||||
description: |
|
||||
Please provide a clear and concise description of what the bug is and paste the error log below. It helps improving readability if the error log is wrapped in ```` ```triple quotes blocks``` ````.
|
||||
placeholder: |
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
```
|
||||
# error log
|
||||
```
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Steps to reproduce
|
||||
description: |
|
||||
Please provide a minimal example to help us reproduce the bug. Code should be wrapped with ```` ```triple quotes blocks``` ```` to improve readability. If the code is too long, please attach as a file or create and link a public gist: https://gist.github.com.
|
||||
|
||||
Related data files (images, onnx, etc) should be attached below as well. If the data files are too big, feel free to upload them to a online drive, share them and put the link below.
|
||||
placeholder: |
|
||||
```cpp (replace cpp with python if python code)
|
||||
# sample code to reproduce the bug
|
||||
```
|
||||
|
||||
Test data: [image](https://link/to/the/image), [model.onnx](htts://link/to/the/onnx/model)
|
||||
validations:
|
||||
required: true
|
||||
- type: checkboxes
|
||||
attributes:
|
||||
label: Issue submission checklist
|
||||
options:
|
||||
- label: I report the issue, it's not a question
|
||||
required: true
|
||||
- label: I checked the problem with documentation, FAQ, open issues, forum.opencv.org, Stack Overflow, etc and have not found any solution
|
||||
- label: I updated to the latest OpenCV version and the issue is still there
|
||||
- label: There is reproducer code and related data files (videos, images, onnx, etc)
|
||||
@@ -0,0 +1,5 @@
|
||||
blank_issues_enabled: true
|
||||
contact_links:
|
||||
- name: Questions
|
||||
url: https://forum.opencv.org/
|
||||
about: Ask questions and discuss with OpenCV community members
|
||||
@@ -0,0 +1,26 @@
|
||||
name: Documentation
|
||||
description: Report an issue related to https://docs.opencv.org/
|
||||
labels: ["category: documentation"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Thank you for contributing! Before submitting a doc issue, please make sure it has no duplicate by searching through [the existing and past issues](https://github.com/opencv/opencv/issues?page=1&q=is%3Aissue+sort%3Acreated-desc)
|
||||
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Descripe the doc issue
|
||||
description: >
|
||||
Please provide a clear and concise description of what content in https://docs.opencv.org/ is an issue. Note that there are multiple active branches, such as 3.4, 4.x and 5.x, so please specify the branch with the problem.
|
||||
placeholder: |
|
||||
A clear and concise description of what content in https://docs.opencv.org/ is an issue.
|
||||
|
||||
Link to the doc: https://docs.opencv.org/4.x/d3/d63/classcv_1_1Mat.html
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Fix suggestion
|
||||
description: >
|
||||
Tell us how we could improve the documentation in this regard.
|
||||
@@ -0,0 +1,22 @@
|
||||
name: Feature request
|
||||
description: Submit a request for a new OpenCV feature
|
||||
labels: ["feature"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Thank you for contributing! Before submitting a feature request, please make sure the request has no duplicate by searching through [the existing and past issues](https://github.com/opencv/opencv/issues?page=1&q=is%3Aissue+sort%3Acreated-desc)
|
||||
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Descripe the feature and motivation
|
||||
description: |
|
||||
Please provide a clear and concise proposal of the feature and outline the motivation.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Additional context
|
||||
description: |
|
||||
Add any other context, such as pseudo code, links, diagram, screenshots, to help the community better understand the feature request.
|
||||
@@ -3,9 +3,9 @@
|
||||
See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request
|
||||
|
||||
- [ ] I agree to contribute to the project under Apache 2 License.
|
||||
- [ ] To the best of my knowledge, the proposed patch is not based on a code under GPL or other license that is incompatible with OpenCV
|
||||
- [ ] The PR is proposed to proper branch
|
||||
- [ ] There is reference to original bug report and related work
|
||||
- [ ] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV
|
||||
- [ ] The PR is proposed to the proper branch
|
||||
- [ ] There is a reference to the original bug report and related work
|
||||
- [ ] There is accuracy test, performance test and test data in opencv_extra repository, if applicable
|
||||
Patch to opencv_extra has the same branch name.
|
||||
- [ ] The feature is well documented and sample code can be built with the project CMake
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
name: PR:4.x
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- 4.x
|
||||
|
||||
jobs:
|
||||
Ubuntu2004-ARM64:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-ARM64.yaml@main
|
||||
|
||||
Ubuntu2004-ARM64-Debug:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-ARM64-Debug.yaml@main
|
||||
|
||||
Ubuntu2004-x64:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-U20.yaml@main
|
||||
|
||||
Ubuntu2004-x64-CUDA:
|
||||
if: "${{ contains(github.event.pull_request.labels.*.name, 'category: dnn') }} || ${{ contains(github.event.pull_request.labels.*.name, 'category: dnn (onnx)') }}"
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-U20-Cuda.yaml@main
|
||||
|
||||
Windows10-x64:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-W10.yaml@main
|
||||
|
||||
macOS-ARM64:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-macOS-ARM64.yaml@main
|
||||
|
||||
macOS-x64:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-macOS-x86_64.yaml@main
|
||||
|
||||
iOS:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-iOS.yaml@main
|
||||
|
||||
Android:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-Android.yaml@main
|
||||
|
||||
TIM-VX:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-timvx-backend-tests-4.x.yml@main
|
||||
|
||||
docs:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-4.x-docs.yaml@main
|
||||
@@ -2,6 +2,9 @@ name: arm64 build checks
|
||||
|
||||
on: workflow_dispatch
|
||||
|
||||
permissions:
|
||||
contents: read # to fetch code (actions/checkout)
|
||||
|
||||
jobs:
|
||||
build:
|
||||
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
name: lint_python
|
||||
on: workflow_dispatch
|
||||
permissions:
|
||||
contents: read # to fetch code (actions/checkout)
|
||||
jobs:
|
||||
lint_python:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/setup-python@v2
|
||||
- run: pip install --upgrade pip wheel
|
||||
- run: pip install bandit black codespell flake8 flake8-2020 flake8-bugbear
|
||||
flake8-comprehensions isort mypy pytest pyupgrade safety
|
||||
- run: bandit --recursive --skip B101 . || true # B101 is assert statements
|
||||
- run: black --check . || true
|
||||
- run: codespell || true # --ignore-words-list="" --skip="*.css,*.js,*.lock"
|
||||
- run: flake8 . --count --select=E9,F63,F7 --show-source --statistics
|
||||
- run: flake8 . --count --exit-zero --max-complexity=10 --max-line-length=88
|
||||
--show-source --statistics
|
||||
- run: isort --check-only --profile black . || true
|
||||
- run: pip install -r requirements.txt || pip install --editable . || true
|
||||
- run: mkdir --parents --verbose .mypy_cache
|
||||
- run: mypy --ignore-missing-imports --install-types --non-interactive . || true
|
||||
- run: pytest . || true
|
||||
- run: pytest --doctest-modules . || true
|
||||
- run: shopt -s globstar && pyupgrade --py36-plus **/*.py || true
|
||||
- run: safety check
|
||||
Vendored
+5
-1
@@ -1,4 +1,4 @@
|
||||
cmake_minimum_required(VERSION 2.8.11 FATAL_ERROR)
|
||||
cmake_minimum_required(VERSION ${MIN_VER_CMAKE} FATAL_ERROR)
|
||||
|
||||
project(Carotene)
|
||||
|
||||
@@ -27,6 +27,10 @@ if(CMAKE_COMPILER_IS_GNUCC)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(APPLE AND CV_CLANG AND WITH_NEON)
|
||||
ocv_warnings_disable(CMAKE_CXX_FLAGS -Wno-unused-function)
|
||||
endif()
|
||||
|
||||
add_library(carotene_objs OBJECT EXCLUDE_FROM_ALL
|
||||
${carotene_headers}
|
||||
${carotene_sources}
|
||||
|
||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
||||
cmake_minimum_required(VERSION 2.8.8 FATAL_ERROR)
|
||||
cmake_minimum_required(VERSION ${MIN_VER_CMAKE} FATAL_ERROR)
|
||||
|
||||
include(CheckCCompilerFlag)
|
||||
include(CheckCXXCompilerFlag)
|
||||
|
||||
Vendored
+28
-19
@@ -1296,13 +1296,13 @@ struct MorphCtx
|
||||
CAROTENE_NS::BORDER_MODE border;
|
||||
uchar borderValues[4];
|
||||
};
|
||||
inline int TEGRA_MORPHINIT(cvhalFilter2D **context, int operation, int src_type, int dst_type, int, int,
|
||||
inline int TEGRA_MORPHINIT(cvhalFilter2D **context, int operation, int src_type, int dst_type, int width, int height,
|
||||
int kernel_type, uchar *kernel_data, size_t kernel_step, int kernel_width, int kernel_height, int anchor_x, int anchor_y,
|
||||
int borderType, const double borderValue[4], int iterations, bool allowSubmatrix, bool allowInplace)
|
||||
{
|
||||
if(!context || !kernel_data || src_type != dst_type ||
|
||||
CV_MAT_DEPTH(src_type) != CV_8U || src_type < 0 || (src_type >> CV_CN_SHIFT) > 3 ||
|
||||
|
||||
width < kernel_width || height < kernel_height ||
|
||||
allowSubmatrix || allowInplace || iterations != 1 ||
|
||||
!CAROTENE_NS::isSupportedConfiguration())
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
@@ -1778,30 +1778,30 @@ TegraCvtColor_Invoker(bgrx2hsvf, bgrx2hsv, src_data + static_cast<size_t>(range.
|
||||
: CV_HAL_ERROR_NOT_IMPLEMENTED \
|
||||
)
|
||||
|
||||
#define TEGRA_CVT2PYUVTOBGR(src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx) \
|
||||
#define TEGRA_CVT2PYUVTOBGR_EX(y_data, y_step, uv_data, uv_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx) \
|
||||
( \
|
||||
CAROTENE_NS::isSupportedConfiguration() ? \
|
||||
dcn == 3 ? \
|
||||
uIdx == 0 ? \
|
||||
(swapBlue ? \
|
||||
CAROTENE_NS::yuv420i2rgb(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step) : \
|
||||
CAROTENE_NS::yuv420i2bgr(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step)), \
|
||||
CV_HAL_ERROR_OK : \
|
||||
uIdx == 1 ? \
|
||||
(swapBlue ? \
|
||||
CAROTENE_NS::yuv420sp2rgb(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step) : \
|
||||
CAROTENE_NS::yuv420sp2bgr(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step)), \
|
||||
CV_HAL_ERROR_OK : \
|
||||
CV_HAL_ERROR_NOT_IMPLEMENTED : \
|
||||
@@ -1809,29 +1809,32 @@ TegraCvtColor_Invoker(bgrx2hsvf, bgrx2hsv, src_data + static_cast<size_t>(range.
|
||||
uIdx == 0 ? \
|
||||
(swapBlue ? \
|
||||
CAROTENE_NS::yuv420i2rgbx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step) : \
|
||||
CAROTENE_NS::yuv420i2bgrx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step)), \
|
||||
CV_HAL_ERROR_OK : \
|
||||
uIdx == 1 ? \
|
||||
(swapBlue ? \
|
||||
CAROTENE_NS::yuv420sp2rgbx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step) : \
|
||||
CAROTENE_NS::yuv420sp2bgrx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||
src_data, src_step, \
|
||||
src_data + src_step * dst_height, src_step, \
|
||||
y_data, y_step, \
|
||||
uv_data, uv_step, \
|
||||
dst_data, dst_step)), \
|
||||
CV_HAL_ERROR_OK : \
|
||||
CV_HAL_ERROR_NOT_IMPLEMENTED : \
|
||||
CV_HAL_ERROR_NOT_IMPLEMENTED \
|
||||
: CV_HAL_ERROR_NOT_IMPLEMENTED \
|
||||
)
|
||||
#define TEGRA_CVT2PYUVTOBGR(src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx) \
|
||||
TEGRA_CVT2PYUVTOBGR_EX(src_data, src_step, src_data + src_step * dst_height, src_step, dst_data, dst_step, \
|
||||
dst_width, dst_height, dcn, swapBlue, uIdx);
|
||||
|
||||
#undef cv_hal_cvtBGRtoBGR
|
||||
#define cv_hal_cvtBGRtoBGR TEGRA_CVTBGRTOBGR
|
||||
@@ -1841,12 +1844,18 @@ TegraCvtColor_Invoker(bgrx2hsvf, bgrx2hsv, src_data + static_cast<size_t>(range.
|
||||
#define cv_hal_cvtBGRtoGray TEGRA_CVTBGRTOGRAY
|
||||
#undef cv_hal_cvtGraytoBGR
|
||||
#define cv_hal_cvtGraytoBGR TEGRA_CVTGRAYTOBGR
|
||||
#if 0 // bit-exact tests are failed
|
||||
#undef cv_hal_cvtBGRtoYUV
|
||||
#define cv_hal_cvtBGRtoYUV TEGRA_CVTBGRTOYUV
|
||||
#endif
|
||||
#undef cv_hal_cvtBGRtoHSV
|
||||
#define cv_hal_cvtBGRtoHSV TEGRA_CVTBGRTOHSV
|
||||
#if 0 // bit-exact tests are failed
|
||||
#undef cv_hal_cvtTwoPlaneYUVtoBGR
|
||||
#define cv_hal_cvtTwoPlaneYUVtoBGR TEGRA_CVT2PYUVTOBGR
|
||||
#undef cv_hal_cvtTwoPlaneYUVtoBGREx
|
||||
#define cv_hal_cvtTwoPlaneYUVtoBGREx TEGRA_CVT2PYUVTOBGR_EX
|
||||
#endif
|
||||
|
||||
#endif // OPENCV_IMGPROC_HAL_INTERFACE_H
|
||||
|
||||
|
||||
+18
-18
@@ -109,9 +109,9 @@ template <> struct wAdd<s32>
|
||||
vgamma = vdupq_n_f32(_gamma + 0.5);
|
||||
}
|
||||
|
||||
void operator() (const typename VecTraits<s32>::vec128 & v_src0,
|
||||
const typename VecTraits<s32>::vec128 & v_src1,
|
||||
typename VecTraits<s32>::vec128 & v_dst) const
|
||||
void operator() (const VecTraits<s32>::vec128 & v_src0,
|
||||
const VecTraits<s32>::vec128 & v_src1,
|
||||
VecTraits<s32>::vec128 & v_dst) const
|
||||
{
|
||||
float32x4_t vs1 = vcvtq_f32_s32(v_src0);
|
||||
float32x4_t vs2 = vcvtq_f32_s32(v_src1);
|
||||
@@ -121,9 +121,9 @@ template <> struct wAdd<s32>
|
||||
v_dst = vcvtq_s32_f32(vs1);
|
||||
}
|
||||
|
||||
void operator() (const typename VecTraits<s32>::vec64 & v_src0,
|
||||
const typename VecTraits<s32>::vec64 & v_src1,
|
||||
typename VecTraits<s32>::vec64 & v_dst) const
|
||||
void operator() (const VecTraits<s32>::vec64 & v_src0,
|
||||
const VecTraits<s32>::vec64 & v_src1,
|
||||
VecTraits<s32>::vec64 & v_dst) const
|
||||
{
|
||||
float32x2_t vs1 = vcvt_f32_s32(v_src0);
|
||||
float32x2_t vs2 = vcvt_f32_s32(v_src1);
|
||||
@@ -153,9 +153,9 @@ template <> struct wAdd<u32>
|
||||
vgamma = vdupq_n_f32(_gamma + 0.5);
|
||||
}
|
||||
|
||||
void operator() (const typename VecTraits<u32>::vec128 & v_src0,
|
||||
const typename VecTraits<u32>::vec128 & v_src1,
|
||||
typename VecTraits<u32>::vec128 & v_dst) const
|
||||
void operator() (const VecTraits<u32>::vec128 & v_src0,
|
||||
const VecTraits<u32>::vec128 & v_src1,
|
||||
VecTraits<u32>::vec128 & v_dst) const
|
||||
{
|
||||
float32x4_t vs1 = vcvtq_f32_u32(v_src0);
|
||||
float32x4_t vs2 = vcvtq_f32_u32(v_src1);
|
||||
@@ -165,9 +165,9 @@ template <> struct wAdd<u32>
|
||||
v_dst = vcvtq_u32_f32(vs1);
|
||||
}
|
||||
|
||||
void operator() (const typename VecTraits<u32>::vec64 & v_src0,
|
||||
const typename VecTraits<u32>::vec64 & v_src1,
|
||||
typename VecTraits<u32>::vec64 & v_dst) const
|
||||
void operator() (const VecTraits<u32>::vec64 & v_src0,
|
||||
const VecTraits<u32>::vec64 & v_src1,
|
||||
VecTraits<u32>::vec64 & v_dst) const
|
||||
{
|
||||
float32x2_t vs1 = vcvt_f32_u32(v_src0);
|
||||
float32x2_t vs2 = vcvt_f32_u32(v_src1);
|
||||
@@ -197,17 +197,17 @@ template <> struct wAdd<f32>
|
||||
vgamma = vdupq_n_f32(_gamma + 0.5);
|
||||
}
|
||||
|
||||
void operator() (const typename VecTraits<f32>::vec128 & v_src0,
|
||||
const typename VecTraits<f32>::vec128 & v_src1,
|
||||
typename VecTraits<f32>::vec128 & v_dst) const
|
||||
void operator() (const VecTraits<f32>::vec128 & v_src0,
|
||||
const VecTraits<f32>::vec128 & v_src1,
|
||||
VecTraits<f32>::vec128 & v_dst) const
|
||||
{
|
||||
float32x4_t vs1 = vmlaq_f32(vgamma, v_src0, valpha);
|
||||
v_dst = vmlaq_f32(vs1, v_src1, vbeta);
|
||||
}
|
||||
|
||||
void operator() (const typename VecTraits<f32>::vec64 & v_src0,
|
||||
const typename VecTraits<f32>::vec64 & v_src1,
|
||||
typename VecTraits<f32>::vec64 & v_dst) const
|
||||
void operator() (const VecTraits<f32>::vec64 & v_src0,
|
||||
const VecTraits<f32>::vec64 & v_src1,
|
||||
VecTraits<f32>::vec64 & v_dst) const
|
||||
{
|
||||
float32x2_t vs1 = vmla_f32(vget_low(vgamma), v_src0, vget_low(valpha));
|
||||
v_dst = vmla_f32(vs1, v_src1, vget_low(vbeta));
|
||||
|
||||
Vendored
+1
-1
@@ -391,9 +391,9 @@ void blur3x3(const Size2D &size, s32 cn,
|
||||
}
|
||||
else if (borderType == BORDER_MODE_REFLECT101)
|
||||
{
|
||||
tcurr = vsetq_lane_u16(vgetq_lane_u16(tcurr, 3),tcurr, 5);
|
||||
tcurr = vsetq_lane_u16(vgetq_lane_u16(tcurr, 4),tcurr, 6);
|
||||
tcurr = vsetq_lane_u16(vgetq_lane_u16(tcurr, 5),tcurr, 7);
|
||||
tcurr = vsetq_lane_u16(vgetq_lane_u16(tcurr, 3),tcurr, 5);
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
Vendored
+6
-6
@@ -1,9 +1,9 @@
|
||||
# Binaries branch name: ffmpeg/master_20210303
|
||||
# Binaries were created for OpenCV: 7ac6abe02a33bef445a5b77214ad31964e2c5cc1
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "629590c3ba09fb0c8eaa9ab858ff13d3a84ca1aa")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "638065d5a0dab8a828879942375dcac4")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "7f10ae2e6a080ba3714f7a38ee03ae15")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "f8e65dbe4a3b4eedc0d2997e07c3f3fd")
|
||||
# Binaries branch name: ffmpeg/4.x_20221225
|
||||
# Binaries were created for OpenCV: 4abe6dc48d4ec6229f332cc6cf6c7e234ac8027e
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "7dd0d4f1d6fe75f05f3d3b5e38cbc96c1a2d2809")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "e598ae2ece1ddf310bc49b58202fd87a")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "b2a40c142c20aef9fd663fc8f85c2971")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "8862c87496e2e8c375965e1277dee1c7")
|
||||
|
||||
function(download_win_ffmpeg script_var)
|
||||
set(${script_var} "" PARENT_SCOPE)
|
||||
|
||||
Vendored
+1
@@ -54,6 +54,7 @@ set_target_properties(${ITT_LIBRARY} PROPERTIES
|
||||
)
|
||||
|
||||
ocv_warnings_disable(CMAKE_C_FLAGS -Wundef -Wsign-compare)
|
||||
ocv_warnings_disable(CMAKE_C_FLAGS -Wstrict-prototypes) # clang15
|
||||
|
||||
if(ENABLE_SOLUTION_FOLDERS)
|
||||
set_target_properties(${ITT_LIBRARY} PROPERTIES FOLDER "3rdparty")
|
||||
|
||||
+103
-15
@@ -3,10 +3,10 @@ project(${JPEG_LIBRARY} C)
|
||||
ocv_warnings_disable(CMAKE_C_FLAGS -Wunused-parameter -Wsign-compare -Wshorten-64-to-32 -Wimplicit-fallthrough)
|
||||
|
||||
set(VERSION_MAJOR 2)
|
||||
set(VERSION_MINOR 0)
|
||||
set(VERSION_REVISION 6)
|
||||
set(VERSION_MINOR 1)
|
||||
set(VERSION_REVISION 3)
|
||||
set(VERSION ${VERSION_MAJOR}.${VERSION_MINOR}.${VERSION_REVISION})
|
||||
set(LIBJPEG_TURBO_VERSION_NUMBER 2000006)
|
||||
set(LIBJPEG_TURBO_VERSION_NUMBER 2001003)
|
||||
|
||||
string(TIMESTAMP BUILD "opencv-${OPENCV_VERSION}-libjpeg-turbo")
|
||||
if(CMAKE_BUILD_TYPE STREQUAL "Debug")
|
||||
@@ -15,8 +15,53 @@ endif()
|
||||
|
||||
message(STATUS "libjpeg-turbo: VERSION = ${VERSION}, BUILD = ${BUILD}")
|
||||
|
||||
math(EXPR BITS "${CMAKE_SIZEOF_VOID_P} * 8")
|
||||
string(TOLOWER ${CMAKE_SYSTEM_PROCESSOR} CMAKE_SYSTEM_PROCESSOR_LC)
|
||||
|
||||
if(CMAKE_SYSTEM_PROCESSOR_LC MATCHES "x86_64" OR
|
||||
CMAKE_SYSTEM_PROCESSOR_LC MATCHES "amd64" OR
|
||||
CMAKE_SYSTEM_PROCESSOR_LC MATCHES "i[0-9]86" OR
|
||||
CMAKE_SYSTEM_PROCESSOR_LC MATCHES "x86" OR
|
||||
CMAKE_SYSTEM_PROCESSOR_LC MATCHES "ia32")
|
||||
if(BITS EQUAL 64 OR CMAKE_C_COMPILER_ABI MATCHES "ELF X32")
|
||||
set(CPU_TYPE x86_64)
|
||||
else()
|
||||
set(CPU_TYPE i386)
|
||||
endif()
|
||||
if(NOT CMAKE_SYSTEM_PROCESSOR STREQUAL ${CPU_TYPE})
|
||||
set(CMAKE_SYSTEM_PROCESSOR ${CPU_TYPE})
|
||||
endif()
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR_LC STREQUAL "aarch64" OR
|
||||
CMAKE_SYSTEM_PROCESSOR_LC MATCHES "^arm")
|
||||
if(BITS EQUAL 64)
|
||||
set(CPU_TYPE arm64)
|
||||
else()
|
||||
set(CPU_TYPE arm)
|
||||
endif()
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR_LC MATCHES "^ppc" OR
|
||||
CMAKE_SYSTEM_PROCESSOR_LC MATCHES "^powerpc")
|
||||
set(CPU_TYPE powerpc)
|
||||
else()
|
||||
set(CPU_TYPE ${CMAKE_SYSTEM_PROCESSOR_LC})
|
||||
endif()
|
||||
if(CMAKE_OSX_ARCHITECTURES MATCHES "x86_64" OR
|
||||
CMAKE_OSX_ARCHITECTURES MATCHES "arm64" OR
|
||||
CMAKE_OSX_ARCHITECTURES MATCHES "i386")
|
||||
set(CPU_TYPE ${CMAKE_OSX_ARCHITECTURES})
|
||||
endif()
|
||||
if(CMAKE_OSX_ARCHITECTURES MATCHES "ppc")
|
||||
set(CPU_TYPE powerpc)
|
||||
endif()
|
||||
if(MSVC_IDE AND CMAKE_GENERATOR_PLATFORM MATCHES "arm64")
|
||||
set(CPU_TYPE arm64)
|
||||
endif()
|
||||
|
||||
OCV_OPTION(ENABLE_LIBJPEG_TURBO_SIMD "Include SIMD extensions for libjpeg-turbo, if available for this platform" (NOT CV_DISABLE_OPTIMIZATION)
|
||||
VISIBLE_IF BUILD_JPEG)
|
||||
option(WITH_ARITH_ENC "Include arithmetic encoding support when emulating the libjpeg v6b API/ABI" TRUE)
|
||||
option(WITH_ARITH_DEC "Include arithmetic decoding support when emulating the libjpeg v6b API/ABI" TRUE)
|
||||
set(WITH_SIMD 1)
|
||||
set(HAVE_LIBJPEG_TURBO_SIMD 0 PARENT_SCOPE)
|
||||
|
||||
include(CheckCSourceCompiles)
|
||||
include(CheckIncludeFiles)
|
||||
@@ -46,7 +91,6 @@ if(UNIX)
|
||||
ocv_update(HAVE_UNSIGNED_SHORT 1)
|
||||
# undef INCOMPLETE_TYPES_BROKEN
|
||||
ocv_update(RIGHT_SHIFT_IS_UNSIGNED 0)
|
||||
ocv_update(__CHAR_UNSIGNED__ 0)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -80,14 +124,13 @@ configure_file(jconfigint.h.in jconfigint.h)
|
||||
|
||||
include_directories(${CMAKE_CURRENT_BINARY_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/src)
|
||||
|
||||
set(JPEG_SOURCES
|
||||
jcapimin.c jcapistd.c jccoefct.c jccolor.c jcdctmgr.c jchuff.c jcicc.c
|
||||
jcinit.c jcmainct.c jcmarker.c jcmaster.c jcomapi.c jcparam.c jcphuff.c
|
||||
jcprepct.c jcsample.c jctrans.c jdapimin.c jdapistd.c jdatadst.c jdatasrc.c
|
||||
jdcoefct.c jdcolor.c jddctmgr.c jdhuff.c jdicc.c jdinput.c jdmainct.c jdmarker.c
|
||||
jdmaster.c jdmerge.c jdphuff.c jdpostct.c jdsample.c jdtrans.c jerror.c
|
||||
jfdctflt.c jfdctfst.c jfdctint.c jidctflt.c jidctfst.c jidctint.c jidctred.c
|
||||
jquant1.c jquant2.c jutils.c jmemmgr.c jmemnobs.c)
|
||||
set(JPEG_SOURCES jcapimin.c jcapistd.c jccoefct.c jccolor.c jcdctmgr.c jchuff.c
|
||||
jcicc.c jcinit.c jcmainct.c jcmarker.c jcmaster.c jcomapi.c jcparam.c
|
||||
jcphuff.c jcprepct.c jcsample.c jctrans.c jdapimin.c jdapistd.c jdatadst.c
|
||||
jdatasrc.c jdcoefct.c jdcolor.c jddctmgr.c jdhuff.c jdicc.c jdinput.c
|
||||
jdmainct.c jdmarker.c jdmaster.c jdmerge.c jdphuff.c jdpostct.c jdsample.c
|
||||
jdtrans.c jerror.c jfdctflt.c jfdctfst.c jfdctint.c jidctflt.c jidctfst.c
|
||||
jidctint.c jidctred.c jquant1.c jquant2.c jutils.c jmemmgr.c jmemnobs.c)
|
||||
|
||||
if(WITH_ARITH_ENC OR WITH_ARITH_DEC)
|
||||
set(JPEG_SOURCES ${JPEG_SOURCES} jaricom.c)
|
||||
@@ -101,12 +144,57 @@ if(WITH_ARITH_DEC)
|
||||
set(JPEG_SOURCES ${JPEG_SOURCES} jdarith.c)
|
||||
endif()
|
||||
|
||||
# No SIMD
|
||||
set(JPEG_SOURCES ${JPEG_SOURCES} jsimd_none.c)
|
||||
if(CMAKE_COMPILER_IS_GNUCC OR CMAKE_C_COMPILER_ID STREQUAL "Clang")
|
||||
# Use the maximum optimization level for release builds
|
||||
foreach(var CMAKE_C_FLAGS_RELEASE CMAKE_C_FLAGS_RELWITHDEBINFO)
|
||||
if(${var} MATCHES "-O2")
|
||||
string(REGEX REPLACE "-O2" "-O3" ${var} "${${var}}")
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if(CMAKE_SYSTEM_NAME STREQUAL "SunOS")
|
||||
if(CMAKE_C_COMPILER_ID MATCHES "SunPro")
|
||||
# Use the maximum optimization level for release builds
|
||||
foreach(var CMAKE_C_FLAGS_RELEASE CMAKE_C_FLAGS_RELWITHDEBINFO)
|
||||
if(${var} MATCHES "-xO3")
|
||||
string(REGEX REPLACE "-xO3" "-xO5" ${var} "${${var}}")
|
||||
endif()
|
||||
if(${var} MATCHES "-xO2")
|
||||
string(REGEX REPLACE "-xO2" "-xO5" ${var} "${${var}}")
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(ENABLE_LIBJPEG_TURBO_SIMD)
|
||||
add_subdirectory(src/simd)
|
||||
if(NEON_INTRINSICS)
|
||||
add_definitions(-DNEON_INTRINSICS)
|
||||
endif()
|
||||
else()
|
||||
set(WITH_SIMD 0)
|
||||
endif()
|
||||
|
||||
if(WITH_SIMD)
|
||||
message(STATUS "SIMD extensions: ${CPU_TYPE} (WITH_SIMD = ${WITH_SIMD})")
|
||||
set(HAVE_LIBJPEG_TURBO_SIMD 1 PARENT_SCOPE)
|
||||
if(MSVC_IDE OR XCODE)
|
||||
set_source_files_properties(${SIMD_OBJS} PROPERTIES GENERATED 1)
|
||||
endif()
|
||||
else()
|
||||
add_library(jsimd OBJECT src/jsimd_none.c)
|
||||
set_target_properties(jsimd PROPERTIES FOLDER "3rdparty")
|
||||
if(NOT WIN32 AND (CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED))
|
||||
set_target_properties(jsimd PROPERTIES POSITION_INDEPENDENT_CODE 1)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
ocv_list_add_prefix(JPEG_SOURCES src/)
|
||||
|
||||
add_library(${JPEG_LIBRARY} STATIC ${OPENCV_3RDPARTY_EXCLUDE_FROM_ALL} ${JPEG_SOURCES} ${SIMD_OBJS})
|
||||
set(JPEG_SOURCES ${JPEG_SOURCES} ${SIMD_OBJS})
|
||||
|
||||
add_library(${JPEG_LIBRARY} STATIC ${OPENCV_3RDPARTY_EXCLUDE_FROM_ALL} ${JPEG_SOURCES} $<TARGET_OBJECTS:jsimd> ${SIMD_OBJS})
|
||||
|
||||
set_target_properties(${JPEG_LIBRARY}
|
||||
PROPERTIES OUTPUT_NAME ${JPEG_LIBRARY}
|
||||
|
||||
Vendored
+1
-1
@@ -91,7 +91,7 @@ best of our understanding.
|
||||
The Modified (3-clause) BSD License
|
||||
===================================
|
||||
|
||||
Copyright (C)2009-2020 D. R. Commander. All Rights Reserved.
|
||||
Copyright (C)2009-2022 D. R. Commander. All Rights Reserved.<br>
|
||||
Copyright (C)2015 Viktor Szathmáry. All Rights Reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
|
||||
Vendored
+1
-14
@@ -128,7 +128,7 @@ with respect to this software, its quality, accuracy, merchantability, or
|
||||
fitness for a particular purpose. This software is provided "AS IS", and you,
|
||||
its user, assume the entire risk as to its quality and accuracy.
|
||||
|
||||
This software is copyright (C) 1991-2016, Thomas G. Lane, Guido Vollbeding.
|
||||
This software is copyright (C) 1991-2020, Thomas G. Lane, Guido Vollbeding.
|
||||
All Rights Reserved except as specified below.
|
||||
|
||||
Permission is hereby granted to use, copy, modify, and distribute this
|
||||
@@ -159,19 +159,6 @@ commercial products, provided that all warranty or liability claims are
|
||||
assumed by the product vendor.
|
||||
|
||||
|
||||
The IJG distribution formerly included code to read and write GIF files.
|
||||
To avoid entanglement with the Unisys LZW patent (now expired), GIF reading
|
||||
support has been removed altogether, and the GIF writer has been simplified
|
||||
to produce "uncompressed GIFs". This technique does not use the LZW
|
||||
algorithm; the resulting GIF files are larger than usual, but are readable
|
||||
by all standard GIF decoders.
|
||||
|
||||
We are required to state that
|
||||
"The Graphics Interchange Format(c) is the Copyright property of
|
||||
CompuServe Incorporated. GIF(sm) is a Service Mark property of
|
||||
CompuServe Incorporated."
|
||||
|
||||
|
||||
REFERENCES
|
||||
==========
|
||||
|
||||
|
||||
Vendored
+1
-1
@@ -3,7 +3,7 @@ Background
|
||||
|
||||
libjpeg-turbo is a JPEG image codec that uses SIMD instructions to accelerate
|
||||
baseline JPEG compression and decompression on x86, x86-64, Arm, PowerPC, and
|
||||
MIPS systems, as well as progressive JPEG compression on x86 and x86-64
|
||||
MIPS systems, as well as progressive JPEG compression on x86, x86-64, and Arm
|
||||
systems. On such systems, libjpeg-turbo is generally 2-6x as fast as libjpeg,
|
||||
all else being equal. On other types of systems, libjpeg-turbo can still
|
||||
outperform libjpeg by a significant amount, by virtue of its highly-optimized
|
||||
|
||||
Vendored
-5
@@ -61,11 +61,6 @@
|
||||
unsigned. */
|
||||
#cmakedefine RIGHT_SHIFT_IS_UNSIGNED 1
|
||||
|
||||
/* Define to 1 if type `char' is unsigned and you are not using gcc. */
|
||||
#ifndef __CHAR_UNSIGNED__
|
||||
#cmakedefine __CHAR_UNSIGNED__ 1
|
||||
#endif
|
||||
|
||||
/* Define to empty if `const' does not conform to ANSI C. */
|
||||
/* #undef const */
|
||||
|
||||
|
||||
-1
@@ -18,7 +18,6 @@
|
||||
#define HAVE_UNSIGNED_SHORT
|
||||
#undef INCOMPLETE_TYPES_BROKEN
|
||||
#undef RIGHT_SHIFT_IS_UNSIGNED
|
||||
#undef __CHAR_UNSIGNED__
|
||||
|
||||
/* Define "boolean" as unsigned char, not int, per Windows custom */
|
||||
#ifndef __RPCNDR_H__ /* don't conflict if rpcndr.h already read */
|
||||
|
||||
+10
@@ -40,3 +40,13 @@
|
||||
#define HAVE_BITSCANFORWARD
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if defined(__has_attribute)
|
||||
#if __has_attribute(fallthrough)
|
||||
#define FALLTHROUGH __attribute__((fallthrough));
|
||||
#else
|
||||
#define FALLTHROUGH
|
||||
#endif
|
||||
#else
|
||||
#define FALLTHROUGH
|
||||
#endif
|
||||
|
||||
+3
-3
@@ -4,8 +4,8 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1994-1998, Thomas G. Lane.
|
||||
* Modified 2003-2010 by Guido Vollbeding.
|
||||
* It was modified by The libjpeg-turbo Project to include only code relevant
|
||||
* to libjpeg-turbo.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -52,7 +52,7 @@ jpeg_CreateCompress(j_compress_ptr cinfo, int version, size_t structsize)
|
||||
{
|
||||
struct jpeg_error_mgr *err = cinfo->err;
|
||||
void *client_data = cinfo->client_data; /* ignore Purify complaint here */
|
||||
MEMZERO(cinfo, sizeof(struct jpeg_compress_struct));
|
||||
memset(cinfo, 0, sizeof(struct jpeg_compress_struct));
|
||||
cinfo->err = err;
|
||||
cinfo->client_data = client_data;
|
||||
}
|
||||
|
||||
Vendored
+6
-6
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Developed 1997-2009 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2015, 2018, D. R. Commander.
|
||||
* Copyright (C) 2015, 2018, 2021-2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -338,14 +338,14 @@ emit_restart(j_compress_ptr cinfo, int restart_num)
|
||||
compptr = cinfo->cur_comp_info[ci];
|
||||
/* DC needs no table for refinement scan */
|
||||
if (cinfo->progressive_mode == 0 || (cinfo->Ss == 0 && cinfo->Ah == 0)) {
|
||||
MEMZERO(entropy->dc_stats[compptr->dc_tbl_no], DC_STAT_BINS);
|
||||
memset(entropy->dc_stats[compptr->dc_tbl_no], 0, DC_STAT_BINS);
|
||||
/* Reset DC predictions to 0 */
|
||||
entropy->last_dc_val[ci] = 0;
|
||||
entropy->dc_context[ci] = 0;
|
||||
}
|
||||
/* AC needs no table when not present */
|
||||
if (cinfo->progressive_mode == 0 || cinfo->Se) {
|
||||
MEMZERO(entropy->ac_stats[compptr->ac_tbl_no], AC_STAT_BINS);
|
||||
memset(entropy->ac_stats[compptr->ac_tbl_no], 0, AC_STAT_BINS);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -836,7 +836,7 @@ start_pass(j_compress_ptr cinfo, boolean gather_statistics)
|
||||
* We are fully adaptive here and need no extra
|
||||
* statistics gathering pass!
|
||||
*/
|
||||
ERREXIT(cinfo, JERR_NOT_COMPILED);
|
||||
ERREXIT(cinfo, JERR_NOTIMPL);
|
||||
|
||||
/* We assume jcmaster.c already validated the progressive scan parameters. */
|
||||
|
||||
@@ -867,7 +867,7 @@ start_pass(j_compress_ptr cinfo, boolean gather_statistics)
|
||||
if (entropy->dc_stats[tbl] == NULL)
|
||||
entropy->dc_stats[tbl] = (unsigned char *)(*cinfo->mem->alloc_small)
|
||||
((j_common_ptr)cinfo, JPOOL_IMAGE, DC_STAT_BINS);
|
||||
MEMZERO(entropy->dc_stats[tbl], DC_STAT_BINS);
|
||||
memset(entropy->dc_stats[tbl], 0, DC_STAT_BINS);
|
||||
/* Initialize DC predictions to 0 */
|
||||
entropy->last_dc_val[ci] = 0;
|
||||
entropy->dc_context[ci] = 0;
|
||||
@@ -880,7 +880,7 @@ start_pass(j_compress_ptr cinfo, boolean gather_statistics)
|
||||
if (entropy->ac_stats[tbl] == NULL)
|
||||
entropy->ac_stats[tbl] = (unsigned char *)(*cinfo->mem->alloc_small)
|
||||
((j_common_ptr)cinfo, JPOOL_IMAGE, AC_STAT_BINS);
|
||||
MEMZERO(entropy->ac_stats[tbl], AC_STAT_BINS);
|
||||
memset(entropy->ac_stats[tbl], 0, AC_STAT_BINS);
|
||||
#ifdef CALCULATE_SPECTRAL_CONDITIONING
|
||||
if (cinfo->progressive_mode)
|
||||
/* Section G.1.3.2: Set appropriate arithmetic conditioning value Kx */
|
||||
|
||||
+9
-9
@@ -48,9 +48,9 @@ rgb_ycc_convert_internal(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
outptr2 = output_buf[2][output_row];
|
||||
output_row++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
r = GETJSAMPLE(inptr[RGB_RED]);
|
||||
g = GETJSAMPLE(inptr[RGB_GREEN]);
|
||||
b = GETJSAMPLE(inptr[RGB_BLUE]);
|
||||
r = inptr[RGB_RED];
|
||||
g = inptr[RGB_GREEN];
|
||||
b = inptr[RGB_BLUE];
|
||||
inptr += RGB_PIXELSIZE;
|
||||
/* If the inputs are 0..MAXJSAMPLE, the outputs of these equations
|
||||
* must be too; we do not need an explicit range-limiting operation.
|
||||
@@ -100,9 +100,9 @@ rgb_gray_convert_internal(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
outptr = output_buf[0][output_row];
|
||||
output_row++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
r = GETJSAMPLE(inptr[RGB_RED]);
|
||||
g = GETJSAMPLE(inptr[RGB_GREEN]);
|
||||
b = GETJSAMPLE(inptr[RGB_BLUE]);
|
||||
r = inptr[RGB_RED];
|
||||
g = inptr[RGB_GREEN];
|
||||
b = inptr[RGB_BLUE];
|
||||
inptr += RGB_PIXELSIZE;
|
||||
/* Y */
|
||||
outptr[col] = (JSAMPLE)((ctab[r + R_Y_OFF] + ctab[g + G_Y_OFF] +
|
||||
@@ -135,9 +135,9 @@ rgb_rgb_convert_internal(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
outptr2 = output_buf[2][output_row];
|
||||
output_row++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
outptr0[col] = GETJSAMPLE(inptr[RGB_RED]);
|
||||
outptr1[col] = GETJSAMPLE(inptr[RGB_GREEN]);
|
||||
outptr2[col] = GETJSAMPLE(inptr[RGB_BLUE]);
|
||||
outptr0[col] = inptr[RGB_RED];
|
||||
outptr1[col] = inptr[RGB_GREEN];
|
||||
outptr2[col] = inptr[RGB_BLUE];
|
||||
inptr += RGB_PIXELSIZE;
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+6
-6
@@ -392,11 +392,11 @@ cmyk_ycck_convert(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
outptr3 = output_buf[3][output_row];
|
||||
output_row++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
r = MAXJSAMPLE - GETJSAMPLE(inptr[0]);
|
||||
g = MAXJSAMPLE - GETJSAMPLE(inptr[1]);
|
||||
b = MAXJSAMPLE - GETJSAMPLE(inptr[2]);
|
||||
r = MAXJSAMPLE - inptr[0];
|
||||
g = MAXJSAMPLE - inptr[1];
|
||||
b = MAXJSAMPLE - inptr[2];
|
||||
/* K passes through as-is */
|
||||
outptr3[col] = inptr[3]; /* don't need GETJSAMPLE here */
|
||||
outptr3[col] = inptr[3];
|
||||
inptr += 4;
|
||||
/* If the inputs are 0..MAXJSAMPLE, the outputs of these equations
|
||||
* must be too; we do not need an explicit range-limiting operation.
|
||||
@@ -438,7 +438,7 @@ grayscale_convert(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
outptr = output_buf[0][output_row];
|
||||
output_row++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
outptr[col] = inptr[0]; /* don't need GETJSAMPLE() here */
|
||||
outptr[col] = inptr[0];
|
||||
inptr += instride;
|
||||
}
|
||||
}
|
||||
@@ -497,7 +497,7 @@ null_convert(j_compress_ptr cinfo, JSAMPARRAY input_buf, JSAMPIMAGE output_buf,
|
||||
inptr = *input_buf;
|
||||
outptr = output_buf[ci][output_row];
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
outptr[col] = inptr[ci]; /* don't need GETJSAMPLE() here */
|
||||
outptr[col] = inptr[ci];
|
||||
inptr += nc;
|
||||
}
|
||||
}
|
||||
|
||||
+18
-19
@@ -381,19 +381,19 @@ convsamp(JSAMPARRAY sample_data, JDIMENSION start_col, DCTELEM *workspace)
|
||||
elemptr = sample_data[elemr] + start_col;
|
||||
|
||||
#if DCTSIZE == 8 /* unroll the inner loop */
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
#else
|
||||
{
|
||||
register int elemc;
|
||||
for (elemc = DCTSIZE; elemc > 0; elemc--)
|
||||
*workspaceptr++ = GETJSAMPLE(*elemptr++) - CENTERJSAMPLE;
|
||||
*workspaceptr++ = (*elemptr++) - CENTERJSAMPLE;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -533,20 +533,19 @@ convsamp_float(JSAMPARRAY sample_data, JDIMENSION start_col,
|
||||
for (elemr = 0; elemr < DCTSIZE; elemr++) {
|
||||
elemptr = sample_data[elemr] + start_col;
|
||||
#if DCTSIZE == 8 /* unroll the inner loop */
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
#else
|
||||
{
|
||||
register int elemc;
|
||||
for (elemc = DCTSIZE; elemc > 0; elemc--)
|
||||
*workspaceptr++ = (FAST_FLOAT)
|
||||
(GETJSAMPLE(*elemptr++) - CENTERJSAMPLE);
|
||||
*workspaceptr++ = (FAST_FLOAT)((*elemptr++) - CENTERJSAMPLE);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
Vendored
+229
-188
@@ -4,8 +4,10 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2009-2011, 2014-2016, 2018-2019, D. R. Commander.
|
||||
* Copyright (C) 2009-2011, 2014-2016, 2018-2022, D. R. Commander.
|
||||
* Copyright (C) 2015, Matthieu Darbois.
|
||||
* Copyright (C) 2018, Matthias Räncker.
|
||||
* Copyright (C) 2020, Arm Limited.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -43,14 +45,19 @@
|
||||
*/
|
||||
|
||||
/* NOTE: Both GCC and Clang define __GNUC__ */
|
||||
#if defined(__GNUC__) && (defined(__arm__) || defined(__aarch64__))
|
||||
#if (defined(__GNUC__) && (defined(__arm__) || defined(__aarch64__))) || \
|
||||
defined(_M_ARM) || defined(_M_ARM64)
|
||||
#if !defined(__thumb__) || defined(__thumb2__)
|
||||
#define USE_CLZ_INTRINSIC
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef USE_CLZ_INTRINSIC
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#define JPEG_NBITS_NONZERO(x) (32 - _CountLeadingZeros(x))
|
||||
#else
|
||||
#define JPEG_NBITS_NONZERO(x) (32 - __builtin_clz(x))
|
||||
#endif
|
||||
#define JPEG_NBITS(x) (x ? JPEG_NBITS_NONZERO(x) : 0)
|
||||
#else
|
||||
#include "jpeg_nbits_table.h"
|
||||
@@ -65,32 +72,43 @@
|
||||
* but must not be updated permanently until we complete the MCU.
|
||||
*/
|
||||
|
||||
#if defined(__x86_64__) && defined(__ILP32__)
|
||||
typedef unsigned long long bit_buf_type;
|
||||
#else
|
||||
typedef size_t bit_buf_type;
|
||||
#endif
|
||||
|
||||
/* NOTE: The more optimal Huffman encoding algorithm is only used by the
|
||||
* intrinsics implementation of the Arm Neon SIMD extensions, which is why we
|
||||
* retain the old Huffman encoder behavior when using the GAS implementation.
|
||||
*/
|
||||
#if defined(WITH_SIMD) && !(defined(__arm__) || defined(__aarch64__) || \
|
||||
defined(_M_ARM) || defined(_M_ARM64))
|
||||
typedef unsigned long long simd_bit_buf_type;
|
||||
#else
|
||||
typedef bit_buf_type simd_bit_buf_type;
|
||||
#endif
|
||||
|
||||
#if (defined(SIZEOF_SIZE_T) && SIZEOF_SIZE_T == 8) || defined(_WIN64) || \
|
||||
(defined(__x86_64__) && defined(__ILP32__))
|
||||
#define BIT_BUF_SIZE 64
|
||||
#elif (defined(SIZEOF_SIZE_T) && SIZEOF_SIZE_T == 4) || defined(_WIN32)
|
||||
#define BIT_BUF_SIZE 32
|
||||
#else
|
||||
#error Cannot determine word size
|
||||
#endif
|
||||
#define SIMD_BIT_BUF_SIZE (sizeof(simd_bit_buf_type) * 8)
|
||||
|
||||
typedef struct {
|
||||
size_t put_buffer; /* current bit-accumulation buffer */
|
||||
int put_bits; /* # of bits now in it */
|
||||
union {
|
||||
bit_buf_type c;
|
||||
simd_bit_buf_type simd;
|
||||
} put_buffer; /* current bit accumulation buffer */
|
||||
int free_bits; /* # of bits available in it */
|
||||
/* (Neon GAS: # of bits now in it) */
|
||||
int last_dc_val[MAX_COMPS_IN_SCAN]; /* last DC coef for each component */
|
||||
} savable_state;
|
||||
|
||||
/* This macro is to work around compilers with missing or broken
|
||||
* structure assignment. You'll need to fix this code if you have
|
||||
* such a compiler and you change MAX_COMPS_IN_SCAN.
|
||||
*/
|
||||
|
||||
#ifndef NO_STRUCT_ASSIGN
|
||||
#define ASSIGN_STATE(dest, src) ((dest) = (src))
|
||||
#else
|
||||
#if MAX_COMPS_IN_SCAN == 4
|
||||
#define ASSIGN_STATE(dest, src) \
|
||||
((dest).put_buffer = (src).put_buffer, \
|
||||
(dest).put_bits = (src).put_bits, \
|
||||
(dest).last_dc_val[0] = (src).last_dc_val[0], \
|
||||
(dest).last_dc_val[1] = (src).last_dc_val[1], \
|
||||
(dest).last_dc_val[2] = (src).last_dc_val[2], \
|
||||
(dest).last_dc_val[3] = (src).last_dc_val[3])
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
typedef struct {
|
||||
struct jpeg_entropy_encoder pub; /* public fields */
|
||||
|
||||
@@ -123,6 +141,7 @@ typedef struct {
|
||||
size_t free_in_buffer; /* # of byte spaces remaining in buffer */
|
||||
savable_state cur; /* Current bit buffer & DC state */
|
||||
j_compress_ptr cinfo; /* dump_buffer needs access to this */
|
||||
int simd;
|
||||
} working_state;
|
||||
|
||||
|
||||
@@ -181,12 +200,12 @@ start_pass_huff(j_compress_ptr cinfo, boolean gather_statistics)
|
||||
entropy->dc_count_ptrs[dctbl] = (long *)
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
257 * sizeof(long));
|
||||
MEMZERO(entropy->dc_count_ptrs[dctbl], 257 * sizeof(long));
|
||||
memset(entropy->dc_count_ptrs[dctbl], 0, 257 * sizeof(long));
|
||||
if (entropy->ac_count_ptrs[actbl] == NULL)
|
||||
entropy->ac_count_ptrs[actbl] = (long *)
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
257 * sizeof(long));
|
||||
MEMZERO(entropy->ac_count_ptrs[actbl], 257 * sizeof(long));
|
||||
memset(entropy->ac_count_ptrs[actbl], 0, 257 * sizeof(long));
|
||||
#endif
|
||||
} else {
|
||||
/* Compute derived values for Huffman tables */
|
||||
@@ -201,8 +220,17 @@ start_pass_huff(j_compress_ptr cinfo, boolean gather_statistics)
|
||||
}
|
||||
|
||||
/* Initialize bit buffer to empty */
|
||||
entropy->saved.put_buffer = 0;
|
||||
entropy->saved.put_bits = 0;
|
||||
if (entropy->simd) {
|
||||
entropy->saved.put_buffer.simd = 0;
|
||||
#if defined(__aarch64__) && !defined(NEON_INTRINSICS)
|
||||
entropy->saved.free_bits = 0;
|
||||
#else
|
||||
entropy->saved.free_bits = SIMD_BIT_BUF_SIZE;
|
||||
#endif
|
||||
} else {
|
||||
entropy->saved.put_buffer.c = 0;
|
||||
entropy->saved.free_bits = BIT_BUF_SIZE;
|
||||
}
|
||||
|
||||
/* Initialize restart stuff */
|
||||
entropy->restarts_to_go = cinfo->restart_interval;
|
||||
@@ -287,7 +315,8 @@ jpeg_make_c_derived_tbl(j_compress_ptr cinfo, boolean isDC, int tblno,
|
||||
* this lets us detect duplicate VAL entries here, and later
|
||||
* allows emit_bits to detect any attempt to emit such symbols.
|
||||
*/
|
||||
MEMZERO(dtbl->ehufsi, sizeof(dtbl->ehufsi));
|
||||
memset(dtbl->ehufco, 0, sizeof(dtbl->ehufco));
|
||||
memset(dtbl->ehufsi, 0, sizeof(dtbl->ehufsi));
|
||||
|
||||
/* This is also a convenient place to check for out-of-range
|
||||
* and duplicated VAL entries. We allow 0..255 for AC symbols
|
||||
@@ -334,94 +363,94 @@ dump_buffer(working_state *state)
|
||||
|
||||
/* Outputting bits to the file */
|
||||
|
||||
/* These macros perform the same task as the emit_bits() function in the
|
||||
* original libjpeg code. In addition to reducing overhead by explicitly
|
||||
* inlining the code, additional performance is achieved by taking into
|
||||
* account the size of the bit buffer and waiting until it is almost full
|
||||
* before emptying it. This mostly benefits 64-bit platforms, since 6
|
||||
* bytes can be stored in a 64-bit bit buffer before it has to be emptied.
|
||||
/* Output byte b and, speculatively, an additional 0 byte. 0xFF must be
|
||||
* encoded as 0xFF 0x00, so the output buffer pointer is advanced by 2 if the
|
||||
* byte is 0xFF. Otherwise, the output buffer pointer is advanced by 1, and
|
||||
* the speculative 0 byte will be overwritten by the next byte.
|
||||
*/
|
||||
|
||||
#define EMIT_BYTE() { \
|
||||
JOCTET c; \
|
||||
put_bits -= 8; \
|
||||
c = (JOCTET)GETJOCTET(put_buffer >> put_bits); \
|
||||
*buffer++ = c; \
|
||||
if (c == 0xFF) /* need to stuff a zero byte? */ \
|
||||
*buffer++ = 0; \
|
||||
#define EMIT_BYTE(b) { \
|
||||
buffer[0] = (JOCTET)(b); \
|
||||
buffer[1] = 0; \
|
||||
buffer -= -2 + ((JOCTET)(b) < 0xFF); \
|
||||
}
|
||||
|
||||
#define PUT_BITS(code, size) { \
|
||||
put_bits += size; \
|
||||
put_buffer = (put_buffer << size) | code; \
|
||||
}
|
||||
/* Output the entire bit buffer. If there are no 0xFF bytes in it, then write
|
||||
* directly to the output buffer. Otherwise, use the EMIT_BYTE() macro to
|
||||
* encode 0xFF as 0xFF 0x00.
|
||||
*/
|
||||
#if BIT_BUF_SIZE == 64
|
||||
|
||||
#if SIZEOF_SIZE_T != 8 && !defined(_WIN64)
|
||||
|
||||
#define CHECKBUF15() { \
|
||||
if (put_bits > 15) { \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
#define FLUSH() { \
|
||||
if (put_buffer & 0x8080808080808080 & ~(put_buffer + 0x0101010101010101)) { \
|
||||
EMIT_BYTE(put_buffer >> 56) \
|
||||
EMIT_BYTE(put_buffer >> 48) \
|
||||
EMIT_BYTE(put_buffer >> 40) \
|
||||
EMIT_BYTE(put_buffer >> 32) \
|
||||
EMIT_BYTE(put_buffer >> 24) \
|
||||
EMIT_BYTE(put_buffer >> 16) \
|
||||
EMIT_BYTE(put_buffer >> 8) \
|
||||
EMIT_BYTE(put_buffer ) \
|
||||
} else { \
|
||||
buffer[0] = (JOCTET)(put_buffer >> 56); \
|
||||
buffer[1] = (JOCTET)(put_buffer >> 48); \
|
||||
buffer[2] = (JOCTET)(put_buffer >> 40); \
|
||||
buffer[3] = (JOCTET)(put_buffer >> 32); \
|
||||
buffer[4] = (JOCTET)(put_buffer >> 24); \
|
||||
buffer[5] = (JOCTET)(put_buffer >> 16); \
|
||||
buffer[6] = (JOCTET)(put_buffer >> 8); \
|
||||
buffer[7] = (JOCTET)(put_buffer); \
|
||||
buffer += 8; \
|
||||
} \
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#define CHECKBUF31() { \
|
||||
if (put_bits > 31) { \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
} \
|
||||
}
|
||||
|
||||
#define CHECKBUF47() { \
|
||||
if (put_bits > 47) { \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
EMIT_BYTE() \
|
||||
} \
|
||||
}
|
||||
|
||||
#if !defined(_WIN32) && !defined(SIZEOF_SIZE_T)
|
||||
#error Cannot determine word size
|
||||
#endif
|
||||
|
||||
#if SIZEOF_SIZE_T == 8 || defined(_WIN64)
|
||||
|
||||
#define EMIT_BITS(code, size) { \
|
||||
CHECKBUF47() \
|
||||
PUT_BITS(code, size) \
|
||||
}
|
||||
|
||||
#define EMIT_CODE(code, size) { \
|
||||
temp2 &= (((JLONG)1) << nbits) - 1; \
|
||||
CHECKBUF31() \
|
||||
PUT_BITS(code, size) \
|
||||
PUT_BITS(temp2, nbits) \
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
#define EMIT_BITS(code, size) { \
|
||||
PUT_BITS(code, size) \
|
||||
CHECKBUF15() \
|
||||
}
|
||||
|
||||
#define EMIT_CODE(code, size) { \
|
||||
temp2 &= (((JLONG)1) << nbits) - 1; \
|
||||
PUT_BITS(code, size) \
|
||||
CHECKBUF15() \
|
||||
PUT_BITS(temp2, nbits) \
|
||||
CHECKBUF15() \
|
||||
#define FLUSH() { \
|
||||
if (put_buffer & 0x80808080 & ~(put_buffer + 0x01010101)) { \
|
||||
EMIT_BYTE(put_buffer >> 24) \
|
||||
EMIT_BYTE(put_buffer >> 16) \
|
||||
EMIT_BYTE(put_buffer >> 8) \
|
||||
EMIT_BYTE(put_buffer ) \
|
||||
} else { \
|
||||
buffer[0] = (JOCTET)(put_buffer >> 24); \
|
||||
buffer[1] = (JOCTET)(put_buffer >> 16); \
|
||||
buffer[2] = (JOCTET)(put_buffer >> 8); \
|
||||
buffer[3] = (JOCTET)(put_buffer); \
|
||||
buffer += 4; \
|
||||
} \
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
/* Fill the bit buffer to capacity with the leading bits from code, then output
|
||||
* the bit buffer and put the remaining bits from code into the bit buffer.
|
||||
*/
|
||||
#define PUT_AND_FLUSH(code, size) { \
|
||||
put_buffer = (put_buffer << (size + free_bits)) | (code >> -free_bits); \
|
||||
FLUSH() \
|
||||
free_bits += BIT_BUF_SIZE; \
|
||||
put_buffer = code; \
|
||||
}
|
||||
|
||||
/* Insert code into the bit buffer and output the bit buffer if needed.
|
||||
* NOTE: We can't flush with free_bits == 0, since the left shift in
|
||||
* PUT_AND_FLUSH() would have undefined behavior.
|
||||
*/
|
||||
#define PUT_BITS(code, size) { \
|
||||
free_bits -= size; \
|
||||
if (free_bits < 0) \
|
||||
PUT_AND_FLUSH(code, size) \
|
||||
else \
|
||||
put_buffer = (put_buffer << size) | code; \
|
||||
}
|
||||
|
||||
#define PUT_CODE(code, size) { \
|
||||
temp &= (((JLONG)1) << nbits) - 1; \
|
||||
temp |= code << nbits; \
|
||||
nbits += size; \
|
||||
PUT_BITS(temp, nbits) \
|
||||
}
|
||||
|
||||
|
||||
/* Although it is exceedingly rare, it is possible for a Huffman-encoded
|
||||
* coefficient block to be larger than the 128-byte unencoded block. For each
|
||||
@@ -444,11 +473,12 @@ dump_buffer(working_state *state)
|
||||
|
||||
#define STORE_BUFFER() { \
|
||||
if (localbuf) { \
|
||||
size_t bytes, bytestocopy; \
|
||||
bytes = buffer - _buffer; \
|
||||
buffer = _buffer; \
|
||||
while (bytes > 0) { \
|
||||
bytestocopy = MIN(bytes, state->free_in_buffer); \
|
||||
MEMCOPY(state->next_output_byte, buffer, bytestocopy); \
|
||||
memcpy(state->next_output_byte, buffer, bytestocopy); \
|
||||
state->next_output_byte += bytestocopy; \
|
||||
buffer += bytestocopy; \
|
||||
state->free_in_buffer -= bytestocopy; \
|
||||
@@ -466,20 +496,46 @@ dump_buffer(working_state *state)
|
||||
LOCAL(boolean)
|
||||
flush_bits(working_state *state)
|
||||
{
|
||||
JOCTET _buffer[BUFSIZE], *buffer;
|
||||
size_t put_buffer; int put_bits;
|
||||
size_t bytes, bytestocopy; int localbuf = 0;
|
||||
JOCTET _buffer[BUFSIZE], *buffer, temp;
|
||||
simd_bit_buf_type put_buffer; int put_bits;
|
||||
int localbuf = 0;
|
||||
|
||||
if (state->simd) {
|
||||
#if defined(__aarch64__) && !defined(NEON_INTRINSICS)
|
||||
put_bits = state->cur.free_bits;
|
||||
#else
|
||||
put_bits = SIMD_BIT_BUF_SIZE - state->cur.free_bits;
|
||||
#endif
|
||||
put_buffer = state->cur.put_buffer.simd;
|
||||
} else {
|
||||
put_bits = BIT_BUF_SIZE - state->cur.free_bits;
|
||||
put_buffer = state->cur.put_buffer.c;
|
||||
}
|
||||
|
||||
put_buffer = state->cur.put_buffer;
|
||||
put_bits = state->cur.put_bits;
|
||||
LOAD_BUFFER()
|
||||
|
||||
/* fill any partial byte with ones */
|
||||
PUT_BITS(0x7F, 7)
|
||||
while (put_bits >= 8) EMIT_BYTE()
|
||||
while (put_bits >= 8) {
|
||||
put_bits -= 8;
|
||||
temp = (JOCTET)(put_buffer >> put_bits);
|
||||
EMIT_BYTE(temp)
|
||||
}
|
||||
if (put_bits) {
|
||||
/* fill partial byte with ones */
|
||||
temp = (JOCTET)((put_buffer << (8 - put_bits)) | (0xFF >> put_bits));
|
||||
EMIT_BYTE(temp)
|
||||
}
|
||||
|
||||
state->cur.put_buffer = 0; /* and reset bit-buffer to empty */
|
||||
state->cur.put_bits = 0;
|
||||
if (state->simd) { /* and reset bit buffer to empty */
|
||||
state->cur.put_buffer.simd = 0;
|
||||
#if defined(__aarch64__) && !defined(NEON_INTRINSICS)
|
||||
state->cur.free_bits = 0;
|
||||
#else
|
||||
state->cur.free_bits = SIMD_BIT_BUF_SIZE;
|
||||
#endif
|
||||
} else {
|
||||
state->cur.put_buffer.c = 0;
|
||||
state->cur.free_bits = BIT_BUF_SIZE;
|
||||
}
|
||||
STORE_BUFFER()
|
||||
|
||||
return TRUE;
|
||||
@@ -493,7 +549,7 @@ encode_one_block_simd(working_state *state, JCOEFPTR block, int last_dc_val,
|
||||
c_derived_tbl *dctbl, c_derived_tbl *actbl)
|
||||
{
|
||||
JOCTET _buffer[BUFSIZE], *buffer;
|
||||
size_t bytes, bytestocopy; int localbuf = 0;
|
||||
int localbuf = 0;
|
||||
|
||||
LOAD_BUFFER()
|
||||
|
||||
@@ -509,53 +565,41 @@ LOCAL(boolean)
|
||||
encode_one_block(working_state *state, JCOEFPTR block, int last_dc_val,
|
||||
c_derived_tbl *dctbl, c_derived_tbl *actbl)
|
||||
{
|
||||
int temp, temp2, temp3;
|
||||
int nbits;
|
||||
int r, code, size;
|
||||
int temp, nbits, free_bits;
|
||||
bit_buf_type put_buffer;
|
||||
JOCTET _buffer[BUFSIZE], *buffer;
|
||||
size_t put_buffer; int put_bits;
|
||||
int code_0xf0 = actbl->ehufco[0xf0], size_0xf0 = actbl->ehufsi[0xf0];
|
||||
size_t bytes, bytestocopy; int localbuf = 0;
|
||||
int localbuf = 0;
|
||||
|
||||
put_buffer = state->cur.put_buffer;
|
||||
put_bits = state->cur.put_bits;
|
||||
free_bits = state->cur.free_bits;
|
||||
put_buffer = state->cur.put_buffer.c;
|
||||
LOAD_BUFFER()
|
||||
|
||||
/* Encode the DC coefficient difference per section F.1.2.1 */
|
||||
|
||||
temp = temp2 = block[0] - last_dc_val;
|
||||
temp = block[0] - last_dc_val;
|
||||
|
||||
/* This is a well-known technique for obtaining the absolute value without a
|
||||
* branch. It is derived from an assembly language technique presented in
|
||||
* "How to Optimize for the Pentium Processors", Copyright (c) 1996, 1997 by
|
||||
* Agner Fog.
|
||||
* Agner Fog. This code assumes we are on a two's complement machine.
|
||||
*/
|
||||
temp3 = temp >> (CHAR_BIT * sizeof(int) - 1);
|
||||
temp ^= temp3;
|
||||
temp -= temp3;
|
||||
|
||||
/* For a negative input, want temp2 = bitwise complement of abs(input) */
|
||||
/* This code assumes we are on a two's complement machine */
|
||||
temp2 += temp3;
|
||||
nbits = temp >> (CHAR_BIT * sizeof(int) - 1);
|
||||
temp += nbits;
|
||||
nbits ^= temp;
|
||||
|
||||
/* Find the number of bits needed for the magnitude of the coefficient */
|
||||
nbits = JPEG_NBITS(temp);
|
||||
nbits = JPEG_NBITS(nbits);
|
||||
|
||||
/* Emit the Huffman-coded symbol for the number of bits */
|
||||
code = dctbl->ehufco[nbits];
|
||||
size = dctbl->ehufsi[nbits];
|
||||
EMIT_BITS(code, size)
|
||||
|
||||
/* Mask off any extra bits in code */
|
||||
temp2 &= (((JLONG)1) << nbits) - 1;
|
||||
|
||||
/* Emit that number of bits of the value, if positive, */
|
||||
/* or the complement of its magnitude, if negative. */
|
||||
EMIT_BITS(temp2, nbits)
|
||||
/* Emit the Huffman-coded symbol for the number of bits.
|
||||
* Emit that number of bits of the value, if positive,
|
||||
* or the complement of its magnitude, if negative.
|
||||
*/
|
||||
PUT_CODE(dctbl->ehufco[nbits], dctbl->ehufsi[nbits])
|
||||
|
||||
/* Encode the AC coefficients per section F.1.2.2 */
|
||||
|
||||
r = 0; /* r = run length of zeros */
|
||||
{
|
||||
int r = 0; /* r = run length of zeros */
|
||||
|
||||
/* Manually unroll the k loop to eliminate the counter variable. This
|
||||
* improves performance greatly on systems with a limited number of
|
||||
@@ -563,51 +607,46 @@ encode_one_block(working_state *state, JCOEFPTR block, int last_dc_val,
|
||||
*/
|
||||
#define kloop(jpeg_natural_order_of_k) { \
|
||||
if ((temp = block[jpeg_natural_order_of_k]) == 0) { \
|
||||
r++; \
|
||||
r += 16; \
|
||||
} else { \
|
||||
temp2 = temp; \
|
||||
/* Branch-less absolute value, bitwise complement, etc., same as above */ \
|
||||
temp3 = temp >> (CHAR_BIT * sizeof(int) - 1); \
|
||||
temp ^= temp3; \
|
||||
temp -= temp3; \
|
||||
temp2 += temp3; \
|
||||
nbits = JPEG_NBITS_NONZERO(temp); \
|
||||
nbits = temp >> (CHAR_BIT * sizeof(int) - 1); \
|
||||
temp += nbits; \
|
||||
nbits ^= temp; \
|
||||
nbits = JPEG_NBITS_NONZERO(nbits); \
|
||||
/* if run length > 15, must emit special run-length-16 codes (0xF0) */ \
|
||||
while (r > 15) { \
|
||||
EMIT_BITS(code_0xf0, size_0xf0) \
|
||||
r -= 16; \
|
||||
while (r >= 16 * 16) { \
|
||||
r -= 16 * 16; \
|
||||
PUT_BITS(actbl->ehufco[0xf0], actbl->ehufsi[0xf0]) \
|
||||
} \
|
||||
/* Emit Huffman symbol for run length / number of bits */ \
|
||||
temp3 = (r << 4) + nbits; \
|
||||
code = actbl->ehufco[temp3]; \
|
||||
size = actbl->ehufsi[temp3]; \
|
||||
EMIT_CODE(code, size) \
|
||||
r += nbits; \
|
||||
PUT_CODE(actbl->ehufco[r], actbl->ehufsi[r]) \
|
||||
r = 0; \
|
||||
} \
|
||||
}
|
||||
|
||||
/* One iteration for each value in jpeg_natural_order[] */
|
||||
kloop(1); kloop(8); kloop(16); kloop(9); kloop(2); kloop(3);
|
||||
kloop(10); kloop(17); kloop(24); kloop(32); kloop(25); kloop(18);
|
||||
kloop(11); kloop(4); kloop(5); kloop(12); kloop(19); kloop(26);
|
||||
kloop(33); kloop(40); kloop(48); kloop(41); kloop(34); kloop(27);
|
||||
kloop(20); kloop(13); kloop(6); kloop(7); kloop(14); kloop(21);
|
||||
kloop(28); kloop(35); kloop(42); kloop(49); kloop(56); kloop(57);
|
||||
kloop(50); kloop(43); kloop(36); kloop(29); kloop(22); kloop(15);
|
||||
kloop(23); kloop(30); kloop(37); kloop(44); kloop(51); kloop(58);
|
||||
kloop(59); kloop(52); kloop(45); kloop(38); kloop(31); kloop(39);
|
||||
kloop(46); kloop(53); kloop(60); kloop(61); kloop(54); kloop(47);
|
||||
kloop(55); kloop(62); kloop(63);
|
||||
/* One iteration for each value in jpeg_natural_order[] */
|
||||
kloop(1); kloop(8); kloop(16); kloop(9); kloop(2); kloop(3);
|
||||
kloop(10); kloop(17); kloop(24); kloop(32); kloop(25); kloop(18);
|
||||
kloop(11); kloop(4); kloop(5); kloop(12); kloop(19); kloop(26);
|
||||
kloop(33); kloop(40); kloop(48); kloop(41); kloop(34); kloop(27);
|
||||
kloop(20); kloop(13); kloop(6); kloop(7); kloop(14); kloop(21);
|
||||
kloop(28); kloop(35); kloop(42); kloop(49); kloop(56); kloop(57);
|
||||
kloop(50); kloop(43); kloop(36); kloop(29); kloop(22); kloop(15);
|
||||
kloop(23); kloop(30); kloop(37); kloop(44); kloop(51); kloop(58);
|
||||
kloop(59); kloop(52); kloop(45); kloop(38); kloop(31); kloop(39);
|
||||
kloop(46); kloop(53); kloop(60); kloop(61); kloop(54); kloop(47);
|
||||
kloop(55); kloop(62); kloop(63);
|
||||
|
||||
/* If the last coef(s) were zero, emit an end-of-block code */
|
||||
if (r > 0) {
|
||||
code = actbl->ehufco[0];
|
||||
size = actbl->ehufsi[0];
|
||||
EMIT_BITS(code, size)
|
||||
/* If the last coef(s) were zero, emit an end-of-block code */
|
||||
if (r > 0) {
|
||||
PUT_BITS(actbl->ehufco[0], actbl->ehufsi[0])
|
||||
}
|
||||
}
|
||||
|
||||
state->cur.put_buffer = put_buffer;
|
||||
state->cur.put_bits = put_bits;
|
||||
state->cur.put_buffer.c = put_buffer;
|
||||
state->cur.free_bits = free_bits;
|
||||
STORE_BUFFER()
|
||||
|
||||
return TRUE;
|
||||
@@ -654,8 +693,9 @@ encode_mcu_huff(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
/* Load up working state */
|
||||
state.next_output_byte = cinfo->dest->next_output_byte;
|
||||
state.free_in_buffer = cinfo->dest->free_in_buffer;
|
||||
ASSIGN_STATE(state.cur, entropy->saved);
|
||||
state.cur = entropy->saved;
|
||||
state.cinfo = cinfo;
|
||||
state.simd = entropy->simd;
|
||||
|
||||
/* Emit restart marker if needed */
|
||||
if (cinfo->restart_interval) {
|
||||
@@ -694,7 +734,7 @@ encode_mcu_huff(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
/* Completed MCU, so update state */
|
||||
cinfo->dest->next_output_byte = state.next_output_byte;
|
||||
cinfo->dest->free_in_buffer = state.free_in_buffer;
|
||||
ASSIGN_STATE(entropy->saved, state.cur);
|
||||
entropy->saved = state.cur;
|
||||
|
||||
/* Update restart-interval state too */
|
||||
if (cinfo->restart_interval) {
|
||||
@@ -723,8 +763,9 @@ finish_pass_huff(j_compress_ptr cinfo)
|
||||
/* Load up working state ... flush_bits needs it */
|
||||
state.next_output_byte = cinfo->dest->next_output_byte;
|
||||
state.free_in_buffer = cinfo->dest->free_in_buffer;
|
||||
ASSIGN_STATE(state.cur, entropy->saved);
|
||||
state.cur = entropy->saved;
|
||||
state.cinfo = cinfo;
|
||||
state.simd = entropy->simd;
|
||||
|
||||
/* Flush out the last data */
|
||||
if (!flush_bits(&state))
|
||||
@@ -733,7 +774,7 @@ finish_pass_huff(j_compress_ptr cinfo)
|
||||
/* Update state */
|
||||
cinfo->dest->next_output_byte = state.next_output_byte;
|
||||
cinfo->dest->free_in_buffer = state.free_in_buffer;
|
||||
ASSIGN_STATE(entropy->saved, state.cur);
|
||||
entropy->saved = state.cur;
|
||||
}
|
||||
|
||||
|
||||
@@ -900,8 +941,8 @@ jpeg_gen_optimal_table(j_compress_ptr cinfo, JHUFF_TBL *htbl, long freq[])
|
||||
|
||||
/* This algorithm is explained in section K.2 of the JPEG standard */
|
||||
|
||||
MEMZERO(bits, sizeof(bits));
|
||||
MEMZERO(codesize, sizeof(codesize));
|
||||
memset(bits, 0, sizeof(bits));
|
||||
memset(codesize, 0, sizeof(codesize));
|
||||
for (i = 0; i < 257; i++)
|
||||
others[i] = -1; /* init links to empty */
|
||||
|
||||
@@ -1003,7 +1044,7 @@ jpeg_gen_optimal_table(j_compress_ptr cinfo, JHUFF_TBL *htbl, long freq[])
|
||||
bits[i]--;
|
||||
|
||||
/* Return final symbol counts (only for lengths 0..16) */
|
||||
MEMCOPY(htbl->bits, bits, sizeof(htbl->bits));
|
||||
memcpy(htbl->bits, bits, sizeof(htbl->bits));
|
||||
|
||||
/* Return a list of the symbols sorted by code length */
|
||||
/* It's not real clear to me why we don't need to consider the codelength
|
||||
@@ -1042,8 +1083,8 @@ finish_pass_gather(j_compress_ptr cinfo)
|
||||
/* It's important not to apply jpeg_gen_optimal_table more than once
|
||||
* per table, because it clobbers the input frequency counts!
|
||||
*/
|
||||
MEMZERO(did_dc, sizeof(did_dc));
|
||||
MEMZERO(did_ac, sizeof(did_ac));
|
||||
memset(did_dc, 0, sizeof(did_dc));
|
||||
memset(did_ac, 0, sizeof(did_ac));
|
||||
|
||||
for (ci = 0; ci < cinfo->comps_in_scan; ci++) {
|
||||
compptr = cinfo->cur_comp_info[ci];
|
||||
|
||||
+1
-1
@@ -493,7 +493,7 @@ prepare_for_pass(j_compress_ptr cinfo)
|
||||
master->pass_type = output_pass;
|
||||
master->pass_number++;
|
||||
#endif
|
||||
/*FALLTHROUGH*/
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case output_pass:
|
||||
/* Do a data-output pass. */
|
||||
/* We need not repeat per-scan setup if prior optimization pass did it. */
|
||||
|
||||
Vendored
+23
-14
@@ -4,8 +4,10 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1995-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2011, 2015, 2018, D. R. Commander.
|
||||
* Copyright (C) 2011, 2015, 2018, 2021-2022, D. R. Commander.
|
||||
* Copyright (C) 2016, 2018, Matthieu Darbois.
|
||||
* Copyright (C) 2020, Arm Limited.
|
||||
* Copyright (C) 2021, Alex Richardson.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -52,14 +54,19 @@
|
||||
*/
|
||||
|
||||
/* NOTE: Both GCC and Clang define __GNUC__ */
|
||||
#if defined(__GNUC__) && (defined(__arm__) || defined(__aarch64__))
|
||||
#if (defined(__GNUC__) && (defined(__arm__) || defined(__aarch64__))) || \
|
||||
defined(_M_ARM) || defined(_M_ARM64)
|
||||
#if !defined(__thumb__) || defined(__thumb2__)
|
||||
#define USE_CLZ_INTRINSIC
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef USE_CLZ_INTRINSIC
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#define JPEG_NBITS_NONZERO(x) (32 - _CountLeadingZeros(x))
|
||||
#else
|
||||
#define JPEG_NBITS_NONZERO(x) (32 - __builtin_clz(x))
|
||||
#endif
|
||||
#define JPEG_NBITS(x) (x ? JPEG_NBITS_NONZERO(x) : 0)
|
||||
#else
|
||||
#include "jpeg_nbits_table.h"
|
||||
@@ -169,24 +176,26 @@ INLINE
|
||||
METHODDEF(int)
|
||||
count_zeroes(size_t *x)
|
||||
{
|
||||
int result;
|
||||
#if defined(HAVE_BUILTIN_CTZL)
|
||||
int result;
|
||||
result = __builtin_ctzl(*x);
|
||||
*x >>= result;
|
||||
#elif defined(HAVE_BITSCANFORWARD64)
|
||||
unsigned long result;
|
||||
_BitScanForward64(&result, *x);
|
||||
*x >>= result;
|
||||
#elif defined(HAVE_BITSCANFORWARD)
|
||||
unsigned long result;
|
||||
_BitScanForward(&result, *x);
|
||||
*x >>= result;
|
||||
#else
|
||||
result = 0;
|
||||
int result = 0;
|
||||
while ((*x & 1) == 0) {
|
||||
++result;
|
||||
*x >>= 1;
|
||||
}
|
||||
#endif
|
||||
return result;
|
||||
return (int)result;
|
||||
}
|
||||
|
||||
|
||||
@@ -266,7 +275,7 @@ start_pass_phuff(j_compress_ptr cinfo, boolean gather_statistics)
|
||||
entropy->count_ptrs[tbl] = (long *)
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
257 * sizeof(long));
|
||||
MEMZERO(entropy->count_ptrs[tbl], 257 * sizeof(long));
|
||||
memset(entropy->count_ptrs[tbl], 0, 257 * sizeof(long));
|
||||
} else {
|
||||
/* Compute derived values for Huffman table */
|
||||
/* We may do this more than once for a table, but it's not expensive */
|
||||
@@ -575,8 +584,8 @@ encode_mcu_DC_first(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
continue; \
|
||||
/* For a negative coef, want temp2 = bitwise complement of abs(coef) */ \
|
||||
temp2 ^= temp; \
|
||||
values[k] = temp; \
|
||||
values[k + DCTSIZE2] = temp2; \
|
||||
values[k] = (JCOEF)temp; \
|
||||
values[k + DCTSIZE2] = (JCOEF)temp2; \
|
||||
zerobits |= ((size_t)1U) << k; \
|
||||
} \
|
||||
}
|
||||
@@ -672,7 +681,7 @@ encode_mcu_AC_first(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
emit_restart(entropy, entropy->next_restart_num);
|
||||
|
||||
#ifdef WITH_SIMD
|
||||
cvalue = values = (JCOEF *)PAD((size_t)values_unaligned, 16);
|
||||
cvalue = values = (JCOEF *)PAD((JUINTPTR)values_unaligned, 16);
|
||||
#else
|
||||
/* Not using SIMD, so alignment is not needed */
|
||||
cvalue = values = values_unaligned;
|
||||
@@ -860,7 +869,7 @@ encode_mcu_AC_refine_prepare(const JCOEF *block,
|
||||
|
||||
#define ENCODE_COEFS_AC_REFINE(label) { \
|
||||
while (zerobits) { \
|
||||
int idx = count_zeroes(&zerobits); \
|
||||
idx = count_zeroes(&zerobits); \
|
||||
r += idx; \
|
||||
cabsvalue += idx; \
|
||||
signbits >>= idx; \
|
||||
@@ -917,7 +926,7 @@ METHODDEF(boolean)
|
||||
encode_mcu_AC_refine(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
{
|
||||
phuff_entropy_ptr entropy = (phuff_entropy_ptr)cinfo->entropy;
|
||||
register int temp, r;
|
||||
register int temp, r, idx;
|
||||
char *BR_buffer;
|
||||
unsigned int BR;
|
||||
int Sl = cinfo->Se - cinfo->Ss + 1;
|
||||
@@ -937,7 +946,7 @@ encode_mcu_AC_refine(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
emit_restart(entropy, entropy->next_restart_num);
|
||||
|
||||
#ifdef WITH_SIMD
|
||||
cabsvalue = absvalues = (JCOEF *)PAD((size_t)absvalues_unaligned, 16);
|
||||
cabsvalue = absvalues = (JCOEF *)PAD((JUINTPTR)absvalues_unaligned, 16);
|
||||
#else
|
||||
/* Not using SIMD, so alignment is not needed */
|
||||
cabsvalue = absvalues = absvalues_unaligned;
|
||||
@@ -968,7 +977,7 @@ encode_mcu_AC_refine(j_compress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
|
||||
if (zerobits) {
|
||||
int diff = ((absvalues + DCTSIZE2 / 2) - cabsvalue);
|
||||
int idx = count_zeroes(&zerobits);
|
||||
idx = count_zeroes(&zerobits);
|
||||
signbits >>= idx;
|
||||
idx += diff;
|
||||
r += idx;
|
||||
@@ -1053,7 +1062,7 @@ finish_pass_gather_phuff(j_compress_ptr cinfo)
|
||||
/* It's important not to apply jpeg_gen_optimal_table more than once
|
||||
* per table, because it clobbers the input frequency counts!
|
||||
*/
|
||||
MEMZERO(did, sizeof(did));
|
||||
memset(did, 0, sizeof(did));
|
||||
|
||||
for (ci = 0; ci < cinfo->comps_in_scan; ci++) {
|
||||
compptr = cinfo->cur_comp_info[ci];
|
||||
|
||||
+4
-4
@@ -3,8 +3,8 @@
|
||||
*
|
||||
* This file is part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1994-1996, Thomas G. Lane.
|
||||
* It was modified by The libjpeg-turbo Project to include only code relevant
|
||||
* to libjpeg-turbo.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -289,8 +289,8 @@ create_context_buffer(j_compress_ptr cinfo)
|
||||
cinfo->max_h_samp_factor) / compptr->h_samp_factor),
|
||||
(JDIMENSION)(3 * rgroup_height));
|
||||
/* Copy true buffer row pointers into the middle of the fake row array */
|
||||
MEMCOPY(fake_buffer + rgroup_height, true_buffer,
|
||||
3 * rgroup_height * sizeof(JSAMPROW));
|
||||
memcpy(fake_buffer + rgroup_height, true_buffer,
|
||||
3 * rgroup_height * sizeof(JSAMPROW));
|
||||
/* Fill in the above and below wraparound pointers */
|
||||
for (i = 0; i < rgroup_height; i++) {
|
||||
fake_buffer[i] = true_buffer[2 * rgroup_height + i];
|
||||
|
||||
+23
-40
@@ -6,7 +6,7 @@
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2014, MIPS Technologies, Inc., California.
|
||||
* Copyright (C) 2015, D. R. Commander.
|
||||
* Copyright (C) 2015, 2019, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -103,7 +103,7 @@ expand_right_edge(JSAMPARRAY image_data, int num_rows, JDIMENSION input_cols,
|
||||
if (numcols > 0) {
|
||||
for (row = 0; row < num_rows; row++) {
|
||||
ptr = image_data[row] + input_cols;
|
||||
pixval = ptr[-1]; /* don't need GETJSAMPLE() here */
|
||||
pixval = ptr[-1];
|
||||
for (count = numcols; count > 0; count--)
|
||||
*ptr++ = pixval;
|
||||
}
|
||||
@@ -174,7 +174,7 @@ int_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
for (v = 0; v < v_expand; v++) {
|
||||
inptr = input_data[inrow + v] + outcol_h;
|
||||
for (h = 0; h < h_expand; h++) {
|
||||
outvalue += (JLONG)GETJSAMPLE(*inptr++);
|
||||
outvalue += (JLONG)(*inptr++);
|
||||
}
|
||||
}
|
||||
*outptr++ = (JSAMPLE)((outvalue + numpix2) / numpix);
|
||||
@@ -237,8 +237,7 @@ h2v1_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
inptr = input_data[outrow];
|
||||
bias = 0; /* bias = 0,1,0,1,... for successive samples */
|
||||
for (outcol = 0; outcol < output_cols; outcol++) {
|
||||
*outptr++ =
|
||||
(JSAMPLE)((GETJSAMPLE(*inptr) + GETJSAMPLE(inptr[1]) + bias) >> 1);
|
||||
*outptr++ = (JSAMPLE)((inptr[0] + inptr[1] + bias) >> 1);
|
||||
bias ^= 1; /* 0=>1, 1=>0 */
|
||||
inptr += 2;
|
||||
}
|
||||
@@ -277,8 +276,7 @@ h2v2_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
bias = 1; /* bias = 1,2,1,2,... for successive samples */
|
||||
for (outcol = 0; outcol < output_cols; outcol++) {
|
||||
*outptr++ =
|
||||
(JSAMPLE)((GETJSAMPLE(*inptr0) + GETJSAMPLE(inptr0[1]) +
|
||||
GETJSAMPLE(*inptr1) + GETJSAMPLE(inptr1[1]) + bias) >> 2);
|
||||
(JSAMPLE)((inptr0[0] + inptr0[1] + inptr1[0] + inptr1[1] + bias) >> 2);
|
||||
bias ^= 3; /* 1=>2, 2=>1 */
|
||||
inptr0 += 2; inptr1 += 2;
|
||||
}
|
||||
@@ -337,33 +335,25 @@ h2v2_smooth_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
below_ptr = input_data[inrow + 2];
|
||||
|
||||
/* Special case for first column: pretend column -1 is same as column 0 */
|
||||
membersum = GETJSAMPLE(*inptr0) + GETJSAMPLE(inptr0[1]) +
|
||||
GETJSAMPLE(*inptr1) + GETJSAMPLE(inptr1[1]);
|
||||
neighsum = GETJSAMPLE(*above_ptr) + GETJSAMPLE(above_ptr[1]) +
|
||||
GETJSAMPLE(*below_ptr) + GETJSAMPLE(below_ptr[1]) +
|
||||
GETJSAMPLE(*inptr0) + GETJSAMPLE(inptr0[2]) +
|
||||
GETJSAMPLE(*inptr1) + GETJSAMPLE(inptr1[2]);
|
||||
membersum = inptr0[0] + inptr0[1] + inptr1[0] + inptr1[1];
|
||||
neighsum = above_ptr[0] + above_ptr[1] + below_ptr[0] + below_ptr[1] +
|
||||
inptr0[0] + inptr0[2] + inptr1[0] + inptr1[2];
|
||||
neighsum += neighsum;
|
||||
neighsum += GETJSAMPLE(*above_ptr) + GETJSAMPLE(above_ptr[2]) +
|
||||
GETJSAMPLE(*below_ptr) + GETJSAMPLE(below_ptr[2]);
|
||||
neighsum += above_ptr[0] + above_ptr[2] + below_ptr[0] + below_ptr[2];
|
||||
membersum = membersum * memberscale + neighsum * neighscale;
|
||||
*outptr++ = (JSAMPLE)((membersum + 32768) >> 16);
|
||||
inptr0 += 2; inptr1 += 2; above_ptr += 2; below_ptr += 2;
|
||||
|
||||
for (colctr = output_cols - 2; colctr > 0; colctr--) {
|
||||
/* sum of pixels directly mapped to this output element */
|
||||
membersum = GETJSAMPLE(*inptr0) + GETJSAMPLE(inptr0[1]) +
|
||||
GETJSAMPLE(*inptr1) + GETJSAMPLE(inptr1[1]);
|
||||
membersum = inptr0[0] + inptr0[1] + inptr1[0] + inptr1[1];
|
||||
/* sum of edge-neighbor pixels */
|
||||
neighsum = GETJSAMPLE(*above_ptr) + GETJSAMPLE(above_ptr[1]) +
|
||||
GETJSAMPLE(*below_ptr) + GETJSAMPLE(below_ptr[1]) +
|
||||
GETJSAMPLE(inptr0[-1]) + GETJSAMPLE(inptr0[2]) +
|
||||
GETJSAMPLE(inptr1[-1]) + GETJSAMPLE(inptr1[2]);
|
||||
neighsum = above_ptr[0] + above_ptr[1] + below_ptr[0] + below_ptr[1] +
|
||||
inptr0[-1] + inptr0[2] + inptr1[-1] + inptr1[2];
|
||||
/* The edge-neighbors count twice as much as corner-neighbors */
|
||||
neighsum += neighsum;
|
||||
/* Add in the corner-neighbors */
|
||||
neighsum += GETJSAMPLE(above_ptr[-1]) + GETJSAMPLE(above_ptr[2]) +
|
||||
GETJSAMPLE(below_ptr[-1]) + GETJSAMPLE(below_ptr[2]);
|
||||
neighsum += above_ptr[-1] + above_ptr[2] + below_ptr[-1] + below_ptr[2];
|
||||
/* form final output scaled up by 2^16 */
|
||||
membersum = membersum * memberscale + neighsum * neighscale;
|
||||
/* round, descale and output it */
|
||||
@@ -372,15 +362,11 @@ h2v2_smooth_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
}
|
||||
|
||||
/* Special case for last column */
|
||||
membersum = GETJSAMPLE(*inptr0) + GETJSAMPLE(inptr0[1]) +
|
||||
GETJSAMPLE(*inptr1) + GETJSAMPLE(inptr1[1]);
|
||||
neighsum = GETJSAMPLE(*above_ptr) + GETJSAMPLE(above_ptr[1]) +
|
||||
GETJSAMPLE(*below_ptr) + GETJSAMPLE(below_ptr[1]) +
|
||||
GETJSAMPLE(inptr0[-1]) + GETJSAMPLE(inptr0[1]) +
|
||||
GETJSAMPLE(inptr1[-1]) + GETJSAMPLE(inptr1[1]);
|
||||
membersum = inptr0[0] + inptr0[1] + inptr1[0] + inptr1[1];
|
||||
neighsum = above_ptr[0] + above_ptr[1] + below_ptr[0] + below_ptr[1] +
|
||||
inptr0[-1] + inptr0[1] + inptr1[-1] + inptr1[1];
|
||||
neighsum += neighsum;
|
||||
neighsum += GETJSAMPLE(above_ptr[-1]) + GETJSAMPLE(above_ptr[1]) +
|
||||
GETJSAMPLE(below_ptr[-1]) + GETJSAMPLE(below_ptr[1]);
|
||||
neighsum += above_ptr[-1] + above_ptr[1] + below_ptr[-1] + below_ptr[1];
|
||||
membersum = membersum * memberscale + neighsum * neighscale;
|
||||
*outptr = (JSAMPLE)((membersum + 32768) >> 16);
|
||||
|
||||
@@ -429,21 +415,18 @@ fullsize_smooth_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
below_ptr = input_data[outrow + 1];
|
||||
|
||||
/* Special case for first column */
|
||||
colsum = GETJSAMPLE(*above_ptr++) + GETJSAMPLE(*below_ptr++) +
|
||||
GETJSAMPLE(*inptr);
|
||||
membersum = GETJSAMPLE(*inptr++);
|
||||
nextcolsum = GETJSAMPLE(*above_ptr) + GETJSAMPLE(*below_ptr) +
|
||||
GETJSAMPLE(*inptr);
|
||||
colsum = (*above_ptr++) + (*below_ptr++) + inptr[0];
|
||||
membersum = *inptr++;
|
||||
nextcolsum = above_ptr[0] + below_ptr[0] + inptr[0];
|
||||
neighsum = colsum + (colsum - membersum) + nextcolsum;
|
||||
membersum = membersum * memberscale + neighsum * neighscale;
|
||||
*outptr++ = (JSAMPLE)((membersum + 32768) >> 16);
|
||||
lastcolsum = colsum; colsum = nextcolsum;
|
||||
|
||||
for (colctr = output_cols - 2; colctr > 0; colctr--) {
|
||||
membersum = GETJSAMPLE(*inptr++);
|
||||
membersum = *inptr++;
|
||||
above_ptr++; below_ptr++;
|
||||
nextcolsum = GETJSAMPLE(*above_ptr) + GETJSAMPLE(*below_ptr) +
|
||||
GETJSAMPLE(*inptr);
|
||||
nextcolsum = above_ptr[0] + below_ptr[0] + inptr[0];
|
||||
neighsum = lastcolsum + (colsum - membersum) + nextcolsum;
|
||||
membersum = membersum * memberscale + neighsum * neighscale;
|
||||
*outptr++ = (JSAMPLE)((membersum + 32768) >> 16);
|
||||
@@ -451,7 +434,7 @@ fullsize_smooth_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
}
|
||||
|
||||
/* Special case for last column */
|
||||
membersum = GETJSAMPLE(*inptr);
|
||||
membersum = *inptr;
|
||||
neighsum = lastcolsum + (colsum - membersum) + colsum;
|
||||
membersum = membersum * memberscale + neighsum * neighscale;
|
||||
*outptr = (JSAMPLE)((membersum + 32768) >> 16);
|
||||
|
||||
Vendored
+3
-3
@@ -5,7 +5,7 @@
|
||||
* Copyright (C) 1995-1998, Thomas G. Lane.
|
||||
* Modified 2000-2009 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2020, D. R. Commander.
|
||||
* Copyright (C) 2020, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -100,8 +100,8 @@ jpeg_copy_critical_parameters(j_decompress_ptr srcinfo, j_compress_ptr dstinfo)
|
||||
qtblptr = &dstinfo->quant_tbl_ptrs[tblno];
|
||||
if (*qtblptr == NULL)
|
||||
*qtblptr = jpeg_alloc_quant_table((j_common_ptr)dstinfo);
|
||||
MEMCOPY((*qtblptr)->quantval, srcinfo->quant_tbl_ptrs[tblno]->quantval,
|
||||
sizeof((*qtblptr)->quantval));
|
||||
memcpy((*qtblptr)->quantval, srcinfo->quant_tbl_ptrs[tblno]->quantval,
|
||||
sizeof((*qtblptr)->quantval));
|
||||
(*qtblptr)->sent_table = FALSE;
|
||||
}
|
||||
}
|
||||
|
||||
+5
-4
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1994-1998, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2016, D. R. Commander.
|
||||
* Copyright (C) 2016, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -23,6 +23,7 @@
|
||||
#include "jinclude.h"
|
||||
#include "jpeglib.h"
|
||||
#include "jdmaster.h"
|
||||
#include "jconfigint.h"
|
||||
|
||||
|
||||
/*
|
||||
@@ -52,7 +53,7 @@ jpeg_CreateDecompress(j_decompress_ptr cinfo, int version, size_t structsize)
|
||||
{
|
||||
struct jpeg_error_mgr *err = cinfo->err;
|
||||
void *client_data = cinfo->client_data; /* ignore Purify complaint here */
|
||||
MEMZERO(cinfo, sizeof(struct jpeg_decompress_struct));
|
||||
memset(cinfo, 0, sizeof(struct jpeg_decompress_struct));
|
||||
cinfo->err = err;
|
||||
cinfo->client_data = client_data;
|
||||
}
|
||||
@@ -91,7 +92,7 @@ jpeg_CreateDecompress(j_decompress_ptr cinfo, int version, size_t structsize)
|
||||
cinfo->master = (struct jpeg_decomp_master *)
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_PERMANENT,
|
||||
sizeof(my_decomp_master));
|
||||
MEMZERO(cinfo->master, sizeof(my_decomp_master));
|
||||
memset(cinfo->master, 0, sizeof(my_decomp_master));
|
||||
}
|
||||
|
||||
|
||||
@@ -308,7 +309,7 @@ jpeg_consume_input(j_decompress_ptr cinfo)
|
||||
/* Initialize application's data source module */
|
||||
(*cinfo->src->init_source) (cinfo);
|
||||
cinfo->global_state = DSTATE_INHEADER;
|
||||
/*FALLTHROUGH*/
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case DSTATE_INHEADER:
|
||||
retcode = (*cinfo->inputctl->consume_input) (cinfo);
|
||||
if (retcode == JPEG_REACHED_SOS) { /* Found SOS, prepare to decompress */
|
||||
|
||||
+15
-1
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1994-1996, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2010, 2015-2018, 2020, D. R. Commander.
|
||||
* Copyright (C) 2010, 2015-2020, 2022, D. R. Commander.
|
||||
* Copyright (C) 2015, Google, Inc.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
@@ -159,6 +159,7 @@ jpeg_crop_scanline(j_decompress_ptr cinfo, JDIMENSION *xoffset,
|
||||
JDIMENSION input_xoffset;
|
||||
boolean reinit_upsampler = FALSE;
|
||||
jpeg_component_info *compptr;
|
||||
my_master_ptr master = (my_master_ptr)cinfo->master;
|
||||
|
||||
if (cinfo->global_state != DSTATE_SCANNING || cinfo->output_scanline != 0)
|
||||
ERREXIT1(cinfo, JERR_BAD_STATE, cinfo->global_state);
|
||||
@@ -208,6 +209,11 @@ jpeg_crop_scanline(j_decompress_ptr cinfo, JDIMENSION *xoffset,
|
||||
*/
|
||||
*width = *width + input_xoffset - *xoffset;
|
||||
cinfo->output_width = *width;
|
||||
if (master->using_merged_upsample && cinfo->max_v_samp_factor == 2) {
|
||||
my_merged_upsample_ptr upsample = (my_merged_upsample_ptr)cinfo->upsample;
|
||||
upsample->out_row_width =
|
||||
cinfo->output_width * cinfo->out_color_components;
|
||||
}
|
||||
|
||||
/* Set the first and last iMCU columns that we must decompress. These values
|
||||
* will be used in single-scan decompressions.
|
||||
@@ -319,6 +325,8 @@ read_and_discard_scanlines(j_decompress_ptr cinfo, JDIMENSION num_lines)
|
||||
{
|
||||
JDIMENSION n;
|
||||
my_master_ptr master = (my_master_ptr)cinfo->master;
|
||||
JSAMPLE dummy_sample[1] = { 0 };
|
||||
JSAMPROW dummy_row = dummy_sample;
|
||||
JSAMPARRAY scanlines = NULL;
|
||||
void (*color_convert) (j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
JDIMENSION input_row, JSAMPARRAY output_buf,
|
||||
@@ -329,6 +337,10 @@ read_and_discard_scanlines(j_decompress_ptr cinfo, JDIMENSION num_lines)
|
||||
if (cinfo->cconvert && cinfo->cconvert->color_convert) {
|
||||
color_convert = cinfo->cconvert->color_convert;
|
||||
cinfo->cconvert->color_convert = noop_convert;
|
||||
/* This just prevents UBSan from complaining about adding 0 to a NULL
|
||||
* pointer. The pointer isn't actually used.
|
||||
*/
|
||||
scanlines = &dummy_row;
|
||||
}
|
||||
|
||||
if (cinfo->cquantize && cinfo->cquantize->color_quantize) {
|
||||
@@ -532,6 +544,8 @@ jpeg_skip_scanlines(j_decompress_ptr cinfo, JDIMENSION num_lines)
|
||||
* decoded coefficients. This is ~5% faster for large subsets, but
|
||||
* it's tough to tell a difference for smaller images.
|
||||
*/
|
||||
if (!cinfo->entropy->insufficient_data)
|
||||
cinfo->master->last_good_iMCU_row = cinfo->input_iMCU_row;
|
||||
(*cinfo->entropy->decode_mcu) (cinfo, NULL);
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+22
-13
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Developed 1997-2015 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2015-2018, D. R. Commander.
|
||||
* Copyright (C) 2015-2020, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -80,7 +80,7 @@ get_byte(j_decompress_ptr cinfo)
|
||||
if (!(*src->fill_input_buffer) (cinfo))
|
||||
ERREXIT(cinfo, JERR_CANT_SUSPEND);
|
||||
src->bytes_in_buffer--;
|
||||
return GETJOCTET(*src->next_input_byte++);
|
||||
return *src->next_input_byte++;
|
||||
}
|
||||
|
||||
|
||||
@@ -210,13 +210,13 @@ process_restart(j_decompress_ptr cinfo)
|
||||
for (ci = 0; ci < cinfo->comps_in_scan; ci++) {
|
||||
compptr = cinfo->cur_comp_info[ci];
|
||||
if (!cinfo->progressive_mode || (cinfo->Ss == 0 && cinfo->Ah == 0)) {
|
||||
MEMZERO(entropy->dc_stats[compptr->dc_tbl_no], DC_STAT_BINS);
|
||||
memset(entropy->dc_stats[compptr->dc_tbl_no], 0, DC_STAT_BINS);
|
||||
/* Reset DC predictions to 0 */
|
||||
entropy->last_dc_val[ci] = 0;
|
||||
entropy->dc_context[ci] = 0;
|
||||
}
|
||||
if (!cinfo->progressive_mode || cinfo->Ss) {
|
||||
MEMZERO(entropy->ac_stats[compptr->ac_tbl_no], AC_STAT_BINS);
|
||||
memset(entropy->ac_stats[compptr->ac_tbl_no], 0, AC_STAT_BINS);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -471,17 +471,17 @@ decode_mcu_AC_refine(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
if (*thiscoef) { /* previously nonzero coef */
|
||||
if (arith_decode(cinfo, st + 2)) {
|
||||
if (*thiscoef < 0)
|
||||
*thiscoef += m1;
|
||||
*thiscoef += (JCOEF)m1;
|
||||
else
|
||||
*thiscoef += p1;
|
||||
*thiscoef += (JCOEF)p1;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (arith_decode(cinfo, st + 1)) { /* newly nonzero coef */
|
||||
if (arith_decode(cinfo, entropy->fixed_bin))
|
||||
*thiscoef = m1;
|
||||
*thiscoef = (JCOEF)m1;
|
||||
else
|
||||
*thiscoef = p1;
|
||||
*thiscoef = (JCOEF)p1;
|
||||
break;
|
||||
}
|
||||
st += 3; k++;
|
||||
@@ -665,8 +665,16 @@ bad:
|
||||
for (ci = 0; ci < cinfo->comps_in_scan; ci++) {
|
||||
int coefi, cindex = cinfo->cur_comp_info[ci]->component_index;
|
||||
int *coef_bit_ptr = &cinfo->coef_bits[cindex][0];
|
||||
int *prev_coef_bit_ptr =
|
||||
&cinfo->coef_bits[cindex + cinfo->num_components][0];
|
||||
if (cinfo->Ss && coef_bit_ptr[0] < 0) /* AC without prior DC scan */
|
||||
WARNMS2(cinfo, JWRN_BOGUS_PROGRESSION, cindex, 0);
|
||||
for (coefi = MIN(cinfo->Ss, 1); coefi <= MAX(cinfo->Se, 9); coefi++) {
|
||||
if (cinfo->input_scan_number > 1)
|
||||
prev_coef_bit_ptr[coefi] = coef_bit_ptr[coefi];
|
||||
else
|
||||
prev_coef_bit_ptr[coefi] = 0;
|
||||
}
|
||||
for (coefi = cinfo->Ss; coefi <= cinfo->Se; coefi++) {
|
||||
int expected = (coef_bit_ptr[coefi] < 0) ? 0 : coef_bit_ptr[coefi];
|
||||
if (cinfo->Ah != expected)
|
||||
@@ -690,8 +698,8 @@ bad:
|
||||
/* Check that the scan parameters Ss, Se, Ah/Al are OK for sequential JPEG.
|
||||
* This ought to be an error condition, but we make it a warning.
|
||||
*/
|
||||
if (cinfo->Ss != 0 || cinfo->Ah != 0 || cinfo->Al != 0 ||
|
||||
(cinfo->Se < DCTSIZE2 && cinfo->Se != DCTSIZE2 - 1))
|
||||
if (cinfo->Ss != 0 || cinfo->Se != DCTSIZE2 - 1 ||
|
||||
cinfo->Ah != 0 || cinfo->Al != 0)
|
||||
WARNMS(cinfo, JWRN_NOT_SEQUENTIAL);
|
||||
/* Select MCU decoding routine */
|
||||
entropy->pub.decode_mcu = decode_mcu;
|
||||
@@ -707,7 +715,7 @@ bad:
|
||||
if (entropy->dc_stats[tbl] == NULL)
|
||||
entropy->dc_stats[tbl] = (unsigned char *)(*cinfo->mem->alloc_small)
|
||||
((j_common_ptr)cinfo, JPOOL_IMAGE, DC_STAT_BINS);
|
||||
MEMZERO(entropy->dc_stats[tbl], DC_STAT_BINS);
|
||||
memset(entropy->dc_stats[tbl], 0, DC_STAT_BINS);
|
||||
/* Initialize DC predictions to 0 */
|
||||
entropy->last_dc_val[ci] = 0;
|
||||
entropy->dc_context[ci] = 0;
|
||||
@@ -719,7 +727,7 @@ bad:
|
||||
if (entropy->ac_stats[tbl] == NULL)
|
||||
entropy->ac_stats[tbl] = (unsigned char *)(*cinfo->mem->alloc_small)
|
||||
((j_common_ptr)cinfo, JPOOL_IMAGE, AC_STAT_BINS);
|
||||
MEMZERO(entropy->ac_stats[tbl], AC_STAT_BINS);
|
||||
memset(entropy->ac_stats[tbl], 0, AC_STAT_BINS);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -727,6 +735,7 @@ bad:
|
||||
entropy->c = 0;
|
||||
entropy->a = 0;
|
||||
entropy->ct = -16; /* force reading 2 initial bytes to fill C */
|
||||
entropy->pub.insufficient_data = FALSE;
|
||||
|
||||
/* Initialize restart counter */
|
||||
entropy->restarts_to_go = cinfo->restart_interval;
|
||||
@@ -763,7 +772,7 @@ jinit_arith_decoder(j_decompress_ptr cinfo)
|
||||
int *coef_bit_ptr, ci;
|
||||
cinfo->coef_bits = (int (*)[DCTSIZE2])
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
cinfo->num_components * DCTSIZE2 *
|
||||
cinfo->num_components * 2 * DCTSIZE2 *
|
||||
sizeof(int));
|
||||
coef_bit_ptr = &cinfo->coef_bits[0][0];
|
||||
for (ci = 0; ci < cinfo->num_components; ci++)
|
||||
|
||||
+4
-9
@@ -5,7 +5,7 @@
|
||||
* Copyright (C) 1994-1996, Thomas G. Lane.
|
||||
* Modified 2009-2012 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2013, 2016, D. R. Commander.
|
||||
* Copyright (C) 2013, 2016, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -23,11 +23,6 @@
|
||||
#include "jpeglib.h"
|
||||
#include "jerror.h"
|
||||
|
||||
#ifndef HAVE_STDLIB_H /* <stdlib.h> should declare malloc(),free() */
|
||||
extern void *malloc(size_t size);
|
||||
extern void free(void *ptr);
|
||||
#endif
|
||||
|
||||
|
||||
/* Expanded data destination object for stdio output */
|
||||
|
||||
@@ -116,7 +111,7 @@ empty_output_buffer(j_compress_ptr cinfo)
|
||||
{
|
||||
my_dest_ptr dest = (my_dest_ptr)cinfo->dest;
|
||||
|
||||
if (JFWRITE(dest->outfile, dest->buffer, OUTPUT_BUF_SIZE) !=
|
||||
if (fwrite(dest->buffer, 1, OUTPUT_BUF_SIZE, dest->outfile) !=
|
||||
(size_t)OUTPUT_BUF_SIZE)
|
||||
ERREXIT(cinfo, JERR_FILE_WRITE);
|
||||
|
||||
@@ -141,7 +136,7 @@ empty_mem_output_buffer(j_compress_ptr cinfo)
|
||||
if (nextbuffer == NULL)
|
||||
ERREXIT1(cinfo, JERR_OUT_OF_MEMORY, 10);
|
||||
|
||||
MEMCOPY(nextbuffer, dest->buffer, dest->bufsize);
|
||||
memcpy(nextbuffer, dest->buffer, dest->bufsize);
|
||||
|
||||
free(dest->newbuffer);
|
||||
|
||||
@@ -175,7 +170,7 @@ term_destination(j_compress_ptr cinfo)
|
||||
|
||||
/* Write any data remaining in the buffer */
|
||||
if (datacount > 0) {
|
||||
if (JFWRITE(dest->outfile, dest->buffer, datacount) != datacount)
|
||||
if (fwrite(dest->buffer, 1, datacount, dest->outfile) != datacount)
|
||||
ERREXIT(cinfo, JERR_FILE_WRITE);
|
||||
}
|
||||
fflush(dest->outfile);
|
||||
|
||||
+2
-2
@@ -5,7 +5,7 @@
|
||||
* Copyright (C) 1994-1996, Thomas G. Lane.
|
||||
* Modified 2009-2011 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2013, 2016, D. R. Commander.
|
||||
* Copyright (C) 2013, 2016, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -104,7 +104,7 @@ fill_input_buffer(j_decompress_ptr cinfo)
|
||||
my_src_ptr src = (my_src_ptr)cinfo->src;
|
||||
size_t nbytes;
|
||||
|
||||
nbytes = JFREAD(src->infile, src->buffer, INPUT_BUF_SIZE);
|
||||
nbytes = fread(src->buffer, 1, INPUT_BUF_SIZE, src->infile);
|
||||
|
||||
if (nbytes <= 0) {
|
||||
if (src->start_of_file) /* Treat empty input file as fatal error */
|
||||
|
||||
+237
-53
@@ -5,7 +5,7 @@
|
||||
* Copyright (C) 1994-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2010, 2015-2016, D. R. Commander.
|
||||
* Copyright (C) 2010, 2015-2016, 2019-2020, D. R. Commander.
|
||||
* Copyright (C) 2015, 2020, Google, Inc.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
@@ -102,6 +102,8 @@ decompress_onepass(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
/* Try to fetch an MCU. Entropy decoder expects buffer to be zeroed. */
|
||||
jzero_far((void *)coef->MCU_buffer[0],
|
||||
(size_t)(cinfo->blocks_in_MCU * sizeof(JBLOCK)));
|
||||
if (!cinfo->entropy->insufficient_data)
|
||||
cinfo->master->last_good_iMCU_row = cinfo->input_iMCU_row;
|
||||
if (!(*cinfo->entropy->decode_mcu) (cinfo, coef->MCU_buffer)) {
|
||||
/* Suspension forced; update state counters and exit */
|
||||
coef->MCU_vert_offset = yoffset;
|
||||
@@ -227,6 +229,8 @@ consume_data(j_decompress_ptr cinfo)
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!cinfo->entropy->insufficient_data)
|
||||
cinfo->master->last_good_iMCU_row = cinfo->input_iMCU_row;
|
||||
/* Try to fetch the MCU. */
|
||||
if (!(*cinfo->entropy->decode_mcu) (cinfo, coef->MCU_buffer)) {
|
||||
/* Suspension forced; update state counters and exit */
|
||||
@@ -326,19 +330,22 @@ decompress_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
#ifdef BLOCK_SMOOTHING_SUPPORTED
|
||||
|
||||
/*
|
||||
* This code applies interblock smoothing as described by section K.8
|
||||
* of the JPEG standard: the first 5 AC coefficients are estimated from
|
||||
* the DC values of a DCT block and its 8 neighboring blocks.
|
||||
* This code applies interblock smoothing; the first 9 AC coefficients are
|
||||
* estimated from the DC values of a DCT block and its 24 neighboring blocks.
|
||||
* We apply smoothing only for progressive JPEG decoding, and only if
|
||||
* the coefficients it can estimate are not yet known to full precision.
|
||||
*/
|
||||
|
||||
/* Natural-order array positions of the first 5 zigzag-order coefficients */
|
||||
/* Natural-order array positions of the first 9 zigzag-order coefficients */
|
||||
#define Q01_POS 1
|
||||
#define Q10_POS 8
|
||||
#define Q20_POS 16
|
||||
#define Q11_POS 9
|
||||
#define Q02_POS 2
|
||||
#define Q03_POS 3
|
||||
#define Q12_POS 10
|
||||
#define Q21_POS 17
|
||||
#define Q30_POS 24
|
||||
|
||||
/*
|
||||
* Determine whether block smoothing is applicable and safe.
|
||||
@@ -356,8 +363,8 @@ smoothing_ok(j_decompress_ptr cinfo)
|
||||
int ci, coefi;
|
||||
jpeg_component_info *compptr;
|
||||
JQUANT_TBL *qtable;
|
||||
int *coef_bits;
|
||||
int *coef_bits_latch;
|
||||
int *coef_bits, *prev_coef_bits;
|
||||
int *coef_bits_latch, *prev_coef_bits_latch;
|
||||
|
||||
if (!cinfo->progressive_mode || cinfo->coef_bits == NULL)
|
||||
return FALSE;
|
||||
@@ -366,34 +373,47 @@ smoothing_ok(j_decompress_ptr cinfo)
|
||||
if (coef->coef_bits_latch == NULL)
|
||||
coef->coef_bits_latch = (int *)
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
cinfo->num_components *
|
||||
cinfo->num_components * 2 *
|
||||
(SAVED_COEFS * sizeof(int)));
|
||||
coef_bits_latch = coef->coef_bits_latch;
|
||||
prev_coef_bits_latch =
|
||||
&coef->coef_bits_latch[cinfo->num_components * SAVED_COEFS];
|
||||
|
||||
for (ci = 0, compptr = cinfo->comp_info; ci < cinfo->num_components;
|
||||
ci++, compptr++) {
|
||||
/* All components' quantization values must already be latched. */
|
||||
if ((qtable = compptr->quant_table) == NULL)
|
||||
return FALSE;
|
||||
/* Verify DC & first 5 AC quantizers are nonzero to avoid zero-divide. */
|
||||
/* Verify DC & first 9 AC quantizers are nonzero to avoid zero-divide. */
|
||||
if (qtable->quantval[0] == 0 ||
|
||||
qtable->quantval[Q01_POS] == 0 ||
|
||||
qtable->quantval[Q10_POS] == 0 ||
|
||||
qtable->quantval[Q20_POS] == 0 ||
|
||||
qtable->quantval[Q11_POS] == 0 ||
|
||||
qtable->quantval[Q02_POS] == 0)
|
||||
qtable->quantval[Q02_POS] == 0 ||
|
||||
qtable->quantval[Q03_POS] == 0 ||
|
||||
qtable->quantval[Q12_POS] == 0 ||
|
||||
qtable->quantval[Q21_POS] == 0 ||
|
||||
qtable->quantval[Q30_POS] == 0)
|
||||
return FALSE;
|
||||
/* DC values must be at least partly known for all components. */
|
||||
coef_bits = cinfo->coef_bits[ci];
|
||||
prev_coef_bits = cinfo->coef_bits[ci + cinfo->num_components];
|
||||
if (coef_bits[0] < 0)
|
||||
return FALSE;
|
||||
coef_bits_latch[0] = coef_bits[0];
|
||||
/* Block smoothing is helpful if some AC coefficients remain inaccurate. */
|
||||
for (coefi = 1; coefi <= 5; coefi++) {
|
||||
for (coefi = 1; coefi < SAVED_COEFS; coefi++) {
|
||||
if (cinfo->input_scan_number > 1)
|
||||
prev_coef_bits_latch[coefi] = prev_coef_bits[coefi];
|
||||
else
|
||||
prev_coef_bits_latch[coefi] = -1;
|
||||
coef_bits_latch[coefi] = coef_bits[coefi];
|
||||
if (coef_bits[coefi] != 0)
|
||||
smoothing_useful = TRUE;
|
||||
}
|
||||
coef_bits_latch += SAVED_COEFS;
|
||||
prev_coef_bits_latch += SAVED_COEFS;
|
||||
}
|
||||
|
||||
return smoothing_useful;
|
||||
@@ -412,17 +432,20 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
JDIMENSION block_num, last_block_column;
|
||||
int ci, block_row, block_rows, access_rows;
|
||||
JBLOCKARRAY buffer;
|
||||
JBLOCKROW buffer_ptr, prev_block_row, next_block_row;
|
||||
JBLOCKROW buffer_ptr, prev_prev_block_row, prev_block_row;
|
||||
JBLOCKROW next_block_row, next_next_block_row;
|
||||
JSAMPARRAY output_ptr;
|
||||
JDIMENSION output_col;
|
||||
jpeg_component_info *compptr;
|
||||
inverse_DCT_method_ptr inverse_DCT;
|
||||
boolean first_row, last_row;
|
||||
boolean change_dc;
|
||||
JCOEF *workspace;
|
||||
int *coef_bits;
|
||||
JQUANT_TBL *quanttbl;
|
||||
JLONG Q00, Q01, Q02, Q10, Q11, Q20, num;
|
||||
int DC1, DC2, DC3, DC4, DC5, DC6, DC7, DC8, DC9;
|
||||
JLONG Q00, Q01, Q02, Q03 = 0, Q10, Q11, Q12 = 0, Q20, Q21 = 0, Q30 = 0, num;
|
||||
int DC01, DC02, DC03, DC04, DC05, DC06, DC07, DC08, DC09, DC10, DC11, DC12,
|
||||
DC13, DC14, DC15, DC16, DC17, DC18, DC19, DC20, DC21, DC22, DC23, DC24,
|
||||
DC25;
|
||||
int Al, pred;
|
||||
|
||||
/* Keep a local variable to avoid looking it up more than once */
|
||||
@@ -434,10 +457,10 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
if (cinfo->input_scan_number == cinfo->output_scan_number) {
|
||||
/* If input is working on current scan, we ordinarily want it to
|
||||
* have completed the current row. But if input scan is DC,
|
||||
* we want it to keep one row ahead so that next block row's DC
|
||||
* we want it to keep two rows ahead so that next two block rows' DC
|
||||
* values are up to date.
|
||||
*/
|
||||
JDIMENSION delta = (cinfo->Ss == 0) ? 1 : 0;
|
||||
JDIMENSION delta = (cinfo->Ss == 0) ? 2 : 0;
|
||||
if (cinfo->input_iMCU_row > cinfo->output_iMCU_row + delta)
|
||||
break;
|
||||
}
|
||||
@@ -452,34 +475,53 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
if (!compptr->component_needed)
|
||||
continue;
|
||||
/* Count non-dummy DCT block rows in this iMCU row. */
|
||||
if (cinfo->output_iMCU_row < last_iMCU_row) {
|
||||
if (cinfo->output_iMCU_row < last_iMCU_row - 1) {
|
||||
block_rows = compptr->v_samp_factor;
|
||||
access_rows = block_rows * 3; /* this and next two iMCU rows */
|
||||
} else if (cinfo->output_iMCU_row < last_iMCU_row) {
|
||||
block_rows = compptr->v_samp_factor;
|
||||
access_rows = block_rows * 2; /* this and next iMCU row */
|
||||
last_row = FALSE;
|
||||
} else {
|
||||
/* NB: can't use last_row_height here; it is input-side-dependent! */
|
||||
block_rows = (int)(compptr->height_in_blocks % compptr->v_samp_factor);
|
||||
if (block_rows == 0) block_rows = compptr->v_samp_factor;
|
||||
access_rows = block_rows; /* this iMCU row only */
|
||||
last_row = TRUE;
|
||||
}
|
||||
/* Align the virtual buffer for this component. */
|
||||
if (cinfo->output_iMCU_row > 0) {
|
||||
access_rows += compptr->v_samp_factor; /* prior iMCU row too */
|
||||
if (cinfo->output_iMCU_row > 1) {
|
||||
access_rows += 2 * compptr->v_samp_factor; /* prior two iMCU rows too */
|
||||
buffer = (*cinfo->mem->access_virt_barray)
|
||||
((j_common_ptr)cinfo, coef->whole_image[ci],
|
||||
(cinfo->output_iMCU_row - 2) * compptr->v_samp_factor,
|
||||
(JDIMENSION)access_rows, FALSE);
|
||||
buffer += 2 * compptr->v_samp_factor; /* point to current iMCU row */
|
||||
} else if (cinfo->output_iMCU_row > 0) {
|
||||
buffer = (*cinfo->mem->access_virt_barray)
|
||||
((j_common_ptr)cinfo, coef->whole_image[ci],
|
||||
(cinfo->output_iMCU_row - 1) * compptr->v_samp_factor,
|
||||
(JDIMENSION)access_rows, FALSE);
|
||||
buffer += compptr->v_samp_factor; /* point to current iMCU row */
|
||||
first_row = FALSE;
|
||||
} else {
|
||||
buffer = (*cinfo->mem->access_virt_barray)
|
||||
((j_common_ptr)cinfo, coef->whole_image[ci],
|
||||
(JDIMENSION)0, (JDIMENSION)access_rows, FALSE);
|
||||
first_row = TRUE;
|
||||
}
|
||||
/* Fetch component-dependent info */
|
||||
coef_bits = coef->coef_bits_latch + (ci * SAVED_COEFS);
|
||||
/* Fetch component-dependent info.
|
||||
* If the current scan is incomplete, then we use the component-dependent
|
||||
* info from the previous scan.
|
||||
*/
|
||||
if (cinfo->output_iMCU_row > cinfo->master->last_good_iMCU_row)
|
||||
coef_bits =
|
||||
coef->coef_bits_latch + ((ci + cinfo->num_components) * SAVED_COEFS);
|
||||
else
|
||||
coef_bits = coef->coef_bits_latch + (ci * SAVED_COEFS);
|
||||
|
||||
/* We only do DC interpolation if no AC coefficient data is available. */
|
||||
change_dc =
|
||||
coef_bits[1] == -1 && coef_bits[2] == -1 && coef_bits[3] == -1 &&
|
||||
coef_bits[4] == -1 && coef_bits[5] == -1 && coef_bits[6] == -1 &&
|
||||
coef_bits[7] == -1 && coef_bits[8] == -1 && coef_bits[9] == -1;
|
||||
|
||||
quanttbl = compptr->quant_table;
|
||||
Q00 = quanttbl->quantval[0];
|
||||
Q01 = quanttbl->quantval[Q01_POS];
|
||||
@@ -487,27 +529,51 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
Q20 = quanttbl->quantval[Q20_POS];
|
||||
Q11 = quanttbl->quantval[Q11_POS];
|
||||
Q02 = quanttbl->quantval[Q02_POS];
|
||||
if (change_dc) {
|
||||
Q03 = quanttbl->quantval[Q03_POS];
|
||||
Q12 = quanttbl->quantval[Q12_POS];
|
||||
Q21 = quanttbl->quantval[Q21_POS];
|
||||
Q30 = quanttbl->quantval[Q30_POS];
|
||||
}
|
||||
inverse_DCT = cinfo->idct->inverse_DCT[ci];
|
||||
output_ptr = output_buf[ci];
|
||||
/* Loop over all DCT blocks to be processed. */
|
||||
for (block_row = 0; block_row < block_rows; block_row++) {
|
||||
buffer_ptr = buffer[block_row] + cinfo->master->first_MCU_col[ci];
|
||||
if (first_row && block_row == 0)
|
||||
|
||||
if (block_row > 0 || cinfo->output_iMCU_row > 0)
|
||||
prev_block_row =
|
||||
buffer[block_row - 1] + cinfo->master->first_MCU_col[ci];
|
||||
else
|
||||
prev_block_row = buffer_ptr;
|
||||
|
||||
if (block_row > 1 || cinfo->output_iMCU_row > 1)
|
||||
prev_prev_block_row =
|
||||
buffer[block_row - 2] + cinfo->master->first_MCU_col[ci];
|
||||
else
|
||||
prev_prev_block_row = prev_block_row;
|
||||
|
||||
if (block_row < block_rows - 1 || cinfo->output_iMCU_row < last_iMCU_row)
|
||||
next_block_row =
|
||||
buffer[block_row + 1] + cinfo->master->first_MCU_col[ci];
|
||||
else
|
||||
prev_block_row = buffer[block_row - 1] +
|
||||
cinfo->master->first_MCU_col[ci];
|
||||
if (last_row && block_row == block_rows - 1)
|
||||
next_block_row = buffer_ptr;
|
||||
|
||||
if (block_row < block_rows - 2 ||
|
||||
cinfo->output_iMCU_row < last_iMCU_row - 1)
|
||||
next_next_block_row =
|
||||
buffer[block_row + 2] + cinfo->master->first_MCU_col[ci];
|
||||
else
|
||||
next_block_row = buffer[block_row + 1] +
|
||||
cinfo->master->first_MCU_col[ci];
|
||||
next_next_block_row = next_block_row;
|
||||
|
||||
/* We fetch the surrounding DC values using a sliding-register approach.
|
||||
* Initialize all nine here so as to do the right thing on narrow pics.
|
||||
* Initialize all 25 here so as to do the right thing on narrow pics.
|
||||
*/
|
||||
DC1 = DC2 = DC3 = (int)prev_block_row[0][0];
|
||||
DC4 = DC5 = DC6 = (int)buffer_ptr[0][0];
|
||||
DC7 = DC8 = DC9 = (int)next_block_row[0][0];
|
||||
DC01 = DC02 = DC03 = DC04 = DC05 = (int)prev_prev_block_row[0][0];
|
||||
DC06 = DC07 = DC08 = DC09 = DC10 = (int)prev_block_row[0][0];
|
||||
DC11 = DC12 = DC13 = DC14 = DC15 = (int)buffer_ptr[0][0];
|
||||
DC16 = DC17 = DC18 = DC19 = DC20 = (int)next_block_row[0][0];
|
||||
DC21 = DC22 = DC23 = DC24 = DC25 = (int)next_next_block_row[0][0];
|
||||
output_col = 0;
|
||||
last_block_column = compptr->width_in_blocks - 1;
|
||||
for (block_num = cinfo->master->first_MCU_col[ci];
|
||||
@@ -515,18 +581,39 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
/* Fetch current DCT block into workspace so we can modify it. */
|
||||
jcopy_block_row(buffer_ptr, (JBLOCKROW)workspace, (JDIMENSION)1);
|
||||
/* Update DC values */
|
||||
if (block_num < last_block_column) {
|
||||
DC3 = (int)prev_block_row[1][0];
|
||||
DC6 = (int)buffer_ptr[1][0];
|
||||
DC9 = (int)next_block_row[1][0];
|
||||
if (block_num == cinfo->master->first_MCU_col[ci] &&
|
||||
block_num < last_block_column) {
|
||||
DC04 = (int)prev_prev_block_row[1][0];
|
||||
DC09 = (int)prev_block_row[1][0];
|
||||
DC14 = (int)buffer_ptr[1][0];
|
||||
DC19 = (int)next_block_row[1][0];
|
||||
DC24 = (int)next_next_block_row[1][0];
|
||||
}
|
||||
/* Compute coefficient estimates per K.8.
|
||||
* An estimate is applied only if coefficient is still zero,
|
||||
* and is not known to be fully accurate.
|
||||
if (block_num + 1 < last_block_column) {
|
||||
DC05 = (int)prev_prev_block_row[2][0];
|
||||
DC10 = (int)prev_block_row[2][0];
|
||||
DC15 = (int)buffer_ptr[2][0];
|
||||
DC20 = (int)next_block_row[2][0];
|
||||
DC25 = (int)next_next_block_row[2][0];
|
||||
}
|
||||
/* If DC interpolation is enabled, compute coefficient estimates using
|
||||
* a Gaussian-like kernel, keeping the averages of the DC values.
|
||||
*
|
||||
* If DC interpolation is disabled, compute coefficient estimates using
|
||||
* an algorithm similar to the one described in Section K.8 of the JPEG
|
||||
* standard, except applied to a 5x5 window rather than a 3x3 window.
|
||||
*
|
||||
* An estimate is applied only if the coefficient is still zero and is
|
||||
* not known to be fully accurate.
|
||||
*/
|
||||
/* AC01 */
|
||||
if ((Al = coef_bits[1]) != 0 && workspace[1] == 0) {
|
||||
num = 36 * Q00 * (DC4 - DC6);
|
||||
num = Q00 * (change_dc ?
|
||||
(-DC01 - DC02 + DC04 + DC05 - 3 * DC06 + 13 * DC07 -
|
||||
13 * DC09 + 3 * DC10 - 3 * DC11 + 38 * DC12 - 38 * DC14 +
|
||||
3 * DC15 - 3 * DC16 + 13 * DC17 - 13 * DC19 + 3 * DC20 -
|
||||
DC21 - DC22 + DC24 + DC25) :
|
||||
(-7 * DC11 + 50 * DC12 - 50 * DC14 + 7 * DC15));
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q01 << 7) + num) / (Q01 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
@@ -541,7 +628,12 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
}
|
||||
/* AC10 */
|
||||
if ((Al = coef_bits[2]) != 0 && workspace[8] == 0) {
|
||||
num = 36 * Q00 * (DC2 - DC8);
|
||||
num = Q00 * (change_dc ?
|
||||
(-DC01 - 3 * DC02 - 3 * DC03 - 3 * DC04 - DC05 - DC06 +
|
||||
13 * DC07 + 38 * DC08 + 13 * DC09 - DC10 + DC16 -
|
||||
13 * DC17 - 38 * DC18 - 13 * DC19 + DC20 + DC21 +
|
||||
3 * DC22 + 3 * DC23 + 3 * DC24 + DC25) :
|
||||
(-7 * DC03 + 50 * DC08 - 50 * DC18 + 7 * DC23));
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q10 << 7) + num) / (Q10 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
@@ -556,7 +648,10 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
}
|
||||
/* AC20 */
|
||||
if ((Al = coef_bits[3]) != 0 && workspace[16] == 0) {
|
||||
num = 9 * Q00 * (DC2 + DC8 - 2 * DC5);
|
||||
num = Q00 * (change_dc ?
|
||||
(DC03 + 2 * DC07 + 7 * DC08 + 2 * DC09 - 5 * DC12 - 14 * DC13 -
|
||||
5 * DC14 + 2 * DC17 + 7 * DC18 + 2 * DC19 + DC23) :
|
||||
(-DC03 + 13 * DC08 - 24 * DC13 + 13 * DC18 - DC23));
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q20 << 7) + num) / (Q20 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
@@ -571,7 +666,11 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
}
|
||||
/* AC11 */
|
||||
if ((Al = coef_bits[4]) != 0 && workspace[9] == 0) {
|
||||
num = 5 * Q00 * (DC1 - DC3 - DC7 + DC9);
|
||||
num = Q00 * (change_dc ?
|
||||
(-DC01 + DC05 + 9 * DC07 - 9 * DC09 - 9 * DC17 +
|
||||
9 * DC19 + DC21 - DC25) :
|
||||
(DC10 + DC16 - 10 * DC17 + 10 * DC19 - DC02 - DC20 + DC22 -
|
||||
DC24 + DC04 - DC06 + 10 * DC07 - 10 * DC09));
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q11 << 7) + num) / (Q11 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
@@ -586,7 +685,10 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
}
|
||||
/* AC02 */
|
||||
if ((Al = coef_bits[5]) != 0 && workspace[2] == 0) {
|
||||
num = 9 * Q00 * (DC4 + DC6 - 2 * DC5);
|
||||
num = Q00 * (change_dc ?
|
||||
(2 * DC07 - 5 * DC08 + 2 * DC09 + DC11 + 7 * DC12 - 14 * DC13 +
|
||||
7 * DC14 + DC15 + 2 * DC17 - 5 * DC18 + 2 * DC19) :
|
||||
(-DC11 + 13 * DC12 - 24 * DC13 + 13 * DC14 - DC15));
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q02 << 7) + num) / (Q02 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
@@ -599,14 +701,96 @@ decompress_smooth_data(j_decompress_ptr cinfo, JSAMPIMAGE output_buf)
|
||||
}
|
||||
workspace[2] = (JCOEF)pred;
|
||||
}
|
||||
if (change_dc) {
|
||||
/* AC03 */
|
||||
if ((Al = coef_bits[6]) != 0 && workspace[3] == 0) {
|
||||
num = Q00 * (DC07 - DC09 + 2 * DC12 - 2 * DC14 + DC17 - DC19);
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q03 << 7) + num) / (Q03 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
} else {
|
||||
pred = (int)(((Q03 << 7) - num) / (Q03 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
pred = -pred;
|
||||
}
|
||||
workspace[3] = (JCOEF)pred;
|
||||
}
|
||||
/* AC12 */
|
||||
if ((Al = coef_bits[7]) != 0 && workspace[10] == 0) {
|
||||
num = Q00 * (DC07 - 3 * DC08 + DC09 - DC17 + 3 * DC18 - DC19);
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q12 << 7) + num) / (Q12 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
} else {
|
||||
pred = (int)(((Q12 << 7) - num) / (Q12 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
pred = -pred;
|
||||
}
|
||||
workspace[10] = (JCOEF)pred;
|
||||
}
|
||||
/* AC21 */
|
||||
if ((Al = coef_bits[8]) != 0 && workspace[17] == 0) {
|
||||
num = Q00 * (DC07 - DC09 - 3 * DC12 + 3 * DC14 + DC17 - DC19);
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q21 << 7) + num) / (Q21 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
} else {
|
||||
pred = (int)(((Q21 << 7) - num) / (Q21 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
pred = -pred;
|
||||
}
|
||||
workspace[17] = (JCOEF)pred;
|
||||
}
|
||||
/* AC30 */
|
||||
if ((Al = coef_bits[9]) != 0 && workspace[24] == 0) {
|
||||
num = Q00 * (DC07 + 2 * DC08 + DC09 - DC17 - 2 * DC18 - DC19);
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q30 << 7) + num) / (Q30 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
} else {
|
||||
pred = (int)(((Q30 << 7) - num) / (Q30 << 8));
|
||||
if (Al > 0 && pred >= (1 << Al))
|
||||
pred = (1 << Al) - 1;
|
||||
pred = -pred;
|
||||
}
|
||||
workspace[24] = (JCOEF)pred;
|
||||
}
|
||||
/* coef_bits[0] is non-negative. Otherwise this function would not
|
||||
* be called.
|
||||
*/
|
||||
num = Q00 *
|
||||
(-2 * DC01 - 6 * DC02 - 8 * DC03 - 6 * DC04 - 2 * DC05 -
|
||||
6 * DC06 + 6 * DC07 + 42 * DC08 + 6 * DC09 - 6 * DC10 -
|
||||
8 * DC11 + 42 * DC12 + 152 * DC13 + 42 * DC14 - 8 * DC15 -
|
||||
6 * DC16 + 6 * DC17 + 42 * DC18 + 6 * DC19 - 6 * DC20 -
|
||||
2 * DC21 - 6 * DC22 - 8 * DC23 - 6 * DC24 - 2 * DC25);
|
||||
if (num >= 0) {
|
||||
pred = (int)(((Q00 << 7) + num) / (Q00 << 8));
|
||||
} else {
|
||||
pred = (int)(((Q00 << 7) - num) / (Q00 << 8));
|
||||
pred = -pred;
|
||||
}
|
||||
workspace[0] = (JCOEF)pred;
|
||||
} /* change_dc */
|
||||
|
||||
/* OK, do the IDCT */
|
||||
(*inverse_DCT) (cinfo, compptr, (JCOEFPTR)workspace, output_ptr,
|
||||
output_col);
|
||||
/* Advance for next column */
|
||||
DC1 = DC2; DC2 = DC3;
|
||||
DC4 = DC5; DC5 = DC6;
|
||||
DC7 = DC8; DC8 = DC9;
|
||||
buffer_ptr++, prev_block_row++, next_block_row++;
|
||||
DC01 = DC02; DC02 = DC03; DC03 = DC04; DC04 = DC05;
|
||||
DC06 = DC07; DC07 = DC08; DC08 = DC09; DC09 = DC10;
|
||||
DC11 = DC12; DC12 = DC13; DC13 = DC14; DC14 = DC15;
|
||||
DC16 = DC17; DC17 = DC18; DC18 = DC19; DC19 = DC20;
|
||||
DC21 = DC22; DC22 = DC23; DC23 = DC24; DC24 = DC25;
|
||||
buffer_ptr++, prev_block_row++, next_block_row++,
|
||||
prev_prev_block_row++, next_next_block_row++;
|
||||
output_col += compptr->_DCT_scaled_size;
|
||||
}
|
||||
output_ptr += compptr->_DCT_scaled_size;
|
||||
@@ -655,7 +839,7 @@ jinit_d_coef_controller(j_decompress_ptr cinfo, boolean need_full_buffer)
|
||||
#ifdef BLOCK_SMOOTHING_SUPPORTED
|
||||
/* If block smoothing could be used, need a bigger window */
|
||||
if (cinfo->progressive_mode)
|
||||
access_rows *= 3;
|
||||
access_rows *= 5;
|
||||
#endif
|
||||
coef->whole_image[ci] = (*cinfo->mem->request_virt_barray)
|
||||
((j_common_ptr)cinfo, JPOOL_IMAGE, TRUE,
|
||||
|
||||
+2
-1
@@ -5,6 +5,7 @@
|
||||
* Copyright (C) 1994-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2020, Google, Inc.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*/
|
||||
@@ -51,7 +52,7 @@ typedef struct {
|
||||
#ifdef BLOCK_SMOOTHING_SUPPORTED
|
||||
/* When doing block smoothing, we latch coefficient Al values here */
|
||||
int *coef_bits_latch;
|
||||
#define SAVED_COEFS 6 /* we save coef_bits[0..5] */
|
||||
#define SAVED_COEFS 10 /* we save coef_bits[0..9] */
|
||||
#endif
|
||||
} my_coef_controller;
|
||||
|
||||
|
||||
+48
-48
@@ -45,9 +45,9 @@ ycc_rgb565_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr = *output_buf++;
|
||||
|
||||
if (PACK_NEED_ALIGNMENT(outptr)) {
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
y = *inptr0++;
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
r = range_limit[y + Crrtab[cr]];
|
||||
g = range_limit[y + ((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
SCALEBITS))];
|
||||
@@ -58,18 +58,18 @@ ycc_rgb565_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
num_cols--;
|
||||
}
|
||||
for (col = 0; col < (num_cols >> 1); col++) {
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
y = *inptr0++;
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
r = range_limit[y + Crrtab[cr]];
|
||||
g = range_limit[y + ((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
SCALEBITS))];
|
||||
b = range_limit[y + Cbbtab[cb]];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
y = *inptr0++;
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
r = range_limit[y + Crrtab[cr]];
|
||||
g = range_limit[y + ((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
SCALEBITS))];
|
||||
@@ -80,9 +80,9 @@ ycc_rgb565_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr += 4;
|
||||
}
|
||||
if (num_cols & 1) {
|
||||
y = GETJSAMPLE(*inptr0);
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
y = *inptr0;
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
r = range_limit[y + Crrtab[cr]];
|
||||
g = range_limit[y + ((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
SCALEBITS))];
|
||||
@@ -125,9 +125,9 @@ ycc_rgb565D_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
if (PACK_NEED_ALIGNMENT(outptr)) {
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
y = *inptr0++;
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
r = range_limit[DITHER_565_R(y + Crrtab[cr], d0)];
|
||||
g = range_limit[DITHER_565_G(y +
|
||||
((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
@@ -139,9 +139,9 @@ ycc_rgb565D_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
num_cols--;
|
||||
}
|
||||
for (col = 0; col < (num_cols >> 1); col++) {
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
y = *inptr0++;
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
r = range_limit[DITHER_565_R(y + Crrtab[cr], d0)];
|
||||
g = range_limit[DITHER_565_G(y +
|
||||
((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
@@ -150,9 +150,9 @@ ycc_rgb565D_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
d0 = DITHER_ROTATE(d0);
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
y = *inptr0++;
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
r = range_limit[DITHER_565_R(y + Crrtab[cr], d0)];
|
||||
g = range_limit[DITHER_565_G(y +
|
||||
((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
@@ -165,9 +165,9 @@ ycc_rgb565D_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr += 4;
|
||||
}
|
||||
if (num_cols & 1) {
|
||||
y = GETJSAMPLE(*inptr0);
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
y = *inptr0;
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
r = range_limit[DITHER_565_R(y + Crrtab[cr], d0)];
|
||||
g = range_limit[DITHER_565_G(y +
|
||||
((int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr],
|
||||
@@ -202,32 +202,32 @@ rgb_rgb565_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
if (PACK_NEED_ALIGNMENT(outptr)) {
|
||||
r = GETJSAMPLE(*inptr0++);
|
||||
g = GETJSAMPLE(*inptr1++);
|
||||
b = GETJSAMPLE(*inptr2++);
|
||||
r = *inptr0++;
|
||||
g = *inptr1++;
|
||||
b = *inptr2++;
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
*(INT16 *)outptr = (INT16)rgb;
|
||||
outptr += 2;
|
||||
num_cols--;
|
||||
}
|
||||
for (col = 0; col < (num_cols >> 1); col++) {
|
||||
r = GETJSAMPLE(*inptr0++);
|
||||
g = GETJSAMPLE(*inptr1++);
|
||||
b = GETJSAMPLE(*inptr2++);
|
||||
r = *inptr0++;
|
||||
g = *inptr1++;
|
||||
b = *inptr2++;
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
r = GETJSAMPLE(*inptr0++);
|
||||
g = GETJSAMPLE(*inptr1++);
|
||||
b = GETJSAMPLE(*inptr2++);
|
||||
r = *inptr0++;
|
||||
g = *inptr1++;
|
||||
b = *inptr2++;
|
||||
rgb = PACK_TWO_PIXELS(rgb, PACK_SHORT_565(r, g, b));
|
||||
|
||||
WRITE_TWO_ALIGNED_PIXELS(outptr, rgb);
|
||||
outptr += 4;
|
||||
}
|
||||
if (num_cols & 1) {
|
||||
r = GETJSAMPLE(*inptr0);
|
||||
g = GETJSAMPLE(*inptr1);
|
||||
b = GETJSAMPLE(*inptr2);
|
||||
r = *inptr0;
|
||||
g = *inptr1;
|
||||
b = *inptr2;
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
*(INT16 *)outptr = (INT16)rgb;
|
||||
}
|
||||
@@ -259,24 +259,24 @@ rgb_rgb565D_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
if (PACK_NEED_ALIGNMENT(outptr)) {
|
||||
r = range_limit[DITHER_565_R(GETJSAMPLE(*inptr0++), d0)];
|
||||
g = range_limit[DITHER_565_G(GETJSAMPLE(*inptr1++), d0)];
|
||||
b = range_limit[DITHER_565_B(GETJSAMPLE(*inptr2++), d0)];
|
||||
r = range_limit[DITHER_565_R(*inptr0++, d0)];
|
||||
g = range_limit[DITHER_565_G(*inptr1++, d0)];
|
||||
b = range_limit[DITHER_565_B(*inptr2++, d0)];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
*(INT16 *)outptr = (INT16)rgb;
|
||||
outptr += 2;
|
||||
num_cols--;
|
||||
}
|
||||
for (col = 0; col < (num_cols >> 1); col++) {
|
||||
r = range_limit[DITHER_565_R(GETJSAMPLE(*inptr0++), d0)];
|
||||
g = range_limit[DITHER_565_G(GETJSAMPLE(*inptr1++), d0)];
|
||||
b = range_limit[DITHER_565_B(GETJSAMPLE(*inptr2++), d0)];
|
||||
r = range_limit[DITHER_565_R(*inptr0++, d0)];
|
||||
g = range_limit[DITHER_565_G(*inptr1++, d0)];
|
||||
b = range_limit[DITHER_565_B(*inptr2++, d0)];
|
||||
d0 = DITHER_ROTATE(d0);
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
r = range_limit[DITHER_565_R(GETJSAMPLE(*inptr0++), d0)];
|
||||
g = range_limit[DITHER_565_G(GETJSAMPLE(*inptr1++), d0)];
|
||||
b = range_limit[DITHER_565_B(GETJSAMPLE(*inptr2++), d0)];
|
||||
r = range_limit[DITHER_565_R(*inptr0++, d0)];
|
||||
g = range_limit[DITHER_565_G(*inptr1++, d0)];
|
||||
b = range_limit[DITHER_565_B(*inptr2++, d0)];
|
||||
d0 = DITHER_ROTATE(d0);
|
||||
rgb = PACK_TWO_PIXELS(rgb, PACK_SHORT_565(r, g, b));
|
||||
|
||||
@@ -284,9 +284,9 @@ rgb_rgb565D_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr += 4;
|
||||
}
|
||||
if (num_cols & 1) {
|
||||
r = range_limit[DITHER_565_R(GETJSAMPLE(*inptr0), d0)];
|
||||
g = range_limit[DITHER_565_G(GETJSAMPLE(*inptr1), d0)];
|
||||
b = range_limit[DITHER_565_B(GETJSAMPLE(*inptr2), d0)];
|
||||
r = range_limit[DITHER_565_R(*inptr0, d0)];
|
||||
g = range_limit[DITHER_565_G(*inptr1, d0)];
|
||||
b = range_limit[DITHER_565_B(*inptr2, d0)];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
*(INT16 *)outptr = (INT16)rgb;
|
||||
}
|
||||
|
||||
+3
-5
@@ -53,9 +53,9 @@ ycc_rgb_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
y = GETJSAMPLE(inptr0[col]);
|
||||
cb = GETJSAMPLE(inptr1[col]);
|
||||
cr = GETJSAMPLE(inptr2[col]);
|
||||
y = inptr0[col];
|
||||
cb = inptr1[col];
|
||||
cr = inptr2[col];
|
||||
/* Range-limiting is essential due to noise introduced by DCT losses. */
|
||||
outptr[RGB_RED] = range_limit[y + Crrtab[cr]];
|
||||
outptr[RGB_GREEN] = range_limit[y +
|
||||
@@ -93,7 +93,6 @@ gray_rgb_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
inptr = input_buf[0][input_row++];
|
||||
outptr = *output_buf++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
/* We can dispense with GETJSAMPLE() here */
|
||||
outptr[RGB_RED] = outptr[RGB_GREEN] = outptr[RGB_BLUE] = inptr[col];
|
||||
/* Set unused byte to 0xFF so it can be interpreted as an opaque */
|
||||
/* alpha channel value */
|
||||
@@ -128,7 +127,6 @@ rgb_rgb_convert_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
/* We can dispense with GETJSAMPLE() here */
|
||||
outptr[RGB_RED] = inptr0[col];
|
||||
outptr[RGB_GREEN] = inptr1[col];
|
||||
outptr[RGB_BLUE] = inptr2[col];
|
||||
|
||||
Vendored
+7
-7
@@ -341,9 +341,9 @@ rgb_gray_convert(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
r = GETJSAMPLE(inptr0[col]);
|
||||
g = GETJSAMPLE(inptr1[col]);
|
||||
b = GETJSAMPLE(inptr2[col]);
|
||||
r = inptr0[col];
|
||||
g = inptr1[col];
|
||||
b = inptr2[col];
|
||||
/* Y */
|
||||
outptr[col] = (JSAMPLE)((ctab[r + R_Y_OFF] + ctab[g + G_Y_OFF] +
|
||||
ctab[b + B_Y_OFF]) >> SCALEBITS);
|
||||
@@ -550,9 +550,9 @@ ycck_cmyk_convert(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
for (col = 0; col < num_cols; col++) {
|
||||
y = GETJSAMPLE(inptr0[col]);
|
||||
cb = GETJSAMPLE(inptr1[col]);
|
||||
cr = GETJSAMPLE(inptr2[col]);
|
||||
y = inptr0[col];
|
||||
cb = inptr1[col];
|
||||
cr = inptr2[col];
|
||||
/* Range-limiting is essential due to noise introduced by DCT losses. */
|
||||
outptr[0] = range_limit[MAXJSAMPLE - (y + Crrtab[cr])]; /* red */
|
||||
outptr[1] = range_limit[MAXJSAMPLE - (y + /* green */
|
||||
@@ -560,7 +560,7 @@ ycck_cmyk_convert(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
SCALEBITS)))];
|
||||
outptr[2] = range_limit[MAXJSAMPLE - (y + Cbbtab[cb])]; /* blue */
|
||||
/* K passes through unchanged */
|
||||
outptr[3] = inptr3[col]; /* don't need GETJSAMPLE here */
|
||||
outptr[3] = inptr3[col];
|
||||
outptr += 4;
|
||||
}
|
||||
}
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
* Modified 2002-2010 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2010, 2015, D. R. Commander.
|
||||
* Copyright (C) 2010, 2015, 2022, D. R. Commander.
|
||||
* Copyright (C) 2013, MIPS Technologies, Inc., California.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
@@ -345,7 +345,7 @@ jinit_inverse_dct(j_decompress_ptr cinfo)
|
||||
compptr->dct_table =
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
sizeof(multiplier_table));
|
||||
MEMZERO(compptr->dct_table, sizeof(multiplier_table));
|
||||
memset(compptr->dct_table, 0, sizeof(multiplier_table));
|
||||
/* Mark multiplier table not yet set up for any method */
|
||||
idct->cur_method[ci] = -1;
|
||||
}
|
||||
|
||||
Vendored
+36
-33
@@ -5,6 +5,7 @@
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2009-2011, 2016, 2018-2019, D. R. Commander.
|
||||
* Copyright (C) 2018, Matthias Räncker.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -39,24 +40,6 @@ typedef struct {
|
||||
int last_dc_val[MAX_COMPS_IN_SCAN]; /* last DC coef for each component */
|
||||
} savable_state;
|
||||
|
||||
/* This macro is to work around compilers with missing or broken
|
||||
* structure assignment. You'll need to fix this code if you have
|
||||
* such a compiler and you change MAX_COMPS_IN_SCAN.
|
||||
*/
|
||||
|
||||
#ifndef NO_STRUCT_ASSIGN
|
||||
#define ASSIGN_STATE(dest, src) ((dest) = (src))
|
||||
#else
|
||||
#if MAX_COMPS_IN_SCAN == 4
|
||||
#define ASSIGN_STATE(dest, src) \
|
||||
((dest).last_dc_val[0] = (src).last_dc_val[0], \
|
||||
(dest).last_dc_val[1] = (src).last_dc_val[1], \
|
||||
(dest).last_dc_val[2] = (src).last_dc_val[2], \
|
||||
(dest).last_dc_val[3] = (src).last_dc_val[3])
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
typedef struct {
|
||||
struct jpeg_entropy_decoder pub; /* public fields */
|
||||
|
||||
@@ -325,7 +308,7 @@ jpeg_fill_bit_buffer(bitread_working_state *state,
|
||||
bytes_in_buffer = cinfo->src->bytes_in_buffer;
|
||||
}
|
||||
bytes_in_buffer--;
|
||||
c = GETJOCTET(*next_input_byte++);
|
||||
c = *next_input_byte++;
|
||||
|
||||
/* If it's 0xFF, check and discard stuffed zero byte */
|
||||
if (c == 0xFF) {
|
||||
@@ -342,7 +325,7 @@ jpeg_fill_bit_buffer(bitread_working_state *state,
|
||||
bytes_in_buffer = cinfo->src->bytes_in_buffer;
|
||||
}
|
||||
bytes_in_buffer--;
|
||||
c = GETJOCTET(*next_input_byte++);
|
||||
c = *next_input_byte++;
|
||||
} while (c == 0xFF);
|
||||
|
||||
if (c == 0) {
|
||||
@@ -405,8 +388,8 @@ no_more_bytes:
|
||||
|
||||
#define GET_BYTE { \
|
||||
register int c0, c1; \
|
||||
c0 = GETJOCTET(*buffer++); \
|
||||
c1 = GETJOCTET(*buffer); \
|
||||
c0 = *buffer++; \
|
||||
c1 = *buffer; \
|
||||
/* Pre-execute most common case */ \
|
||||
get_buffer = (get_buffer << 8) | c0; \
|
||||
bits_left += 8; \
|
||||
@@ -423,7 +406,7 @@ no_more_bytes:
|
||||
} \
|
||||
}
|
||||
|
||||
#if SIZEOF_SIZE_T == 8 || defined(_WIN64)
|
||||
#if SIZEOF_SIZE_T == 8 || defined(_WIN64) || (defined(__x86_64__) && defined(__ILP32__))
|
||||
|
||||
/* Pre-fetch 48 bytes, because the holding register is 64-bit */
|
||||
#define FILL_BIT_BUFFER_FAST \
|
||||
@@ -557,6 +540,12 @@ process_restart(j_decompress_ptr cinfo)
|
||||
}
|
||||
|
||||
|
||||
#if defined(__has_feature)
|
||||
#if __has_feature(undefined_behavior_sanitizer)
|
||||
__attribute__((no_sanitize("signed-integer-overflow"),
|
||||
no_sanitize("unsigned-integer-overflow")))
|
||||
#endif
|
||||
#endif
|
||||
LOCAL(boolean)
|
||||
decode_mcu_slow(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
{
|
||||
@@ -568,7 +557,7 @@ decode_mcu_slow(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
|
||||
/* Load up working state */
|
||||
BITREAD_LOAD_STATE(cinfo, entropy->bitstate);
|
||||
ASSIGN_STATE(state, entropy->saved);
|
||||
state = entropy->saved;
|
||||
|
||||
for (blkn = 0; blkn < cinfo->blocks_in_MCU; blkn++) {
|
||||
JBLOCKROW block = MCU_data ? MCU_data[blkn] : NULL;
|
||||
@@ -589,11 +578,15 @@ decode_mcu_slow(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
if (entropy->dc_needed[blkn]) {
|
||||
/* Convert DC difference to actual value, update last_dc_val */
|
||||
int ci = cinfo->MCU_membership[blkn];
|
||||
/* This is really just
|
||||
* s += state.last_dc_val[ci];
|
||||
* It is written this way in order to shut up UBSan.
|
||||
/* Certain malformed JPEG images produce repeated DC coefficient
|
||||
* differences of 2047 or -2047, which causes state.last_dc_val[ci] to
|
||||
* grow until it overflows or underflows a 32-bit signed integer. This
|
||||
* behavior is, to the best of our understanding, innocuous, and it is
|
||||
* unclear how to work around it without potentially affecting
|
||||
* performance. Thus, we (hopefully temporarily) suppress UBSan integer
|
||||
* overflow errors for this function and decode_mcu_fast().
|
||||
*/
|
||||
s = (int)((unsigned int)s + (unsigned int)state.last_dc_val[ci]);
|
||||
s += state.last_dc_val[ci];
|
||||
state.last_dc_val[ci] = s;
|
||||
if (block) {
|
||||
/* Output the DC coefficient (assumes jpeg_natural_order[0] = 0) */
|
||||
@@ -653,11 +646,17 @@ decode_mcu_slow(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
|
||||
/* Completed MCU, so update state */
|
||||
BITREAD_SAVE_STATE(cinfo, entropy->bitstate);
|
||||
ASSIGN_STATE(entropy->saved, state);
|
||||
entropy->saved = state;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
|
||||
#if defined(__has_feature)
|
||||
#if __has_feature(undefined_behavior_sanitizer)
|
||||
__attribute__((no_sanitize("signed-integer-overflow"),
|
||||
no_sanitize("unsigned-integer-overflow")))
|
||||
#endif
|
||||
#endif
|
||||
LOCAL(boolean)
|
||||
decode_mcu_fast(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
{
|
||||
@@ -671,7 +670,7 @@ decode_mcu_fast(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
/* Load up working state */
|
||||
BITREAD_LOAD_STATE(cinfo, entropy->bitstate);
|
||||
buffer = (JOCTET *)br_state.next_input_byte;
|
||||
ASSIGN_STATE(state, entropy->saved);
|
||||
state = entropy->saved;
|
||||
|
||||
for (blkn = 0; blkn < cinfo->blocks_in_MCU; blkn++) {
|
||||
JBLOCKROW block = MCU_data ? MCU_data[blkn] : NULL;
|
||||
@@ -688,7 +687,10 @@ decode_mcu_fast(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
|
||||
if (entropy->dc_needed[blkn]) {
|
||||
int ci = cinfo->MCU_membership[blkn];
|
||||
s = (int)((unsigned int)s + (unsigned int)state.last_dc_val[ci]);
|
||||
/* Refer to the comment in decode_mcu_slow() regarding the supression of
|
||||
* a UBSan integer overflow error in this line of code.
|
||||
*/
|
||||
s += state.last_dc_val[ci];
|
||||
state.last_dc_val[ci] = s;
|
||||
if (block)
|
||||
(*block)[0] = (JCOEF)s;
|
||||
@@ -740,7 +742,7 @@ decode_mcu_fast(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
br_state.bytes_in_buffer -= (buffer - br_state.next_input_byte);
|
||||
br_state.next_input_byte = buffer;
|
||||
BITREAD_SAVE_STATE(cinfo, entropy->bitstate);
|
||||
ASSIGN_STATE(entropy->saved, state);
|
||||
entropy->saved = state;
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
@@ -795,7 +797,8 @@ use_slow:
|
||||
}
|
||||
|
||||
/* Account for restart interval (no-op if not using restarts) */
|
||||
entropy->restarts_to_go--;
|
||||
if (cinfo->restart_interval)
|
||||
entropy->restarts_to_go--;
|
||||
|
||||
return TRUE;
|
||||
}
|
||||
|
||||
Vendored
+11
-2
@@ -4,7 +4,8 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2010-2011, 2015-2016, D. R. Commander.
|
||||
* Copyright (C) 2010-2011, 2015-2016, 2021, D. R. Commander.
|
||||
* Copyright (C) 2018, Matthias Räncker.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -78,6 +79,11 @@ EXTERN(void) jpeg_make_d_derived_tbl(j_decompress_ptr cinfo, boolean isDC,
|
||||
typedef size_t bit_buf_type; /* type of bit-extraction buffer */
|
||||
#define BIT_BUF_SIZE 64 /* size of buffer in bits */
|
||||
|
||||
#elif defined(__x86_64__) && defined(__ILP32__)
|
||||
|
||||
typedef unsigned long long bit_buf_type; /* type of bit-extraction buffer */
|
||||
#define BIT_BUF_SIZE 64 /* size of buffer in bits */
|
||||
|
||||
#else
|
||||
|
||||
typedef unsigned long bit_buf_type; /* type of bit-extraction buffer */
|
||||
@@ -228,7 +234,10 @@ slowlabel: \
|
||||
s |= GET_BITS(1); \
|
||||
nb++; \
|
||||
} \
|
||||
s = htbl->pub->huffval[(int)(s + htbl->valoffset[nb]) & 0xFF]; \
|
||||
if (nb > 16) \
|
||||
s = 0; \
|
||||
else \
|
||||
s = htbl->pub->huffval[(int)(s + htbl->valoffset[nb]) & 0xFF]; \
|
||||
}
|
||||
|
||||
/* Out-of-line case for Huffman code fetching */
|
||||
|
||||
Vendored
+16
-20
@@ -18,10 +18,6 @@
|
||||
#include "jpeglib.h"
|
||||
#include "jerror.h"
|
||||
|
||||
#ifndef HAVE_STDLIB_H /* <stdlib.h> should declare malloc() */
|
||||
extern void *malloc(size_t size);
|
||||
#endif
|
||||
|
||||
|
||||
#define ICC_MARKER (JPEG_APP0 + 2) /* JPEG marker code for ICC */
|
||||
#define ICC_OVERHEAD_LEN 14 /* size of non-profile data in APP2 */
|
||||
@@ -38,18 +34,18 @@ marker_is_icc(jpeg_saved_marker_ptr marker)
|
||||
marker->marker == ICC_MARKER &&
|
||||
marker->data_length >= ICC_OVERHEAD_LEN &&
|
||||
/* verify the identifying string */
|
||||
GETJOCTET(marker->data[0]) == 0x49 &&
|
||||
GETJOCTET(marker->data[1]) == 0x43 &&
|
||||
GETJOCTET(marker->data[2]) == 0x43 &&
|
||||
GETJOCTET(marker->data[3]) == 0x5F &&
|
||||
GETJOCTET(marker->data[4]) == 0x50 &&
|
||||
GETJOCTET(marker->data[5]) == 0x52 &&
|
||||
GETJOCTET(marker->data[6]) == 0x4F &&
|
||||
GETJOCTET(marker->data[7]) == 0x46 &&
|
||||
GETJOCTET(marker->data[8]) == 0x49 &&
|
||||
GETJOCTET(marker->data[9]) == 0x4C &&
|
||||
GETJOCTET(marker->data[10]) == 0x45 &&
|
||||
GETJOCTET(marker->data[11]) == 0x0;
|
||||
marker->data[0] == 0x49 &&
|
||||
marker->data[1] == 0x43 &&
|
||||
marker->data[2] == 0x43 &&
|
||||
marker->data[3] == 0x5F &&
|
||||
marker->data[4] == 0x50 &&
|
||||
marker->data[5] == 0x52 &&
|
||||
marker->data[6] == 0x4F &&
|
||||
marker->data[7] == 0x46 &&
|
||||
marker->data[8] == 0x49 &&
|
||||
marker->data[9] == 0x4C &&
|
||||
marker->data[10] == 0x45 &&
|
||||
marker->data[11] == 0x0;
|
||||
}
|
||||
|
||||
|
||||
@@ -102,12 +98,12 @@ jpeg_read_icc_profile(j_decompress_ptr cinfo, JOCTET **icc_data_ptr,
|
||||
for (marker = cinfo->marker_list; marker != NULL; marker = marker->next) {
|
||||
if (marker_is_icc(marker)) {
|
||||
if (num_markers == 0)
|
||||
num_markers = GETJOCTET(marker->data[13]);
|
||||
else if (num_markers != GETJOCTET(marker->data[13])) {
|
||||
num_markers = marker->data[13];
|
||||
else if (num_markers != marker->data[13]) {
|
||||
WARNMS(cinfo, JWRN_BOGUS_ICC); /* inconsistent num_markers fields */
|
||||
return FALSE;
|
||||
}
|
||||
seq_no = GETJOCTET(marker->data[12]);
|
||||
seq_no = marker->data[12];
|
||||
if (seq_no <= 0 || seq_no > num_markers) {
|
||||
WARNMS(cinfo, JWRN_BOGUS_ICC); /* bogus sequence number */
|
||||
return FALSE;
|
||||
@@ -154,7 +150,7 @@ jpeg_read_icc_profile(j_decompress_ptr cinfo, JOCTET **icc_data_ptr,
|
||||
JOCTET FAR *src_ptr;
|
||||
JOCTET *dst_ptr;
|
||||
unsigned int length;
|
||||
seq_no = GETJOCTET(marker->data[12]);
|
||||
seq_no = marker->data[12];
|
||||
dst_ptr = icc_data + data_offset[seq_no];
|
||||
src_ptr = marker->data + ICC_OVERHEAD_LEN;
|
||||
length = data_length[seq_no];
|
||||
|
||||
Vendored
+2
-2
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2010, 2016, 2018, D. R. Commander.
|
||||
* Copyright (C) 2010, 2016, 2018, 2022, D. R. Commander.
|
||||
* Copyright (C) 2015, Google, Inc.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
@@ -264,7 +264,7 @@ latch_quant_tables(j_decompress_ptr cinfo)
|
||||
qtbl = (JQUANT_TBL *)
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
sizeof(JQUANT_TBL));
|
||||
MEMCOPY(qtbl, cinfo->quant_tbl_ptrs[qtblno], sizeof(JQUANT_TBL));
|
||||
memcpy(qtbl, cinfo->quant_tbl_ptrs[qtblno], sizeof(JQUANT_TBL));
|
||||
compptr->quant_table = qtbl;
|
||||
}
|
||||
}
|
||||
|
||||
+3
-2
@@ -18,6 +18,7 @@
|
||||
|
||||
#include "jinclude.h"
|
||||
#include "jdmainct.h"
|
||||
#include "jconfigint.h"
|
||||
|
||||
|
||||
/*
|
||||
@@ -360,7 +361,7 @@ process_data_context_main(j_decompress_ptr cinfo, JSAMPARRAY output_buf,
|
||||
main_ptr->context_state = CTX_PREPARE_FOR_IMCU;
|
||||
if (*out_row_ctr >= out_rows_avail)
|
||||
return; /* Postprocessor exactly filled output buf */
|
||||
/*FALLTHROUGH*/
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case CTX_PREPARE_FOR_IMCU:
|
||||
/* Prepare to process first M-1 row groups of this iMCU row */
|
||||
main_ptr->rowgroup_ctr = 0;
|
||||
@@ -371,7 +372,7 @@ process_data_context_main(j_decompress_ptr cinfo, JSAMPARRAY output_buf,
|
||||
if (main_ptr->iMCU_row_ctr == cinfo->total_iMCU_rows)
|
||||
set_bottom_pointers(cinfo);
|
||||
main_ptr->context_state = CTX_PROCESS_IMCU;
|
||||
/*FALLTHROUGH*/
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case CTX_PROCESS_IMCU:
|
||||
/* Call postprocessor using previously set pointers */
|
||||
(*cinfo->post->post_process_data) (cinfo,
|
||||
|
||||
+36
-39
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1998, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2012, 2015, D. R. Commander.
|
||||
* Copyright (C) 2012, 2015, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -151,7 +151,7 @@ typedef my_marker_reader *my_marker_ptr;
|
||||
#define INPUT_BYTE(cinfo, V, action) \
|
||||
MAKESTMT( MAKE_BYTE_AVAIL(cinfo, action); \
|
||||
bytes_in_buffer--; \
|
||||
V = GETJOCTET(*next_input_byte++); )
|
||||
V = *next_input_byte++; )
|
||||
|
||||
/* As above, but read two bytes interpreted as an unsigned 16-bit integer.
|
||||
* V should be declared unsigned int or perhaps JLONG.
|
||||
@@ -159,10 +159,10 @@ typedef my_marker_reader *my_marker_ptr;
|
||||
#define INPUT_2BYTES(cinfo, V, action) \
|
||||
MAKESTMT( MAKE_BYTE_AVAIL(cinfo, action); \
|
||||
bytes_in_buffer--; \
|
||||
V = ((unsigned int)GETJOCTET(*next_input_byte++)) << 8; \
|
||||
V = ((unsigned int)(*next_input_byte++)) << 8; \
|
||||
MAKE_BYTE_AVAIL(cinfo, action); \
|
||||
bytes_in_buffer--; \
|
||||
V += GETJOCTET(*next_input_byte++); )
|
||||
V += *next_input_byte++; )
|
||||
|
||||
|
||||
/*
|
||||
@@ -473,7 +473,7 @@ get_dht(j_decompress_ptr cinfo)
|
||||
for (i = 0; i < count; i++)
|
||||
INPUT_BYTE(cinfo, huffval[i], return FALSE);
|
||||
|
||||
MEMZERO(&huffval[count], (256 - count) * sizeof(UINT8));
|
||||
memset(&huffval[count], 0, (256 - count) * sizeof(UINT8));
|
||||
|
||||
length -= count;
|
||||
|
||||
@@ -491,8 +491,8 @@ get_dht(j_decompress_ptr cinfo)
|
||||
if (*htblptr == NULL)
|
||||
*htblptr = jpeg_alloc_huff_table((j_common_ptr)cinfo);
|
||||
|
||||
MEMCOPY((*htblptr)->bits, bits, sizeof((*htblptr)->bits));
|
||||
MEMCOPY((*htblptr)->huffval, huffval, sizeof((*htblptr)->huffval));
|
||||
memcpy((*htblptr)->bits, bits, sizeof((*htblptr)->bits));
|
||||
memcpy((*htblptr)->huffval, huffval, sizeof((*htblptr)->huffval));
|
||||
}
|
||||
|
||||
if (length != 0)
|
||||
@@ -608,18 +608,18 @@ examine_app0(j_decompress_ptr cinfo, JOCTET *data, unsigned int datalen,
|
||||
JLONG totallen = (JLONG)datalen + remaining;
|
||||
|
||||
if (datalen >= APP0_DATA_LEN &&
|
||||
GETJOCTET(data[0]) == 0x4A &&
|
||||
GETJOCTET(data[1]) == 0x46 &&
|
||||
GETJOCTET(data[2]) == 0x49 &&
|
||||
GETJOCTET(data[3]) == 0x46 &&
|
||||
GETJOCTET(data[4]) == 0) {
|
||||
data[0] == 0x4A &&
|
||||
data[1] == 0x46 &&
|
||||
data[2] == 0x49 &&
|
||||
data[3] == 0x46 &&
|
||||
data[4] == 0) {
|
||||
/* Found JFIF APP0 marker: save info */
|
||||
cinfo->saw_JFIF_marker = TRUE;
|
||||
cinfo->JFIF_major_version = GETJOCTET(data[5]);
|
||||
cinfo->JFIF_minor_version = GETJOCTET(data[6]);
|
||||
cinfo->density_unit = GETJOCTET(data[7]);
|
||||
cinfo->X_density = (GETJOCTET(data[8]) << 8) + GETJOCTET(data[9]);
|
||||
cinfo->Y_density = (GETJOCTET(data[10]) << 8) + GETJOCTET(data[11]);
|
||||
cinfo->JFIF_major_version = data[5];
|
||||
cinfo->JFIF_minor_version = data[6];
|
||||
cinfo->density_unit = data[7];
|
||||
cinfo->X_density = (data[8] << 8) + data[9];
|
||||
cinfo->Y_density = (data[10] << 8) + data[11];
|
||||
/* Check version.
|
||||
* Major version must be 1, anything else signals an incompatible change.
|
||||
* (We used to treat this as an error, but now it's a nonfatal warning,
|
||||
@@ -634,24 +634,22 @@ examine_app0(j_decompress_ptr cinfo, JOCTET *data, unsigned int datalen,
|
||||
cinfo->JFIF_major_version, cinfo->JFIF_minor_version,
|
||||
cinfo->X_density, cinfo->Y_density, cinfo->density_unit);
|
||||
/* Validate thumbnail dimensions and issue appropriate messages */
|
||||
if (GETJOCTET(data[12]) | GETJOCTET(data[13]))
|
||||
TRACEMS2(cinfo, 1, JTRC_JFIF_THUMBNAIL,
|
||||
GETJOCTET(data[12]), GETJOCTET(data[13]));
|
||||
if (data[12] | data[13])
|
||||
TRACEMS2(cinfo, 1, JTRC_JFIF_THUMBNAIL, data[12], data[13]);
|
||||
totallen -= APP0_DATA_LEN;
|
||||
if (totallen !=
|
||||
((JLONG)GETJOCTET(data[12]) * (JLONG)GETJOCTET(data[13]) * (JLONG)3))
|
||||
if (totallen != ((JLONG)data[12] * (JLONG)data[13] * (JLONG)3))
|
||||
TRACEMS1(cinfo, 1, JTRC_JFIF_BADTHUMBNAILSIZE, (int)totallen);
|
||||
} else if (datalen >= 6 &&
|
||||
GETJOCTET(data[0]) == 0x4A &&
|
||||
GETJOCTET(data[1]) == 0x46 &&
|
||||
GETJOCTET(data[2]) == 0x58 &&
|
||||
GETJOCTET(data[3]) == 0x58 &&
|
||||
GETJOCTET(data[4]) == 0) {
|
||||
data[0] == 0x4A &&
|
||||
data[1] == 0x46 &&
|
||||
data[2] == 0x58 &&
|
||||
data[3] == 0x58 &&
|
||||
data[4] == 0) {
|
||||
/* Found JFIF "JFXX" extension APP0 marker */
|
||||
/* The library doesn't actually do anything with these,
|
||||
* but we try to produce a helpful trace message.
|
||||
*/
|
||||
switch (GETJOCTET(data[5])) {
|
||||
switch (data[5]) {
|
||||
case 0x10:
|
||||
TRACEMS1(cinfo, 1, JTRC_THUMB_JPEG, (int)totallen);
|
||||
break;
|
||||
@@ -662,8 +660,7 @@ examine_app0(j_decompress_ptr cinfo, JOCTET *data, unsigned int datalen,
|
||||
TRACEMS1(cinfo, 1, JTRC_THUMB_RGB, (int)totallen);
|
||||
break;
|
||||
default:
|
||||
TRACEMS2(cinfo, 1, JTRC_JFIF_EXTENSION,
|
||||
GETJOCTET(data[5]), (int)totallen);
|
||||
TRACEMS2(cinfo, 1, JTRC_JFIF_EXTENSION, data[5], (int)totallen);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
@@ -684,16 +681,16 @@ examine_app14(j_decompress_ptr cinfo, JOCTET *data, unsigned int datalen,
|
||||
unsigned int version, flags0, flags1, transform;
|
||||
|
||||
if (datalen >= APP14_DATA_LEN &&
|
||||
GETJOCTET(data[0]) == 0x41 &&
|
||||
GETJOCTET(data[1]) == 0x64 &&
|
||||
GETJOCTET(data[2]) == 0x6F &&
|
||||
GETJOCTET(data[3]) == 0x62 &&
|
||||
GETJOCTET(data[4]) == 0x65) {
|
||||
data[0] == 0x41 &&
|
||||
data[1] == 0x64 &&
|
||||
data[2] == 0x6F &&
|
||||
data[3] == 0x62 &&
|
||||
data[4] == 0x65) {
|
||||
/* Found Adobe APP14 marker */
|
||||
version = (GETJOCTET(data[5]) << 8) + GETJOCTET(data[6]);
|
||||
flags0 = (GETJOCTET(data[7]) << 8) + GETJOCTET(data[8]);
|
||||
flags1 = (GETJOCTET(data[9]) << 8) + GETJOCTET(data[10]);
|
||||
transform = GETJOCTET(data[11]);
|
||||
version = (data[5] << 8) + data[6];
|
||||
flags0 = (data[7] << 8) + data[8];
|
||||
flags1 = (data[9] << 8) + data[10];
|
||||
transform = data[11];
|
||||
TRACEMS4(cinfo, 1, JTRC_ADOBE, version, flags0, flags1, transform);
|
||||
cinfo->saw_Adobe_marker = TRUE;
|
||||
cinfo->Adobe_transform = (UINT8)transform;
|
||||
|
||||
+7
-18
@@ -5,7 +5,7 @@
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* Modified 2002-2009 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2009-2011, 2016, D. R. Commander.
|
||||
* Copyright (C) 2009-2011, 2016, 2019, 2022, D. R. Commander.
|
||||
* Copyright (C) 2013, Linaro Limited.
|
||||
* Copyright (C) 2015, Google, Inc.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
@@ -22,7 +22,6 @@
|
||||
#include "jpeglib.h"
|
||||
#include "jpegcomp.h"
|
||||
#include "jdmaster.h"
|
||||
#include "jsimd.h"
|
||||
|
||||
|
||||
/*
|
||||
@@ -70,17 +69,6 @@ use_merged_upsample(j_decompress_ptr cinfo)
|
||||
cinfo->comp_info[1]._DCT_scaled_size != cinfo->_min_DCT_scaled_size ||
|
||||
cinfo->comp_info[2]._DCT_scaled_size != cinfo->_min_DCT_scaled_size)
|
||||
return FALSE;
|
||||
#ifdef WITH_SIMD
|
||||
/* If YCbCr-to-RGB color conversion is SIMD-accelerated but merged upsampling
|
||||
isn't, then disabling merged upsampling is likely to be faster when
|
||||
decompressing YCbCr JPEG images. */
|
||||
if (!jsimd_can_h2v2_merged_upsample() && !jsimd_can_h2v1_merged_upsample() &&
|
||||
jsimd_can_ycc_rgb() && cinfo->jpeg_color_space == JCS_YCbCr &&
|
||||
(cinfo->out_color_space == JCS_RGB ||
|
||||
(cinfo->out_color_space >= JCS_EXT_RGB &&
|
||||
cinfo->out_color_space <= JCS_EXT_ARGB)))
|
||||
return FALSE;
|
||||
#endif
|
||||
/* ??? also need to test for upsample-time rescaling, when & if supported */
|
||||
return TRUE; /* by golly, it'll work... */
|
||||
#else
|
||||
@@ -429,7 +417,7 @@ prepare_range_limit_table(j_decompress_ptr cinfo)
|
||||
table += (MAXJSAMPLE + 1); /* allow negative subscripts of simple table */
|
||||
cinfo->sample_range_limit = table;
|
||||
/* First segment of "simple" table: limit[x] = 0 for x < 0 */
|
||||
MEMZERO(table - (MAXJSAMPLE + 1), (MAXJSAMPLE + 1) * sizeof(JSAMPLE));
|
||||
memset(table - (MAXJSAMPLE + 1), 0, (MAXJSAMPLE + 1) * sizeof(JSAMPLE));
|
||||
/* Main part of "simple" table: limit[x] = x */
|
||||
for (i = 0; i <= MAXJSAMPLE; i++)
|
||||
table[i] = (JSAMPLE)i;
|
||||
@@ -438,10 +426,10 @@ prepare_range_limit_table(j_decompress_ptr cinfo)
|
||||
for (i = CENTERJSAMPLE; i < 2 * (MAXJSAMPLE + 1); i++)
|
||||
table[i] = MAXJSAMPLE;
|
||||
/* Second half of post-IDCT table */
|
||||
MEMZERO(table + (2 * (MAXJSAMPLE + 1)),
|
||||
(2 * (MAXJSAMPLE + 1) - CENTERJSAMPLE) * sizeof(JSAMPLE));
|
||||
MEMCOPY(table + (4 * (MAXJSAMPLE + 1) - CENTERJSAMPLE),
|
||||
cinfo->sample_range_limit, CENTERJSAMPLE * sizeof(JSAMPLE));
|
||||
memset(table + (2 * (MAXJSAMPLE + 1)), 0,
|
||||
(2 * (MAXJSAMPLE + 1) - CENTERJSAMPLE) * sizeof(JSAMPLE));
|
||||
memcpy(table + (4 * (MAXJSAMPLE + 1) - CENTERJSAMPLE),
|
||||
cinfo->sample_range_limit, CENTERJSAMPLE * sizeof(JSAMPLE));
|
||||
}
|
||||
|
||||
|
||||
@@ -580,6 +568,7 @@ master_selection(j_decompress_ptr cinfo)
|
||||
*/
|
||||
cinfo->master->first_iMCU_col = 0;
|
||||
cinfo->master->last_iMCU_col = cinfo->MCUs_per_row - 1;
|
||||
cinfo->master->last_good_iMCU_row = 0;
|
||||
|
||||
#ifdef D_MULTISCAN_FILES_SUPPORTED
|
||||
/* If jpeg_start_decompress will read the whole file, initialize
|
||||
|
||||
+34
-34
@@ -43,20 +43,20 @@ h2v1_merged_upsample_565_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
/* Loop for each pair of output pixels */
|
||||
for (col = cinfo->output_width >> 1; col > 0; col--) {
|
||||
/* Do the chroma part of the calculation */
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
|
||||
/* Fetch 2 Y values and emit 2 pixels */
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
y = *inptr0++;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
y = *inptr0++;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
@@ -68,12 +68,12 @@ h2v1_merged_upsample_565_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
|
||||
/* If image width is odd, do the last output column separately */
|
||||
if (cinfo->output_width & 1) {
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
y = GETJSAMPLE(*inptr0);
|
||||
y = *inptr0;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
@@ -115,21 +115,21 @@ h2v1_merged_upsample_565D_internal(j_decompress_ptr cinfo,
|
||||
/* Loop for each pair of output pixels */
|
||||
for (col = cinfo->output_width >> 1; col > 0; col--) {
|
||||
/* Do the chroma part of the calculation */
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
|
||||
/* Fetch 2 Y values and emit 2 pixels */
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
y = *inptr0++;
|
||||
r = range_limit[DITHER_565_R(y + cred, d0)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d0)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d0)];
|
||||
d0 = DITHER_ROTATE(d0);
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
y = *inptr0++;
|
||||
r = range_limit[DITHER_565_R(y + cred, d0)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d0)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d0)];
|
||||
@@ -142,12 +142,12 @@ h2v1_merged_upsample_565D_internal(j_decompress_ptr cinfo,
|
||||
|
||||
/* If image width is odd, do the last output column separately */
|
||||
if (cinfo->output_width & 1) {
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
y = GETJSAMPLE(*inptr0);
|
||||
y = *inptr0;
|
||||
r = range_limit[DITHER_565_R(y + cred, d0)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d0)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d0)];
|
||||
@@ -189,20 +189,20 @@ h2v2_merged_upsample_565_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
/* Loop for each group of output pixels */
|
||||
for (col = cinfo->output_width >> 1; col > 0; col--) {
|
||||
/* Do the chroma part of the calculation */
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
|
||||
/* Fetch 4 Y values and emit 4 pixels */
|
||||
y = GETJSAMPLE(*inptr00++);
|
||||
y = *inptr00++;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr00++);
|
||||
y = *inptr00++;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
@@ -211,13 +211,13 @@ h2v2_merged_upsample_565_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
WRITE_TWO_PIXELS(outptr0, rgb);
|
||||
outptr0 += 4;
|
||||
|
||||
y = GETJSAMPLE(*inptr01++);
|
||||
y = *inptr01++;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr01++);
|
||||
y = *inptr01++;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
@@ -229,20 +229,20 @@ h2v2_merged_upsample_565_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
|
||||
/* If image width is odd, do the last output column separately */
|
||||
if (cinfo->output_width & 1) {
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
|
||||
y = GETJSAMPLE(*inptr00);
|
||||
y = *inptr00;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
*(INT16 *)outptr0 = (INT16)rgb;
|
||||
|
||||
y = GETJSAMPLE(*inptr01);
|
||||
y = *inptr01;
|
||||
r = range_limit[y + cred];
|
||||
g = range_limit[y + cgreen];
|
||||
b = range_limit[y + cblue];
|
||||
@@ -287,21 +287,21 @@ h2v2_merged_upsample_565D_internal(j_decompress_ptr cinfo,
|
||||
/* Loop for each group of output pixels */
|
||||
for (col = cinfo->output_width >> 1; col > 0; col--) {
|
||||
/* Do the chroma part of the calculation */
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
|
||||
/* Fetch 4 Y values and emit 4 pixels */
|
||||
y = GETJSAMPLE(*inptr00++);
|
||||
y = *inptr00++;
|
||||
r = range_limit[DITHER_565_R(y + cred, d0)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d0)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d0)];
|
||||
d0 = DITHER_ROTATE(d0);
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr00++);
|
||||
y = *inptr00++;
|
||||
r = range_limit[DITHER_565_R(y + cred, d0)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d0)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d0)];
|
||||
@@ -311,14 +311,14 @@ h2v2_merged_upsample_565D_internal(j_decompress_ptr cinfo,
|
||||
WRITE_TWO_PIXELS(outptr0, rgb);
|
||||
outptr0 += 4;
|
||||
|
||||
y = GETJSAMPLE(*inptr01++);
|
||||
y = *inptr01++;
|
||||
r = range_limit[DITHER_565_R(y + cred, d1)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d1)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d1)];
|
||||
d1 = DITHER_ROTATE(d1);
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
|
||||
y = GETJSAMPLE(*inptr01++);
|
||||
y = *inptr01++;
|
||||
r = range_limit[DITHER_565_R(y + cred, d1)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d1)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d1)];
|
||||
@@ -331,20 +331,20 @@ h2v2_merged_upsample_565D_internal(j_decompress_ptr cinfo,
|
||||
|
||||
/* If image width is odd, do the last output column separately */
|
||||
if (cinfo->output_width & 1) {
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
|
||||
y = GETJSAMPLE(*inptr00);
|
||||
y = *inptr00;
|
||||
r = range_limit[DITHER_565_R(y + cred, d0)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d0)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d0)];
|
||||
rgb = PACK_SHORT_565(r, g, b);
|
||||
*(INT16 *)outptr0 = (INT16)rgb;
|
||||
|
||||
y = GETJSAMPLE(*inptr01);
|
||||
y = *inptr01;
|
||||
r = range_limit[DITHER_565_R(y + cred, d1)];
|
||||
g = range_limit[DITHER_565_G(y + cgreen, d1)];
|
||||
b = range_limit[DITHER_565_B(y + cblue, d1)];
|
||||
|
||||
+17
-17
@@ -46,13 +46,13 @@ h2v1_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
/* Loop for each pair of output pixels */
|
||||
for (col = cinfo->output_width >> 1; col > 0; col--) {
|
||||
/* Do the chroma part of the calculation */
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
/* Fetch 2 Y values and emit 2 pixels */
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
y = *inptr0++;
|
||||
outptr[RGB_RED] = range_limit[y + cred];
|
||||
outptr[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -60,7 +60,7 @@ h2v1_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr[RGB_ALPHA] = 0xFF;
|
||||
#endif
|
||||
outptr += RGB_PIXELSIZE;
|
||||
y = GETJSAMPLE(*inptr0++);
|
||||
y = *inptr0++;
|
||||
outptr[RGB_RED] = range_limit[y + cred];
|
||||
outptr[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -71,12 +71,12 @@ h2v1_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
}
|
||||
/* If image width is odd, do the last output column separately */
|
||||
if (cinfo->output_width & 1) {
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
y = GETJSAMPLE(*inptr0);
|
||||
y = *inptr0;
|
||||
outptr[RGB_RED] = range_limit[y + cred];
|
||||
outptr[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -120,13 +120,13 @@ h2v2_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
/* Loop for each group of output pixels */
|
||||
for (col = cinfo->output_width >> 1; col > 0; col--) {
|
||||
/* Do the chroma part of the calculation */
|
||||
cb = GETJSAMPLE(*inptr1++);
|
||||
cr = GETJSAMPLE(*inptr2++);
|
||||
cb = *inptr1++;
|
||||
cr = *inptr2++;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
/* Fetch 4 Y values and emit 4 pixels */
|
||||
y = GETJSAMPLE(*inptr00++);
|
||||
y = *inptr00++;
|
||||
outptr0[RGB_RED] = range_limit[y + cred];
|
||||
outptr0[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr0[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -134,7 +134,7 @@ h2v2_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr0[RGB_ALPHA] = 0xFF;
|
||||
#endif
|
||||
outptr0 += RGB_PIXELSIZE;
|
||||
y = GETJSAMPLE(*inptr00++);
|
||||
y = *inptr00++;
|
||||
outptr0[RGB_RED] = range_limit[y + cred];
|
||||
outptr0[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr0[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -142,7 +142,7 @@ h2v2_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr0[RGB_ALPHA] = 0xFF;
|
||||
#endif
|
||||
outptr0 += RGB_PIXELSIZE;
|
||||
y = GETJSAMPLE(*inptr01++);
|
||||
y = *inptr01++;
|
||||
outptr1[RGB_RED] = range_limit[y + cred];
|
||||
outptr1[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr1[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -150,7 +150,7 @@ h2v2_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
outptr1[RGB_ALPHA] = 0xFF;
|
||||
#endif
|
||||
outptr1 += RGB_PIXELSIZE;
|
||||
y = GETJSAMPLE(*inptr01++);
|
||||
y = *inptr01++;
|
||||
outptr1[RGB_RED] = range_limit[y + cred];
|
||||
outptr1[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr1[RGB_BLUE] = range_limit[y + cblue];
|
||||
@@ -161,19 +161,19 @@ h2v2_merged_upsample_internal(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
}
|
||||
/* If image width is odd, do the last output column separately */
|
||||
if (cinfo->output_width & 1) {
|
||||
cb = GETJSAMPLE(*inptr1);
|
||||
cr = GETJSAMPLE(*inptr2);
|
||||
cb = *inptr1;
|
||||
cr = *inptr2;
|
||||
cred = Crrtab[cr];
|
||||
cgreen = (int)RIGHT_SHIFT(Cbgtab[cb] + Crgtab[cr], SCALEBITS);
|
||||
cblue = Cbbtab[cb];
|
||||
y = GETJSAMPLE(*inptr00);
|
||||
y = *inptr00;
|
||||
outptr0[RGB_RED] = range_limit[y + cred];
|
||||
outptr0[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr0[RGB_BLUE] = range_limit[y + cblue];
|
||||
#ifdef RGB_ALPHA
|
||||
outptr0[RGB_ALPHA] = 0xFF;
|
||||
#endif
|
||||
y = GETJSAMPLE(*inptr01);
|
||||
y = *inptr01;
|
||||
outptr1[RGB_RED] = range_limit[y + cred];
|
||||
outptr1[RGB_GREEN] = range_limit[y + cgreen];
|
||||
outptr1[RGB_BLUE] = range_limit[y + cblue];
|
||||
|
||||
Vendored
+24
-32
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1995-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2015-2016, 2018, D. R. Commander.
|
||||
* Copyright (C) 2015-2016, 2018-2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -41,25 +41,6 @@ typedef struct {
|
||||
int last_dc_val[MAX_COMPS_IN_SCAN]; /* last DC coef for each component */
|
||||
} savable_state;
|
||||
|
||||
/* This macro is to work around compilers with missing or broken
|
||||
* structure assignment. You'll need to fix this code if you have
|
||||
* such a compiler and you change MAX_COMPS_IN_SCAN.
|
||||
*/
|
||||
|
||||
#ifndef NO_STRUCT_ASSIGN
|
||||
#define ASSIGN_STATE(dest, src) ((dest) = (src))
|
||||
#else
|
||||
#if MAX_COMPS_IN_SCAN == 4
|
||||
#define ASSIGN_STATE(dest, src) \
|
||||
((dest).EOBRUN = (src).EOBRUN, \
|
||||
(dest).last_dc_val[0] = (src).last_dc_val[0], \
|
||||
(dest).last_dc_val[1] = (src).last_dc_val[1], \
|
||||
(dest).last_dc_val[2] = (src).last_dc_val[2], \
|
||||
(dest).last_dc_val[3] = (src).last_dc_val[3])
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
typedef struct {
|
||||
struct jpeg_entropy_decoder pub; /* public fields */
|
||||
|
||||
@@ -102,7 +83,7 @@ start_pass_phuff_decoder(j_decompress_ptr cinfo)
|
||||
boolean is_DC_band, bad;
|
||||
int ci, coefi, tbl;
|
||||
d_derived_tbl **pdtbl;
|
||||
int *coef_bit_ptr;
|
||||
int *coef_bit_ptr, *prev_coef_bit_ptr;
|
||||
jpeg_component_info *compptr;
|
||||
|
||||
is_DC_band = (cinfo->Ss == 0);
|
||||
@@ -143,8 +124,15 @@ start_pass_phuff_decoder(j_decompress_ptr cinfo)
|
||||
for (ci = 0; ci < cinfo->comps_in_scan; ci++) {
|
||||
int cindex = cinfo->cur_comp_info[ci]->component_index;
|
||||
coef_bit_ptr = &cinfo->coef_bits[cindex][0];
|
||||
prev_coef_bit_ptr = &cinfo->coef_bits[cindex + cinfo->num_components][0];
|
||||
if (!is_DC_band && coef_bit_ptr[0] < 0) /* AC without prior DC scan */
|
||||
WARNMS2(cinfo, JWRN_BOGUS_PROGRESSION, cindex, 0);
|
||||
for (coefi = MIN(cinfo->Ss, 1); coefi <= MAX(cinfo->Se, 9); coefi++) {
|
||||
if (cinfo->input_scan_number > 1)
|
||||
prev_coef_bit_ptr[coefi] = coef_bit_ptr[coefi];
|
||||
else
|
||||
prev_coef_bit_ptr[coefi] = 0;
|
||||
}
|
||||
for (coefi = cinfo->Ss; coefi <= cinfo->Se; coefi++) {
|
||||
int expected = (coef_bit_ptr[coefi] < 0) ? 0 : coef_bit_ptr[coefi];
|
||||
if (cinfo->Ah != expected)
|
||||
@@ -323,7 +311,7 @@ decode_mcu_DC_first(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
|
||||
/* Load up working state */
|
||||
BITREAD_LOAD_STATE(cinfo, entropy->bitstate);
|
||||
ASSIGN_STATE(state, entropy->saved);
|
||||
state = entropy->saved;
|
||||
|
||||
/* Outer loop handles each block in the MCU */
|
||||
|
||||
@@ -356,11 +344,12 @@ decode_mcu_DC_first(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
|
||||
/* Completed MCU, so update state */
|
||||
BITREAD_SAVE_STATE(cinfo, entropy->bitstate);
|
||||
ASSIGN_STATE(entropy->saved, state);
|
||||
entropy->saved = state;
|
||||
}
|
||||
|
||||
/* Account for restart interval (no-op if not using restarts) */
|
||||
entropy->restarts_to_go--;
|
||||
if (cinfo->restart_interval)
|
||||
entropy->restarts_to_go--;
|
||||
|
||||
return TRUE;
|
||||
}
|
||||
@@ -444,7 +433,8 @@ decode_mcu_AC_first(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
}
|
||||
|
||||
/* Account for restart interval (no-op if not using restarts) */
|
||||
entropy->restarts_to_go--;
|
||||
if (cinfo->restart_interval)
|
||||
entropy->restarts_to_go--;
|
||||
|
||||
return TRUE;
|
||||
}
|
||||
@@ -495,7 +485,8 @@ decode_mcu_DC_refine(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
BITREAD_SAVE_STATE(cinfo, entropy->bitstate);
|
||||
|
||||
/* Account for restart interval (no-op if not using restarts) */
|
||||
entropy->restarts_to_go--;
|
||||
if (cinfo->restart_interval)
|
||||
entropy->restarts_to_go--;
|
||||
|
||||
return TRUE;
|
||||
}
|
||||
@@ -587,9 +578,9 @@ decode_mcu_AC_refine(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
if (GET_BITS(1)) {
|
||||
if ((*thiscoef & p1) == 0) { /* do nothing if already set it */
|
||||
if (*thiscoef >= 0)
|
||||
*thiscoef += p1;
|
||||
*thiscoef += (JCOEF)p1;
|
||||
else
|
||||
*thiscoef += m1;
|
||||
*thiscoef += (JCOEF)m1;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -621,9 +612,9 @@ decode_mcu_AC_refine(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
if (GET_BITS(1)) {
|
||||
if ((*thiscoef & p1) == 0) { /* do nothing if already changed it */
|
||||
if (*thiscoef >= 0)
|
||||
*thiscoef += p1;
|
||||
*thiscoef += (JCOEF)p1;
|
||||
else
|
||||
*thiscoef += m1;
|
||||
*thiscoef += (JCOEF)m1;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -638,7 +629,8 @@ decode_mcu_AC_refine(j_decompress_ptr cinfo, JBLOCKROW *MCU_data)
|
||||
}
|
||||
|
||||
/* Account for restart interval (no-op if not using restarts) */
|
||||
entropy->restarts_to_go--;
|
||||
if (cinfo->restart_interval)
|
||||
entropy->restarts_to_go--;
|
||||
|
||||
return TRUE;
|
||||
|
||||
@@ -676,7 +668,7 @@ jinit_phuff_decoder(j_decompress_ptr cinfo)
|
||||
/* Create progression status table */
|
||||
cinfo->coef_bits = (int (*)[DCTSIZE2])
|
||||
(*cinfo->mem->alloc_small) ((j_common_ptr)cinfo, JPOOL_IMAGE,
|
||||
cinfo->num_components * DCTSIZE2 *
|
||||
cinfo->num_components * 2 * DCTSIZE2 *
|
||||
sizeof(int));
|
||||
coef_bit_ptr = &cinfo->coef_bits[0][0];
|
||||
for (ci = 0; ci < cinfo->num_components; ci++)
|
||||
|
||||
+22
-16
@@ -8,7 +8,7 @@
|
||||
* Copyright (C) 2010, 2015-2016, D. R. Commander.
|
||||
* Copyright (C) 2014, MIPS Technologies, Inc., California.
|
||||
* Copyright (C) 2015, Google, Inc.
|
||||
* Copyright (C) 2019, Arm Limited.
|
||||
* Copyright (C) 2019-2020, Arm Limited.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -177,7 +177,7 @@ int_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
outptr = output_data[outrow];
|
||||
outend = outptr + cinfo->output_width;
|
||||
while (outptr < outend) {
|
||||
invalue = *inptr++; /* don't need GETJSAMPLE() here */
|
||||
invalue = *inptr++;
|
||||
for (h = h_expand; h > 0; h--) {
|
||||
*outptr++ = invalue;
|
||||
}
|
||||
@@ -213,7 +213,7 @@ h2v1_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
outptr = output_data[inrow];
|
||||
outend = outptr + cinfo->output_width;
|
||||
while (outptr < outend) {
|
||||
invalue = *inptr++; /* don't need GETJSAMPLE() here */
|
||||
invalue = *inptr++;
|
||||
*outptr++ = invalue;
|
||||
*outptr++ = invalue;
|
||||
}
|
||||
@@ -242,7 +242,7 @@ h2v2_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
outptr = output_data[outrow];
|
||||
outend = outptr + cinfo->output_width;
|
||||
while (outptr < outend) {
|
||||
invalue = *inptr++; /* don't need GETJSAMPLE() here */
|
||||
invalue = *inptr++;
|
||||
*outptr++ = invalue;
|
||||
*outptr++ = invalue;
|
||||
}
|
||||
@@ -283,20 +283,20 @@ h2v1_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
inptr = input_data[inrow];
|
||||
outptr = output_data[inrow];
|
||||
/* Special case for first column */
|
||||
invalue = GETJSAMPLE(*inptr++);
|
||||
invalue = *inptr++;
|
||||
*outptr++ = (JSAMPLE)invalue;
|
||||
*outptr++ = (JSAMPLE)((invalue * 3 + GETJSAMPLE(*inptr) + 2) >> 2);
|
||||
*outptr++ = (JSAMPLE)((invalue * 3 + inptr[0] + 2) >> 2);
|
||||
|
||||
for (colctr = compptr->downsampled_width - 2; colctr > 0; colctr--) {
|
||||
/* General case: 3/4 * nearer pixel + 1/4 * further pixel */
|
||||
invalue = GETJSAMPLE(*inptr++) * 3;
|
||||
*outptr++ = (JSAMPLE)((invalue + GETJSAMPLE(inptr[-2]) + 1) >> 2);
|
||||
*outptr++ = (JSAMPLE)((invalue + GETJSAMPLE(*inptr) + 2) >> 2);
|
||||
invalue = (*inptr++) * 3;
|
||||
*outptr++ = (JSAMPLE)((invalue + inptr[-2] + 1) >> 2);
|
||||
*outptr++ = (JSAMPLE)((invalue + inptr[0] + 2) >> 2);
|
||||
}
|
||||
|
||||
/* Special case for last column */
|
||||
invalue = GETJSAMPLE(*inptr);
|
||||
*outptr++ = (JSAMPLE)((invalue * 3 + GETJSAMPLE(inptr[-1]) + 1) >> 2);
|
||||
invalue = *inptr;
|
||||
*outptr++ = (JSAMPLE)((invalue * 3 + inptr[-1] + 1) >> 2);
|
||||
*outptr++ = (JSAMPLE)invalue;
|
||||
}
|
||||
}
|
||||
@@ -338,7 +338,7 @@ h1v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
outptr = output_data[outrow++];
|
||||
|
||||
for (colctr = 0; colctr < compptr->downsampled_width; colctr++) {
|
||||
thiscolsum = GETJSAMPLE(*inptr0++) * 3 + GETJSAMPLE(*inptr1++);
|
||||
thiscolsum = (*inptr0++) * 3 + (*inptr1++);
|
||||
*outptr++ = (JSAMPLE)((thiscolsum + bias) >> 2);
|
||||
}
|
||||
}
|
||||
@@ -381,8 +381,8 @@ h2v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
outptr = output_data[outrow++];
|
||||
|
||||
/* Special case for first column */
|
||||
thiscolsum = GETJSAMPLE(*inptr0++) * 3 + GETJSAMPLE(*inptr1++);
|
||||
nextcolsum = GETJSAMPLE(*inptr0++) * 3 + GETJSAMPLE(*inptr1++);
|
||||
thiscolsum = (*inptr0++) * 3 + (*inptr1++);
|
||||
nextcolsum = (*inptr0++) * 3 + (*inptr1++);
|
||||
*outptr++ = (JSAMPLE)((thiscolsum * 4 + 8) >> 4);
|
||||
*outptr++ = (JSAMPLE)((thiscolsum * 3 + nextcolsum + 7) >> 4);
|
||||
lastcolsum = thiscolsum; thiscolsum = nextcolsum;
|
||||
@@ -390,7 +390,7 @@ h2v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
for (colctr = compptr->downsampled_width - 2; colctr > 0; colctr--) {
|
||||
/* General case: 3/4 * nearer pixel + 1/4 * further pixel in each */
|
||||
/* dimension, thus 9/16, 3/16, 3/16, 1/16 overall */
|
||||
nextcolsum = GETJSAMPLE(*inptr0++) * 3 + GETJSAMPLE(*inptr1++);
|
||||
nextcolsum = (*inptr0++) * 3 + (*inptr1++);
|
||||
*outptr++ = (JSAMPLE)((thiscolsum * 3 + lastcolsum + 8) >> 4);
|
||||
*outptr++ = (JSAMPLE)((thiscolsum * 3 + nextcolsum + 7) >> 4);
|
||||
lastcolsum = thiscolsum; thiscolsum = nextcolsum;
|
||||
@@ -477,7 +477,13 @@ jinit_upsampler(j_decompress_ptr cinfo)
|
||||
} else if (h_in_group == h_out_group &&
|
||||
v_in_group * 2 == v_out_group && do_fancy) {
|
||||
/* Non-fancy upsampling is handled by the generic method */
|
||||
upsample->methods[ci] = h1v2_fancy_upsample;
|
||||
#if defined(__arm__) || defined(__aarch64__) || \
|
||||
defined(_M_ARM) || defined(_M_ARM64)
|
||||
if (jsimd_can_h1v2_fancy_upsample())
|
||||
upsample->methods[ci] = jsimd_h1v2_fancy_upsample;
|
||||
else
|
||||
#endif
|
||||
upsample->methods[ci] = h1v2_fancy_upsample;
|
||||
upsample->pub.need_context_rows = TRUE;
|
||||
} else if (h_in_group * 2 == h_out_group &&
|
||||
v_in_group * 2 == v_out_group) {
|
||||
|
||||
Vendored
+8
-8
@@ -3,8 +3,8 @@
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1998, Thomas G. Lane.
|
||||
* It was modified by The libjpeg-turbo Project to include only code relevant
|
||||
* to libjpeg-turbo.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -189,13 +189,13 @@ format_message(j_common_ptr cinfo, char *buffer)
|
||||
|
||||
/* Format the message into the passed buffer */
|
||||
if (isstring)
|
||||
sprintf(buffer, msgtext, err->msg_parm.s);
|
||||
snprintf(buffer, JMSG_LENGTH_MAX, msgtext, err->msg_parm.s);
|
||||
else
|
||||
sprintf(buffer, msgtext,
|
||||
err->msg_parm.i[0], err->msg_parm.i[1],
|
||||
err->msg_parm.i[2], err->msg_parm.i[3],
|
||||
err->msg_parm.i[4], err->msg_parm.i[5],
|
||||
err->msg_parm.i[6], err->msg_parm.i[7]);
|
||||
snprintf(buffer, JMSG_LENGTH_MAX, msgtext,
|
||||
err->msg_parm.i[0], err->msg_parm.i[1],
|
||||
err->msg_parm.i[2], err->msg_parm.i[3],
|
||||
err->msg_parm.i[4], err->msg_parm.i[5],
|
||||
err->msg_parm.i[6], err->msg_parm.i[7]);
|
||||
}
|
||||
|
||||
|
||||
|
||||
Vendored
+17
-2
@@ -5,7 +5,7 @@
|
||||
* Copyright (C) 1994-1997, Thomas G. Lane.
|
||||
* Modified 1997-2009 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2014, 2017, D. R. Commander.
|
||||
* Copyright (C) 2014, 2017, 2021-2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -103,7 +103,7 @@ JMESSAGE(JERR_MISMATCHED_QUANT_TABLE,
|
||||
"Cannot transcode due to multiple use of quantization table %d")
|
||||
JMESSAGE(JERR_MISSING_DATA, "Scan script does not transmit all data")
|
||||
JMESSAGE(JERR_MODE_CHANGE, "Invalid color quantization mode change")
|
||||
JMESSAGE(JERR_NOTIMPL, "Not implemented yet")
|
||||
JMESSAGE(JERR_NOTIMPL, "Requested features are incompatible")
|
||||
JMESSAGE(JERR_NOT_COMPILED, "Requested feature was omitted at compile time")
|
||||
#if JPEG_LIB_VERSION >= 70
|
||||
JMESSAGE(JERR_NO_ARITH_TABLE, "Arithmetic table 0x%02x was not defined")
|
||||
@@ -207,6 +207,10 @@ JMESSAGE(JWRN_ARITH_BAD_CODE, "Corrupt JPEG data: bad arithmetic code")
|
||||
#endif
|
||||
#endif
|
||||
JMESSAGE(JWRN_BOGUS_ICC, "Corrupt JPEG data: bad ICC marker")
|
||||
#if JPEG_LIB_VERSION < 70
|
||||
JMESSAGE(JERR_BAD_DROP_SAMPLING,
|
||||
"Component index %d: mismatching sampling ratio %d:%d, %d:%d, %c")
|
||||
#endif
|
||||
|
||||
#ifdef JMAKE_ENUM_LIST
|
||||
|
||||
@@ -252,9 +256,19 @@ JMESSAGE(JWRN_BOGUS_ICC, "Corrupt JPEG data: bad ICC marker")
|
||||
(cinfo)->err->msg_parm.i[2] = (p3), \
|
||||
(cinfo)->err->msg_parm.i[3] = (p4), \
|
||||
(*(cinfo)->err->error_exit) ((j_common_ptr)(cinfo)))
|
||||
#define ERREXIT6(cinfo, code, p1, p2, p3, p4, p5, p6) \
|
||||
((cinfo)->err->msg_code = (code), \
|
||||
(cinfo)->err->msg_parm.i[0] = (p1), \
|
||||
(cinfo)->err->msg_parm.i[1] = (p2), \
|
||||
(cinfo)->err->msg_parm.i[2] = (p3), \
|
||||
(cinfo)->err->msg_parm.i[3] = (p4), \
|
||||
(cinfo)->err->msg_parm.i[4] = (p5), \
|
||||
(cinfo)->err->msg_parm.i[5] = (p6), \
|
||||
(*(cinfo)->err->error_exit) ((j_common_ptr)(cinfo)))
|
||||
#define ERREXITS(cinfo, code, str) \
|
||||
((cinfo)->err->msg_code = (code), \
|
||||
strncpy((cinfo)->err->msg_parm.s, (str), JMSG_STR_PARM_MAX), \
|
||||
(cinfo)->err->msg_parm.s[JMSG_STR_PARM_MAX - 1] = '\0', \
|
||||
(*(cinfo)->err->error_exit) ((j_common_ptr)(cinfo)))
|
||||
|
||||
#define MAKESTMT(stuff) do { stuff } while (0)
|
||||
@@ -311,6 +325,7 @@ JMESSAGE(JWRN_BOGUS_ICC, "Corrupt JPEG data: bad ICC marker")
|
||||
#define TRACEMSS(cinfo, lvl, code, str) \
|
||||
((cinfo)->err->msg_code = (code), \
|
||||
strncpy((cinfo)->err->msg_parm.s, (str), JMSG_STR_PARM_MAX), \
|
||||
(cinfo)->err->msg_parm.s[JMSG_STR_PARM_MAX - 1] = '\0', \
|
||||
(*(cinfo)->err->emit_message) ((j_common_ptr)(cinfo), (lvl)))
|
||||
|
||||
#endif /* JERROR_H */
|
||||
|
||||
+4
-4
@@ -3,7 +3,7 @@
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1998, Thomas G. Lane.
|
||||
* Modification developed 2002-2009 by Guido Vollbeding.
|
||||
* Modification developed 2002-2018 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2015, 2020, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
@@ -417,7 +417,7 @@ jpeg_idct_islow(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
|
||||
/*
|
||||
* Perform dequantization and inverse DCT on one block of coefficients,
|
||||
* producing a 7x7 output block.
|
||||
* producing a reduced-size 7x7 output block.
|
||||
*
|
||||
* Optimized algorithm with 12 multiplications in the 1-D kernel.
|
||||
* cK represents sqrt(2) * cos(K*pi/14).
|
||||
@@ -1258,7 +1258,7 @@ jpeg_idct_10x10(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
|
||||
/*
|
||||
* Perform dequantization and inverse DCT on one block of coefficients,
|
||||
* producing a 11x11 output block.
|
||||
* producing an 11x11 output block.
|
||||
*
|
||||
* Optimized algorithm with 24 multiplications in the 1-D kernel.
|
||||
* cK represents sqrt(2) * cos(K*pi/22).
|
||||
@@ -2398,7 +2398,7 @@ jpeg_idct_16x16(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
tmp0 = DEQUANTIZE(inptr[DCTSIZE * 0], quantptr[DCTSIZE * 0]);
|
||||
tmp0 = LEFT_SHIFT(tmp0, CONST_BITS);
|
||||
/* Add fudge factor here for final descale. */
|
||||
tmp0 += 1 << (CONST_BITS - PASS1_BITS - 1);
|
||||
tmp0 += ONE << (CONST_BITS - PASS1_BITS - 1);
|
||||
|
||||
z1 = DEQUANTIZE(inptr[DCTSIZE * 4], quantptr[DCTSIZE * 4]);
|
||||
tmp1 = MULTIPLY(z1, FIX(1.306562965)); /* c4[16] = c2[8] */
|
||||
|
||||
+95
-50
@@ -3,8 +3,8 @@
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1994, Thomas G. Lane.
|
||||
* It was modified by The libjpeg-turbo Project to include only code relevant
|
||||
* to libjpeg-turbo.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -17,72 +17,117 @@
|
||||
* JPEG library. Most applications need only include jpeglib.h.
|
||||
*/
|
||||
|
||||
#ifndef __JINCLUDE_H__
|
||||
#define __JINCLUDE_H__
|
||||
|
||||
/* Include auto-config file to find out which system include files we need. */
|
||||
|
||||
#include "jconfig.h" /* auto configuration options */
|
||||
#include "jconfigint.h"
|
||||
#define JCONFIG_INCLUDED /* so that jpeglib.h doesn't do it again */
|
||||
|
||||
/*
|
||||
* We need the NULL macro and size_t typedef.
|
||||
* On an ANSI-conforming system it is sufficient to include <stddef.h>.
|
||||
* Otherwise, we get them from <stdlib.h> or <stdio.h>; we may have to
|
||||
* pull in <sys/types.h> as well.
|
||||
* Note that the core JPEG library does not require <stdio.h>;
|
||||
* only the default error handler and data source/destination modules do.
|
||||
* But we must pull it in because of the references to FILE in jpeglib.h.
|
||||
* You can remove those references if you want to compile without <stdio.h>.
|
||||
*/
|
||||
|
||||
#ifdef HAVE_STDDEF_H
|
||||
#include <stddef.h>
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_STDLIB_H
|
||||
#include <stdlib.h>
|
||||
#endif
|
||||
|
||||
#ifdef NEED_SYS_TYPES_H
|
||||
#include <sys/types.h>
|
||||
#endif
|
||||
|
||||
#include <stdio.h>
|
||||
|
||||
/*
|
||||
* We need memory copying and zeroing functions, plus strncpy().
|
||||
* ANSI and System V implementations declare these in <string.h>.
|
||||
* BSD doesn't have the mem() functions, but it does have bcopy()/bzero().
|
||||
* Some systems may declare memset and memcpy in <memory.h>.
|
||||
*
|
||||
* NOTE: we assume the size parameters to these functions are of type size_t.
|
||||
* Change the casts in these macros if not!
|
||||
*/
|
||||
|
||||
#ifdef NEED_BSD_STRINGS
|
||||
|
||||
#include <strings.h>
|
||||
#define MEMZERO(target, size) \
|
||||
bzero((void *)(target), (size_t)(size))
|
||||
#define MEMCOPY(dest, src, size) \
|
||||
bcopy((const void *)(src), (void *)(dest), (size_t)(size))
|
||||
|
||||
#else /* not BSD, assume ANSI/SysV string lib */
|
||||
|
||||
#include <string.h>
|
||||
#define MEMZERO(target, size) \
|
||||
memset((void *)(target), 0, (size_t)(size))
|
||||
#define MEMCOPY(dest, src, size) \
|
||||
memcpy((void *)(dest), (const void *)(src), (size_t)(size))
|
||||
|
||||
#endif
|
||||
|
||||
/*
|
||||
* The modules that use fread() and fwrite() always invoke them through
|
||||
* these macros. On some systems you may need to twiddle the argument casts.
|
||||
* CAUTION: argument order is different from underlying functions!
|
||||
* These macros/inline functions facilitate using Microsoft's "safe string"
|
||||
* functions with Visual Studio builds without the need to scatter #ifdefs
|
||||
* throughout the code base.
|
||||
*/
|
||||
|
||||
#define JFREAD(file, buf, sizeofbuf) \
|
||||
((size_t)fread((void *)(buf), (size_t)1, (size_t)(sizeofbuf), (file)))
|
||||
#define JFWRITE(file, buf, sizeofbuf) \
|
||||
((size_t)fwrite((const void *)(buf), (size_t)1, (size_t)(sizeofbuf), (file)))
|
||||
|
||||
#ifndef NO_GETENV
|
||||
|
||||
#ifdef _MSC_VER
|
||||
|
||||
static INLINE int GETENV_S(char *buffer, size_t buffer_size, const char *name)
|
||||
{
|
||||
size_t required_size;
|
||||
|
||||
return (int)getenv_s(&required_size, buffer, buffer_size, name);
|
||||
}
|
||||
|
||||
#else /* _MSC_VER */
|
||||
|
||||
#include <errno.h>
|
||||
|
||||
/* This provides a similar interface to the Microsoft/C11 getenv_s() function,
|
||||
* but other than parameter validation, it has no advantages over getenv().
|
||||
*/
|
||||
|
||||
static INLINE int GETENV_S(char *buffer, size_t buffer_size, const char *name)
|
||||
{
|
||||
char *env;
|
||||
|
||||
if (!buffer) {
|
||||
if (buffer_size == 0)
|
||||
return 0;
|
||||
else
|
||||
return (errno = EINVAL);
|
||||
}
|
||||
if (buffer_size == 0)
|
||||
return (errno = EINVAL);
|
||||
if (!name) {
|
||||
*buffer = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
env = getenv(name);
|
||||
if (!env)
|
||||
{
|
||||
*buffer = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (strlen(env) + 1 > buffer_size) {
|
||||
*buffer = 0;
|
||||
return ERANGE;
|
||||
}
|
||||
|
||||
strncpy(buffer, env, buffer_size);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
#endif /* _MSC_VER */
|
||||
|
||||
#endif /* NO_GETENV */
|
||||
|
||||
|
||||
#ifndef NO_PUTENV
|
||||
|
||||
#ifdef _WIN32
|
||||
|
||||
#define PUTENV_S(name, value) _putenv_s(name, value)
|
||||
|
||||
#else
|
||||
|
||||
/* This provides a similar interface to the Microsoft _putenv_s() function, but
|
||||
* other than parameter validation, it has no advantages over setenv().
|
||||
*/
|
||||
|
||||
static INLINE int PUTENV_S(const char *name, const char *value)
|
||||
{
|
||||
if (!name || !value)
|
||||
return (errno = EINVAL);
|
||||
|
||||
setenv(name, value, 1);
|
||||
|
||||
return errno;
|
||||
}
|
||||
|
||||
#endif /* _WIN32 */
|
||||
|
||||
#endif /* NO_PUTENV */
|
||||
|
||||
|
||||
#endif /* JINCLUDE_H */
|
||||
|
||||
Vendored
+9
-11
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2016, D. R. Commander.
|
||||
* Copyright (C) 2016, 2021-2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -37,12 +37,6 @@
|
||||
#endif
|
||||
#include <limits.h>
|
||||
|
||||
#ifndef NO_GETENV
|
||||
#ifndef HAVE_STDLIB_H /* <stdlib.h> should declare getenv() */
|
||||
extern char *getenv(const char *name);
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
LOCAL(size_t)
|
||||
round_up_pow2(size_t a, size_t b)
|
||||
@@ -1032,7 +1026,7 @@ free_pool(j_common_ptr cinfo, int pool_id)
|
||||
large_pool_ptr next_lhdr_ptr = lhdr_ptr->next;
|
||||
space_freed = lhdr_ptr->bytes_used +
|
||||
lhdr_ptr->bytes_left +
|
||||
sizeof(large_pool_hdr);
|
||||
sizeof(large_pool_hdr) + ALIGN_SIZE - 1;
|
||||
jpeg_free_large(cinfo, (void *)lhdr_ptr, space_freed);
|
||||
mem->total_space_allocated -= space_freed;
|
||||
lhdr_ptr = next_lhdr_ptr;
|
||||
@@ -1045,7 +1039,7 @@ free_pool(j_common_ptr cinfo, int pool_id)
|
||||
while (shdr_ptr != NULL) {
|
||||
small_pool_ptr next_shdr_ptr = shdr_ptr->next;
|
||||
space_freed = shdr_ptr->bytes_used + shdr_ptr->bytes_left +
|
||||
sizeof(small_pool_hdr);
|
||||
sizeof(small_pool_hdr) + ALIGN_SIZE - 1;
|
||||
jpeg_free_small(cinfo, (void *)shdr_ptr, space_freed);
|
||||
mem->total_space_allocated -= space_freed;
|
||||
shdr_ptr = next_shdr_ptr;
|
||||
@@ -1162,12 +1156,16 @@ jinit_memory_mgr(j_common_ptr cinfo)
|
||||
*/
|
||||
#ifndef NO_GETENV
|
||||
{
|
||||
char *memenv;
|
||||
char memenv[30] = { 0 };
|
||||
|
||||
if ((memenv = getenv("JPEGMEM")) != NULL) {
|
||||
if (!GETENV_S(memenv, 30, "JPEGMEM") && strlen(memenv) > 0) {
|
||||
char ch = 'x';
|
||||
|
||||
#ifdef _MSC_VER
|
||||
if (sscanf_s(memenv, "%ld%c", &max_to_use, &ch, 1) > 0) {
|
||||
#else
|
||||
if (sscanf(memenv, "%ld%c", &max_to_use, &ch) > 0) {
|
||||
#endif
|
||||
if (ch == 'm' || ch == 'M')
|
||||
max_to_use *= 1000L;
|
||||
mem->pub.max_memory_to_use = max_to_use * 1000L;
|
||||
|
||||
-5
@@ -22,11 +22,6 @@
|
||||
#include "jpeglib.h"
|
||||
#include "jmemsys.h" /* import the system-dependent declarations */
|
||||
|
||||
#ifndef HAVE_STDLIB_H /* <stdlib.h> should declare malloc(),free() */
|
||||
extern void *malloc(size_t size);
|
||||
extern void free(void *ptr);
|
||||
#endif
|
||||
|
||||
|
||||
/*
|
||||
* Memory allocation and freeing are controlled by the regular library
|
||||
|
||||
-39
@@ -43,25 +43,11 @@
|
||||
|
||||
#if BITS_IN_JSAMPLE == 8
|
||||
/* JSAMPLE should be the smallest type that will hold the values 0..255.
|
||||
* You can use a signed char by having GETJSAMPLE mask it with 0xFF.
|
||||
*/
|
||||
|
||||
#ifdef HAVE_UNSIGNED_CHAR
|
||||
|
||||
typedef unsigned char JSAMPLE;
|
||||
#define GETJSAMPLE(value) ((int)(value))
|
||||
|
||||
#else /* not HAVE_UNSIGNED_CHAR */
|
||||
|
||||
typedef char JSAMPLE;
|
||||
#ifdef __CHAR_UNSIGNED__
|
||||
#define GETJSAMPLE(value) ((int)(value))
|
||||
#else
|
||||
#define GETJSAMPLE(value) ((int)(value) & 0xFF)
|
||||
#endif /* __CHAR_UNSIGNED__ */
|
||||
|
||||
#endif /* HAVE_UNSIGNED_CHAR */
|
||||
|
||||
#define MAXJSAMPLE 255
|
||||
#define CENTERJSAMPLE 128
|
||||
|
||||
@@ -97,22 +83,9 @@ typedef short JCOEF;
|
||||
* managers, this is also the data type passed to fread/fwrite.
|
||||
*/
|
||||
|
||||
#ifdef HAVE_UNSIGNED_CHAR
|
||||
|
||||
typedef unsigned char JOCTET;
|
||||
#define GETJOCTET(value) (value)
|
||||
|
||||
#else /* not HAVE_UNSIGNED_CHAR */
|
||||
|
||||
typedef char JOCTET;
|
||||
#ifdef __CHAR_UNSIGNED__
|
||||
#define GETJOCTET(value) (value)
|
||||
#else
|
||||
#define GETJOCTET(value) ((value) & 0xFF)
|
||||
#endif /* __CHAR_UNSIGNED__ */
|
||||
|
||||
#endif /* HAVE_UNSIGNED_CHAR */
|
||||
|
||||
|
||||
/* These typedefs are used for various table entries and so forth.
|
||||
* They must be at least as wide as specified; but making them too big
|
||||
@@ -123,23 +96,11 @@ typedef char JOCTET;
|
||||
|
||||
/* UINT8 must hold at least the values 0..255. */
|
||||
|
||||
#ifdef HAVE_UNSIGNED_CHAR
|
||||
typedef unsigned char UINT8;
|
||||
#else /* not HAVE_UNSIGNED_CHAR */
|
||||
#ifdef __CHAR_UNSIGNED__
|
||||
typedef char UINT8;
|
||||
#else /* not __CHAR_UNSIGNED__ */
|
||||
typedef short UINT8;
|
||||
#endif /* __CHAR_UNSIGNED__ */
|
||||
#endif /* HAVE_UNSIGNED_CHAR */
|
||||
|
||||
/* UINT16 must hold at least the values 0..65535. */
|
||||
|
||||
#ifdef HAVE_UNSIGNED_SHORT
|
||||
typedef unsigned short UINT16;
|
||||
#else /* not HAVE_UNSIGNED_SHORT */
|
||||
typedef unsigned int UINT16;
|
||||
#endif /* HAVE_UNSIGNED_SHORT */
|
||||
|
||||
/* INT16 must hold at least the values -32768..32767. */
|
||||
|
||||
|
||||
Vendored
+17
-10
@@ -5,8 +5,9 @@
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* Modified 1997-2009 by Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2015-2016, D. R. Commander.
|
||||
* Copyright (C) 2015-2016, 2019, 2021, D. R. Commander.
|
||||
* Copyright (C) 2015, Google, Inc.
|
||||
* Copyright (C) 2021, Alex Richardson.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -47,6 +48,18 @@ typedef enum { /* Operating modes for buffer controllers */
|
||||
/* JLONG must hold at least signed 32-bit values. */
|
||||
typedef long JLONG;
|
||||
|
||||
/* JUINTPTR must hold pointer values. */
|
||||
#ifdef __UINTPTR_TYPE__
|
||||
/*
|
||||
* __UINTPTR_TYPE__ is GNU-specific and available in GCC 4.6+ and Clang 3.0+.
|
||||
* Fortunately, that is sufficient to support the few architectures for which
|
||||
* sizeof(void *) != sizeof(size_t). The only other options would require C99
|
||||
* or Clang-specific builtins.
|
||||
*/
|
||||
typedef __UINTPTR_TYPE__ JUINTPTR;
|
||||
#else
|
||||
typedef size_t JUINTPTR;
|
||||
#endif
|
||||
|
||||
/*
|
||||
* Left shift macro that handles a negative operand without causing any
|
||||
@@ -158,6 +171,9 @@ struct jpeg_decomp_master {
|
||||
JDIMENSION first_MCU_col[MAX_COMPONENTS];
|
||||
JDIMENSION last_MCU_col[MAX_COMPONENTS];
|
||||
boolean jinit_upsampler_no_alloc;
|
||||
|
||||
/* Last iMCU row that was successfully decoded */
|
||||
JDIMENSION last_good_iMCU_row;
|
||||
};
|
||||
|
||||
/* Input control module */
|
||||
@@ -357,12 +373,3 @@ extern const int jpeg_natural_order[]; /* zigzag coef order to natural order */
|
||||
|
||||
/* Arithmetic coding probability estimation tables in jaricom.c */
|
||||
extern const JLONG jpeg_aritab[];
|
||||
|
||||
/* Suppress undefined-structure complaints if necessary. */
|
||||
|
||||
#ifdef INCOMPLETE_TYPES_BROKEN
|
||||
#ifndef AM_MEMORY_MANAGER /* only jmemmgr.c defines these */
|
||||
struct jvirt_sarray_control { long dummy; };
|
||||
struct jvirt_barray_control { long dummy; };
|
||||
#endif
|
||||
#endif /* INCOMPLETE_TYPES_BROKEN */
|
||||
|
||||
Vendored
+12
-15
@@ -479,7 +479,7 @@ color_quantize(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
for (col = width; col > 0; col--) {
|
||||
pixcode = 0;
|
||||
for (ci = 0; ci < nc; ci++) {
|
||||
pixcode += GETJSAMPLE(colorindex[ci][GETJSAMPLE(*ptrin++)]);
|
||||
pixcode += colorindex[ci][*ptrin++];
|
||||
}
|
||||
*ptrout++ = (JSAMPLE)pixcode;
|
||||
}
|
||||
@@ -506,9 +506,9 @@ color_quantize3(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
ptrin = input_buf[row];
|
||||
ptrout = output_buf[row];
|
||||
for (col = width; col > 0; col--) {
|
||||
pixcode = GETJSAMPLE(colorindex0[GETJSAMPLE(*ptrin++)]);
|
||||
pixcode += GETJSAMPLE(colorindex1[GETJSAMPLE(*ptrin++)]);
|
||||
pixcode += GETJSAMPLE(colorindex2[GETJSAMPLE(*ptrin++)]);
|
||||
pixcode = colorindex0[*ptrin++];
|
||||
pixcode += colorindex1[*ptrin++];
|
||||
pixcode += colorindex2[*ptrin++];
|
||||
*ptrout++ = (JSAMPLE)pixcode;
|
||||
}
|
||||
}
|
||||
@@ -552,7 +552,7 @@ quantize_ord_dither(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
* required amount of padding.
|
||||
*/
|
||||
*output_ptr +=
|
||||
colorindex_ci[GETJSAMPLE(*input_ptr) + dither[col_index]];
|
||||
colorindex_ci[*input_ptr + dither[col_index]];
|
||||
input_ptr += nc;
|
||||
output_ptr++;
|
||||
col_index = (col_index + 1) & ODITHER_MASK;
|
||||
@@ -595,12 +595,9 @@ quantize3_ord_dither(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
col_index = 0;
|
||||
|
||||
for (col = width; col > 0; col--) {
|
||||
pixcode =
|
||||
GETJSAMPLE(colorindex0[GETJSAMPLE(*input_ptr++) + dither0[col_index]]);
|
||||
pixcode +=
|
||||
GETJSAMPLE(colorindex1[GETJSAMPLE(*input_ptr++) + dither1[col_index]]);
|
||||
pixcode +=
|
||||
GETJSAMPLE(colorindex2[GETJSAMPLE(*input_ptr++) + dither2[col_index]]);
|
||||
pixcode = colorindex0[(*input_ptr++) + dither0[col_index]];
|
||||
pixcode += colorindex1[(*input_ptr++) + dither1[col_index]];
|
||||
pixcode += colorindex2[(*input_ptr++) + dither2[col_index]];
|
||||
*output_ptr++ = (JSAMPLE)pixcode;
|
||||
col_index = (col_index + 1) & ODITHER_MASK;
|
||||
}
|
||||
@@ -677,15 +674,15 @@ quantize_fs_dither(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
* The maximum error is +- MAXJSAMPLE; this sets the required size
|
||||
* of the range_limit array.
|
||||
*/
|
||||
cur += GETJSAMPLE(*input_ptr);
|
||||
cur = GETJSAMPLE(range_limit[cur]);
|
||||
cur += *input_ptr;
|
||||
cur = range_limit[cur];
|
||||
/* Select output value, accumulate into output code for this pixel */
|
||||
pixcode = GETJSAMPLE(colorindex_ci[cur]);
|
||||
pixcode = colorindex_ci[cur];
|
||||
*output_ptr += (JSAMPLE)pixcode;
|
||||
/* Compute actual representation error at this pixel */
|
||||
/* Note: we can do this even though we don't have the final */
|
||||
/* pixel code, because the colormap is orthogonal. */
|
||||
cur -= GETJSAMPLE(colormap_ci[pixcode]);
|
||||
cur -= colormap_ci[pixcode];
|
||||
/* Compute error fractions to be propagated to adjacent pixels.
|
||||
* Add these into the running sums, and simultaneously shift the
|
||||
* next-line error sums left by 1 column.
|
||||
|
||||
Vendored
+23
-23
@@ -215,9 +215,9 @@ prescan_quantize(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
ptr = input_buf[row];
|
||||
for (col = width; col > 0; col--) {
|
||||
/* get pixel value and index into the histogram */
|
||||
histp = &histogram[GETJSAMPLE(ptr[0]) >> C0_SHIFT]
|
||||
[GETJSAMPLE(ptr[1]) >> C1_SHIFT]
|
||||
[GETJSAMPLE(ptr[2]) >> C2_SHIFT];
|
||||
histp = &histogram[ptr[0] >> C0_SHIFT]
|
||||
[ptr[1] >> C1_SHIFT]
|
||||
[ptr[2] >> C2_SHIFT];
|
||||
/* increment, check for overflow and undo increment if so. */
|
||||
if (++(*histp) <= 0)
|
||||
(*histp)--;
|
||||
@@ -665,7 +665,7 @@ find_nearby_colors(j_decompress_ptr cinfo, int minc0, int minc1, int minc2,
|
||||
|
||||
for (i = 0; i < numcolors; i++) {
|
||||
/* We compute the squared-c0-distance term, then add in the other two. */
|
||||
x = GETJSAMPLE(cinfo->colormap[0][i]);
|
||||
x = cinfo->colormap[0][i];
|
||||
if (x < minc0) {
|
||||
tdist = (x - minc0) * C0_SCALE;
|
||||
min_dist = tdist * tdist;
|
||||
@@ -688,7 +688,7 @@ find_nearby_colors(j_decompress_ptr cinfo, int minc0, int minc1, int minc2,
|
||||
}
|
||||
}
|
||||
|
||||
x = GETJSAMPLE(cinfo->colormap[1][i]);
|
||||
x = cinfo->colormap[1][i];
|
||||
if (x < minc1) {
|
||||
tdist = (x - minc1) * C1_SCALE;
|
||||
min_dist += tdist * tdist;
|
||||
@@ -710,7 +710,7 @@ find_nearby_colors(j_decompress_ptr cinfo, int minc0, int minc1, int minc2,
|
||||
}
|
||||
}
|
||||
|
||||
x = GETJSAMPLE(cinfo->colormap[2][i]);
|
||||
x = cinfo->colormap[2][i];
|
||||
if (x < minc2) {
|
||||
tdist = (x - minc2) * C2_SCALE;
|
||||
min_dist += tdist * tdist;
|
||||
@@ -788,13 +788,13 @@ find_best_colors(j_decompress_ptr cinfo, int minc0, int minc1, int minc2,
|
||||
#define STEP_C2 ((1 << C2_SHIFT) * C2_SCALE)
|
||||
|
||||
for (i = 0; i < numcolors; i++) {
|
||||
icolor = GETJSAMPLE(colorlist[i]);
|
||||
icolor = colorlist[i];
|
||||
/* Compute (square of) distance from minc0/c1/c2 to this color */
|
||||
inc0 = (minc0 - GETJSAMPLE(cinfo->colormap[0][icolor])) * C0_SCALE;
|
||||
inc0 = (minc0 - cinfo->colormap[0][icolor]) * C0_SCALE;
|
||||
dist0 = inc0 * inc0;
|
||||
inc1 = (minc1 - GETJSAMPLE(cinfo->colormap[1][icolor])) * C1_SCALE;
|
||||
inc1 = (minc1 - cinfo->colormap[1][icolor]) * C1_SCALE;
|
||||
dist0 += inc1 * inc1;
|
||||
inc2 = (minc2 - GETJSAMPLE(cinfo->colormap[2][icolor])) * C2_SCALE;
|
||||
inc2 = (minc2 - cinfo->colormap[2][icolor]) * C2_SCALE;
|
||||
dist0 += inc2 * inc2;
|
||||
/* Form the initial difference increments */
|
||||
inc0 = inc0 * (2 * STEP_C0) + STEP_C0 * STEP_C0;
|
||||
@@ -879,7 +879,7 @@ fill_inverse_cmap(j_decompress_ptr cinfo, int c0, int c1, int c2)
|
||||
for (ic1 = 0; ic1 < BOX_C1_ELEMS; ic1++) {
|
||||
cachep = &histogram[c0 + ic0][c1 + ic1][c2];
|
||||
for (ic2 = 0; ic2 < BOX_C2_ELEMS; ic2++) {
|
||||
*cachep++ = (histcell)(GETJSAMPLE(*cptr++) + 1);
|
||||
*cachep++ = (histcell)((*cptr++) + 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -909,9 +909,9 @@ pass2_no_dither(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
outptr = output_buf[row];
|
||||
for (col = width; col > 0; col--) {
|
||||
/* get pixel value and index into the cache */
|
||||
c0 = GETJSAMPLE(*inptr++) >> C0_SHIFT;
|
||||
c1 = GETJSAMPLE(*inptr++) >> C1_SHIFT;
|
||||
c2 = GETJSAMPLE(*inptr++) >> C2_SHIFT;
|
||||
c0 = (*inptr++) >> C0_SHIFT;
|
||||
c1 = (*inptr++) >> C1_SHIFT;
|
||||
c2 = (*inptr++) >> C2_SHIFT;
|
||||
cachep = &histogram[c0][c1][c2];
|
||||
/* If we have not seen this color before, find nearest colormap entry */
|
||||
/* and update the cache */
|
||||
@@ -996,12 +996,12 @@ pass2_fs_dither(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
* The maximum error is +- MAXJSAMPLE (or less with error limiting);
|
||||
* this sets the required size of the range_limit array.
|
||||
*/
|
||||
cur0 += GETJSAMPLE(inptr[0]);
|
||||
cur1 += GETJSAMPLE(inptr[1]);
|
||||
cur2 += GETJSAMPLE(inptr[2]);
|
||||
cur0 = GETJSAMPLE(range_limit[cur0]);
|
||||
cur1 = GETJSAMPLE(range_limit[cur1]);
|
||||
cur2 = GETJSAMPLE(range_limit[cur2]);
|
||||
cur0 += inptr[0];
|
||||
cur1 += inptr[1];
|
||||
cur2 += inptr[2];
|
||||
cur0 = range_limit[cur0];
|
||||
cur1 = range_limit[cur1];
|
||||
cur2 = range_limit[cur2];
|
||||
/* Index into the cache with adjusted pixel value */
|
||||
cachep =
|
||||
&histogram[cur0 >> C0_SHIFT][cur1 >> C1_SHIFT][cur2 >> C2_SHIFT];
|
||||
@@ -1015,9 +1015,9 @@ pass2_fs_dither(j_decompress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
register int pixcode = *cachep - 1;
|
||||
*outptr = (JSAMPLE)pixcode;
|
||||
/* Compute representation error for this pixel */
|
||||
cur0 -= GETJSAMPLE(colormap0[pixcode]);
|
||||
cur1 -= GETJSAMPLE(colormap1[pixcode]);
|
||||
cur2 -= GETJSAMPLE(colormap2[pixcode]);
|
||||
cur0 -= colormap0[pixcode];
|
||||
cur1 -= colormap1[pixcode];
|
||||
cur2 -= colormap2[pixcode];
|
||||
}
|
||||
/* Compute error fractions to be propagated to adjacent pixels.
|
||||
* Add these into the running sums, and simultaneously shift the
|
||||
|
||||
Vendored
+6
@@ -4,6 +4,7 @@
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2011, 2014, D. R. Commander.
|
||||
* Copyright (C) 2015-2016, 2018, Matthieu Darbois.
|
||||
* Copyright (C) 2020, Arm Limited.
|
||||
*
|
||||
* Based on the x86 SIMD extension for IJG JPEG library,
|
||||
* Copyright (C) 1999-2006, MIYASAKA Masaru.
|
||||
@@ -75,6 +76,7 @@ EXTERN(void) jsimd_int_upsample(j_decompress_ptr cinfo,
|
||||
|
||||
EXTERN(int) jsimd_can_h2v2_fancy_upsample(void);
|
||||
EXTERN(int) jsimd_can_h2v1_fancy_upsample(void);
|
||||
EXTERN(int) jsimd_can_h1v2_fancy_upsample(void);
|
||||
|
||||
EXTERN(void) jsimd_h2v2_fancy_upsample(j_decompress_ptr cinfo,
|
||||
jpeg_component_info *compptr,
|
||||
@@ -84,6 +86,10 @@ EXTERN(void) jsimd_h2v1_fancy_upsample(j_decompress_ptr cinfo,
|
||||
jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr);
|
||||
EXTERN(void) jsimd_h1v2_fancy_upsample(j_decompress_ptr cinfo,
|
||||
jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr);
|
||||
|
||||
EXTERN(int) jsimd_can_h2v2_merged_upsample(void);
|
||||
EXTERN(int) jsimd_can_h2v1_merged_upsample(void);
|
||||
|
||||
+13
@@ -4,6 +4,7 @@
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2009-2011, 2014, D. R. Commander.
|
||||
* Copyright (C) 2015-2016, 2018, Matthieu Darbois.
|
||||
* Copyright (C) 2020, Arm Limited.
|
||||
*
|
||||
* Based on the x86 SIMD extension for IJG JPEG library,
|
||||
* Copyright (C) 1999-2006, MIYASAKA Masaru.
|
||||
@@ -169,6 +170,12 @@ jsimd_can_h2v1_fancy_upsample(void)
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h1v2_fancy_upsample(void)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
@@ -181,6 +188,12 @@ jsimd_h2v1_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
{
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h1v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v2_merged_upsample(void)
|
||||
{
|
||||
|
||||
+5
-4
@@ -4,7 +4,7 @@
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1998, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2013, D. R. Commander.
|
||||
* Copyright (C) 2013, 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -29,7 +29,7 @@ add_huff_table(j_common_ptr cinfo, JHUFF_TBL **htblptr, const UINT8 *bits,
|
||||
return;
|
||||
|
||||
/* Copy the number-of-symbols-of-each-code-length counts */
|
||||
MEMCOPY((*htblptr)->bits, bits, sizeof((*htblptr)->bits));
|
||||
memcpy((*htblptr)->bits, bits, sizeof((*htblptr)->bits));
|
||||
|
||||
/* Validate the counts. We do this here mainly so we can copy the right
|
||||
* number of symbols from the val[] array, without risking marching off
|
||||
@@ -41,8 +41,9 @@ add_huff_table(j_common_ptr cinfo, JHUFF_TBL **htblptr, const UINT8 *bits,
|
||||
if (nsymbols < 1 || nsymbols > 256)
|
||||
ERREXIT(cinfo, JERR_BAD_HUFF_TABLE);
|
||||
|
||||
MEMCOPY((*htblptr)->huffval, val, nsymbols * sizeof(UINT8));
|
||||
MEMZERO(&((*htblptr)->huffval[nsymbols]), (256 - nsymbols) * sizeof(UINT8));
|
||||
memcpy((*htblptr)->huffval, val, nsymbols * sizeof(UINT8));
|
||||
memset(&((*htblptr)->huffval[nsymbols]), 0,
|
||||
(256 - nsymbols) * sizeof(UINT8));
|
||||
|
||||
/* Initialize sent_table FALSE so table will be written to JPEG file. */
|
||||
(*htblptr)->sent_table = FALSE;
|
||||
|
||||
Vendored
+5
-5
@@ -3,8 +3,8 @@
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1996, Thomas G. Lane.
|
||||
* It was modified by The libjpeg-turbo Project to include only code
|
||||
* relevant to libjpeg-turbo.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -110,7 +110,7 @@ jcopy_sample_rows(JSAMPARRAY input_array, int source_row,
|
||||
for (row = num_rows; row > 0; row--) {
|
||||
inptr = *input_array++;
|
||||
outptr = *output_array++;
|
||||
MEMCOPY(outptr, inptr, count);
|
||||
memcpy(outptr, inptr, count);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -120,7 +120,7 @@ jcopy_block_row(JBLOCKROW input_row, JBLOCKROW output_row,
|
||||
JDIMENSION num_blocks)
|
||||
/* Copy a row of coefficient blocks from one place to another. */
|
||||
{
|
||||
MEMCOPY(output_row, input_row, num_blocks * (DCTSIZE2 * sizeof(JCOEF)));
|
||||
memcpy(output_row, input_row, num_blocks * (DCTSIZE2 * sizeof(JCOEF)));
|
||||
}
|
||||
|
||||
|
||||
@@ -129,5 +129,5 @@ jzero_far(void *target, size_t bytestozero)
|
||||
/* Zero out a chunk of memory. */
|
||||
/* This might be sample-array data, block-array data, or alloc_large data. */
|
||||
{
|
||||
MEMZERO(target, bytestozero);
|
||||
memset(target, 0, bytestozero);
|
||||
}
|
||||
|
||||
+6
-6
@@ -2,9 +2,9 @@
|
||||
* jversion.h
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-2012, Thomas G. Lane, Guido Vollbeding.
|
||||
* Copyright (C) 1991-2020, Thomas G. Lane, Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2010, 2012-2020, D. R. Commander.
|
||||
* Copyright (C) 2010, 2012-2021, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
@@ -37,9 +37,9 @@
|
||||
*/
|
||||
|
||||
#define JCOPYRIGHT \
|
||||
"Copyright (C) 2009-2020 D. R. Commander\n" \
|
||||
"Copyright (C) 2009-2021 D. R. Commander\n" \
|
||||
"Copyright (C) 2015, 2020 Google, Inc.\n" \
|
||||
"Copyright (C) 2019 Arm Limited\n" \
|
||||
"Copyright (C) 2019-2020 Arm Limited\n" \
|
||||
"Copyright (C) 2015-2016, 2018 Matthieu Darbois\n" \
|
||||
"Copyright (C) 2011-2016 Siarhei Siamashka\n" \
|
||||
"Copyright (C) 2015 Intel Corporation\n" \
|
||||
@@ -48,7 +48,7 @@
|
||||
"Copyright (C) 2009, 2012 Pierre Ossman for Cendio AB\n" \
|
||||
"Copyright (C) 2009-2011 Nokia Corporation and/or its subsidiary(-ies)\n" \
|
||||
"Copyright (C) 1999-2006 MIYASAKA Masaru\n" \
|
||||
"Copyright (C) 1991-2017 Thomas G. Lane, Guido Vollbeding"
|
||||
"Copyright (C) 1991-2020 Thomas G. Lane, Guido Vollbeding"
|
||||
|
||||
#define JCOPYRIGHT_SHORT \
|
||||
"Copyright (C) 1991-2020 The libjpeg-turbo Project and many others"
|
||||
"Copyright (C) 1991-2021 The libjpeg-turbo Project and many others"
|
||||
|
||||
+54
@@ -0,0 +1,54 @@
|
||||
/*
|
||||
* jversion.h
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-2020, Thomas G. Lane, Guido Vollbeding.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2010, 2012-2022, D. R. Commander.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*
|
||||
* This file contains software version identification.
|
||||
*/
|
||||
|
||||
|
||||
#if JPEG_LIB_VERSION >= 80
|
||||
|
||||
#define JVERSION "8d 15-Jan-2012"
|
||||
|
||||
#elif JPEG_LIB_VERSION >= 70
|
||||
|
||||
#define JVERSION "7 27-Jun-2009"
|
||||
|
||||
#else
|
||||
|
||||
#define JVERSION "6b 27-Mar-1998"
|
||||
|
||||
#endif
|
||||
|
||||
/*
|
||||
* NOTE: It is our convention to place the authors in the following order:
|
||||
* - libjpeg-turbo authors (2009-) in descending order of the date of their
|
||||
* most recent contribution to the project, then in ascending order of the
|
||||
* date of their first contribution to the project, then in alphabetical
|
||||
* order
|
||||
* - Upstream authors in descending order of the date of the first inclusion of
|
||||
* their code
|
||||
*/
|
||||
|
||||
#define JCOPYRIGHT \
|
||||
"Copyright (C) 2009-2022 D. R. Commander\n" \
|
||||
"Copyright (C) 2015, 2020 Google, Inc.\n" \
|
||||
"Copyright (C) 2019-2020 Arm Limited\n" \
|
||||
"Copyright (C) 2015-2016, 2018 Matthieu Darbois\n" \
|
||||
"Copyright (C) 2011-2016 Siarhei Siamashka\n" \
|
||||
"Copyright (C) 2015 Intel Corporation\n" \
|
||||
"Copyright (C) 2013-2014 Linaro Limited\n" \
|
||||
"Copyright (C) 2013-2014 MIPS Technologies, Inc.\n" \
|
||||
"Copyright (C) 2009, 2012 Pierre Ossman for Cendio AB\n" \
|
||||
"Copyright (C) 2009-2011 Nokia Corporation and/or its subsidiary(-ies)\n" \
|
||||
"Copyright (C) 1999-2006 MIYASAKA Masaru\n" \
|
||||
"Copyright (C) 1991-2020 Thomas G. Lane, Guido Vollbeding"
|
||||
|
||||
#define JCOPYRIGHT_SHORT \
|
||||
"Copyright (C) @COPYRIGHT_YEAR@ The libjpeg-turbo Project and many others"
|
||||
+545
@@ -0,0 +1,545 @@
|
||||
macro(simd_fail message)
|
||||
message(STATUS "libjpeg-turbo(SIMD): ${message}. Performance will suffer.")
|
||||
set(WITH_SIMD 0 PARENT_SCOPE)
|
||||
endmacro()
|
||||
|
||||
macro(boolean_number var)
|
||||
if(${var})
|
||||
set(${var} 1 ${ARGN})
|
||||
else()
|
||||
set(${var} 0 ${ARGN})
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
|
||||
###############################################################################
|
||||
# x86[-64] (NASM)
|
||||
###############################################################################
|
||||
|
||||
if(CPU_TYPE STREQUAL "x86_64" OR CPU_TYPE STREQUAL "i386")
|
||||
|
||||
set(CMAKE_ASM_NASM_FLAGS_DEBUG_INIT "-g")
|
||||
set(CMAKE_ASM_NASM_FLAGS_RELWITHDEBINFO_INIT "-g")
|
||||
|
||||
# Allow the location of the NASM executable to be specified using the ASM_NASM
|
||||
# environment variable. This should happen automatically, but unfortunately
|
||||
# enable_language(ASM_NASM) doesn't parse the ASM_NASM environment variable
|
||||
# until after CMAKE_ASM_NASM_COMPILER has been populated with the results of
|
||||
# searching for NASM or Yasm in the PATH.
|
||||
if(NOT DEFINED CMAKE_ASM_NASM_COMPILER AND DEFINED ENV{ASM_NASM})
|
||||
set(CMAKE_ASM_NASM_COMPILER $ENV{ASM_NASM})
|
||||
endif()
|
||||
|
||||
if(CPU_TYPE STREQUAL "x86_64")
|
||||
if(CYGWIN)
|
||||
set(CMAKE_ASM_NASM_OBJECT_FORMAT win64)
|
||||
endif()
|
||||
if(CMAKE_C_COMPILER_ABI MATCHES "ELF X32")
|
||||
set(CMAKE_ASM_NASM_OBJECT_FORMAT elfx32)
|
||||
endif()
|
||||
elseif(CPU_TYPE STREQUAL "i386")
|
||||
if(BORLAND)
|
||||
set(CMAKE_ASM_NASM_OBJECT_FORMAT obj)
|
||||
elseif(CYGWIN)
|
||||
set(CMAKE_ASM_NASM_OBJECT_FORMAT win32)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
|
||||
include(CheckLanguage)
|
||||
check_language(ASM_NASM)
|
||||
if(NOT CMAKE_ASM_NASM_COMPILER)
|
||||
simd_fail("SIMD extensions disabled: could not find NASM compiler")
|
||||
return()
|
||||
endif()
|
||||
|
||||
enable_language(ASM_NASM)
|
||||
message(STATUS "CMAKE_ASM_NASM_COMPILER = ${CMAKE_ASM_NASM_COMPILER}")
|
||||
|
||||
if(CMAKE_ASM_NASM_OBJECT_FORMAT MATCHES "^macho")
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -DMACHO")
|
||||
elseif(CMAKE_ASM_NASM_OBJECT_FORMAT MATCHES "^elf")
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -DELF")
|
||||
set(CMAKE_ASM_NASM_DEBUG_FORMAT "dwarf2")
|
||||
endif()
|
||||
if(CPU_TYPE STREQUAL "x86_64")
|
||||
if(WIN32 OR CYGWIN)
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -DWIN64")
|
||||
endif()
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -D__x86_64__")
|
||||
elseif(CPU_TYPE STREQUAL "i386")
|
||||
if(BORLAND)
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -DOBJ32")
|
||||
elseif(WIN32 OR CYGWIN)
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -DWIN32")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(NOT CMAKE_ASM_NASM_OBJECT_FORMAT)
|
||||
simd_fail("SIMD extensions disabled: could not determine NASM object format")
|
||||
return()
|
||||
endif()
|
||||
|
||||
get_filename_component(CMAKE_ASM_NASM_COMPILER_TYPE
|
||||
"${CMAKE_ASM_NASM_COMPILER}" NAME_WE)
|
||||
if(CMAKE_ASM_NASM_COMPILER_TYPE MATCHES "yasm")
|
||||
foreach(var CMAKE_ASM_NASM_FLAGS_DEBUG CMAKE_ASM_NASM_FLAGS_RELWITHDEBINFO)
|
||||
if(${var} STREQUAL "-g")
|
||||
if(CMAKE_ASM_NASM_DEBUG_FORMAT)
|
||||
set_property(CACHE ${var} PROPERTY VALUE "-g ${CMAKE_ASM_NASM_DEBUG_FORMAT}")
|
||||
else()
|
||||
set_property(CACHE ${var} PROPERTY VALUE "")
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
if(NOT WIN32 AND (CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED))
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -DPIC")
|
||||
endif()
|
||||
|
||||
string(TOUPPER ${CMAKE_BUILD_TYPE} CMAKE_BUILD_TYPE_UC)
|
||||
set(EFFECTIVE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} ${CMAKE_ASM_NASM_FLAGS_${CMAKE_BUILD_TYPE_UC}}")
|
||||
|
||||
set(CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS} -I\"${CMAKE_CURRENT_SOURCE_DIR}/nasm/\" -I\"${CMAKE_CURRENT_SOURCE_DIR}/${CPU_TYPE}/\"")
|
||||
|
||||
set(GREP grep)
|
||||
if(CMAKE_SYSTEM_NAME STREQUAL "SunOS")
|
||||
set(GREP ggrep)
|
||||
endif()
|
||||
add_custom_target(jsimdcfg COMMAND
|
||||
${CMAKE_C_COMPILER} -E -I${CMAKE_BINARY_DIR} -I${CMAKE_CURRENT_BINARY_DIR}
|
||||
-I${CMAKE_CURRENT_SOURCE_DIR}
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/nasm/jsimdcfg.inc.h |
|
||||
${GREP} -E '^[\;%]|^\ %' | sed 's%_cpp_protection_%%' |
|
||||
sed 's@% define@%define@g' >${CMAKE_CURRENT_SOURCE_DIR}/nasm/jsimdcfg.inc)
|
||||
set_target_properties(jsimdcfg PROPERTIES FOLDER "3rdparty")
|
||||
|
||||
if(CPU_TYPE STREQUAL "x86_64")
|
||||
set(SIMD_SOURCES x86_64/jsimdcpu.asm x86_64/jfdctflt-sse.asm
|
||||
x86_64/jccolor-sse2.asm x86_64/jcgray-sse2.asm x86_64/jchuff-sse2.asm
|
||||
x86_64/jcphuff-sse2.asm x86_64/jcsample-sse2.asm x86_64/jdcolor-sse2.asm
|
||||
x86_64/jdmerge-sse2.asm x86_64/jdsample-sse2.asm x86_64/jfdctfst-sse2.asm
|
||||
x86_64/jfdctint-sse2.asm x86_64/jidctflt-sse2.asm x86_64/jidctfst-sse2.asm
|
||||
x86_64/jidctint-sse2.asm x86_64/jidctred-sse2.asm x86_64/jquantf-sse2.asm
|
||||
x86_64/jquanti-sse2.asm
|
||||
x86_64/jccolor-avx2.asm x86_64/jcgray-avx2.asm x86_64/jcsample-avx2.asm
|
||||
x86_64/jdcolor-avx2.asm x86_64/jdmerge-avx2.asm x86_64/jdsample-avx2.asm
|
||||
x86_64/jfdctint-avx2.asm x86_64/jidctint-avx2.asm x86_64/jquanti-avx2.asm)
|
||||
else()
|
||||
set(SIMD_SOURCES i386/jsimdcpu.asm i386/jfdctflt-3dn.asm
|
||||
i386/jidctflt-3dn.asm i386/jquant-3dn.asm
|
||||
i386/jccolor-mmx.asm i386/jcgray-mmx.asm i386/jcsample-mmx.asm
|
||||
i386/jdcolor-mmx.asm i386/jdmerge-mmx.asm i386/jdsample-mmx.asm
|
||||
i386/jfdctfst-mmx.asm i386/jfdctint-mmx.asm i386/jidctfst-mmx.asm
|
||||
i386/jidctint-mmx.asm i386/jidctred-mmx.asm i386/jquant-mmx.asm
|
||||
i386/jfdctflt-sse.asm i386/jidctflt-sse.asm i386/jquant-sse.asm
|
||||
i386/jccolor-sse2.asm i386/jcgray-sse2.asm i386/jchuff-sse2.asm
|
||||
i386/jcphuff-sse2.asm i386/jcsample-sse2.asm i386/jdcolor-sse2.asm
|
||||
i386/jdmerge-sse2.asm i386/jdsample-sse2.asm i386/jfdctfst-sse2.asm
|
||||
i386/jfdctint-sse2.asm i386/jidctflt-sse2.asm i386/jidctfst-sse2.asm
|
||||
i386/jidctint-sse2.asm i386/jidctred-sse2.asm i386/jquantf-sse2.asm
|
||||
i386/jquanti-sse2.asm
|
||||
i386/jccolor-avx2.asm i386/jcgray-avx2.asm i386/jcsample-avx2.asm
|
||||
i386/jdcolor-avx2.asm i386/jdmerge-avx2.asm i386/jdsample-avx2.asm
|
||||
i386/jfdctint-avx2.asm i386/jidctint-avx2.asm i386/jquanti-avx2.asm)
|
||||
endif()
|
||||
|
||||
if(MSVC_IDE)
|
||||
set(OBJDIR "${CMAKE_CURRENT_BINARY_DIR}/${CMAKE_CFG_INTDIR}")
|
||||
string(REGEX REPLACE " " ";" CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS}")
|
||||
elseif(XCODE)
|
||||
set(OBJDIR "${CMAKE_CURRENT_BINARY_DIR}")
|
||||
string(REGEX REPLACE " " ";" CMAKE_ASM_NASM_FLAGS "${CMAKE_ASM_NASM_FLAGS}")
|
||||
endif()
|
||||
|
||||
file(GLOB INC_FILES nasm/*.inc)
|
||||
|
||||
foreach(file ${SIMD_SOURCES})
|
||||
set(OBJECT_DEPENDS "")
|
||||
if(${file} MATCHES jccolor)
|
||||
string(REGEX REPLACE "jccolor" "jccolext" DEPFILE ${file})
|
||||
set(OBJECT_DEPENDS ${OBJECT_DEPENDS}
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/${DEPFILE})
|
||||
endif()
|
||||
if(${file} MATCHES jcgray)
|
||||
string(REGEX REPLACE "jcgray" "jcgryext" DEPFILE ${file})
|
||||
set(OBJECT_DEPENDS ${OBJECT_DEPENDS}
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/${DEPFILE})
|
||||
endif()
|
||||
if(${file} MATCHES jdcolor)
|
||||
string(REGEX REPLACE "jdcolor" "jdcolext" DEPFILE ${file})
|
||||
set(OBJECT_DEPENDS ${OBJECT_DEPENDS}
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/${DEPFILE})
|
||||
endif()
|
||||
if(${file} MATCHES jdmerge)
|
||||
string(REGEX REPLACE "jdmerge" "jdmrgext" DEPFILE ${file})
|
||||
set(OBJECT_DEPENDS ${OBJECT_DEPENDS}
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/${DEPFILE})
|
||||
endif()
|
||||
set(OBJECT_DEPENDS ${OBJECT_DEPENDS} ${INC_FILES})
|
||||
if(MSVC_IDE OR XCODE)
|
||||
# The CMake Visual Studio generators do not work properly with the ASM_NASM
|
||||
# language, so we have to go rogue here and use a custom command like we
|
||||
# did in prior versions of libjpeg-turbo. (This is why we can't have nice
|
||||
# things.)
|
||||
string(REGEX REPLACE "${CPU_TYPE}/" "" filename ${file})
|
||||
set(SIMD_OBJ ${OBJDIR}/${filename}${CMAKE_C_OUTPUT_EXTENSION})
|
||||
add_custom_command(OUTPUT ${SIMD_OBJ} DEPENDS ${file} ${OBJECT_DEPENDS}
|
||||
COMMAND ${CMAKE_ASM_NASM_COMPILER} -f${CMAKE_ASM_NASM_OBJECT_FORMAT}
|
||||
${CMAKE_ASM_NASM_FLAGS} ${CMAKE_CURRENT_SOURCE_DIR}/${file}
|
||||
-o${SIMD_OBJ})
|
||||
set(SIMD_OBJS ${SIMD_OBJS} ${SIMD_OBJ})
|
||||
else()
|
||||
set_source_files_properties(${file} PROPERTIES OBJECT_DEPENDS
|
||||
"${OBJECT_DEPENDS}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if(MSVC_IDE OR XCODE)
|
||||
set(SIMD_OBJS ${SIMD_OBJS} PARENT_SCOPE)
|
||||
add_library(jsimd OBJECT ${CPU_TYPE}/jsimd.c)
|
||||
add_custom_target(jsimd-objs DEPENDS ${SIMD_OBJS})
|
||||
add_dependencies(jsimd jsimd-objs)
|
||||
set_target_properties(jsimd PROPERTIES FOLDER "3rdparty")
|
||||
set_target_properties(jsimd-objs PROPERTIES FOLDER "3rdparty")
|
||||
else()
|
||||
add_library(jsimd OBJECT ${SIMD_SOURCES} ${CPU_TYPE}/jsimd.c)
|
||||
endif()
|
||||
if(NOT WIN32 AND (CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED))
|
||||
set_target_properties(jsimd PROPERTIES POSITION_INDEPENDENT_CODE 1)
|
||||
endif()
|
||||
|
||||
|
||||
###############################################################################
|
||||
# Arm (Intrinsics or GAS)
|
||||
###############################################################################
|
||||
|
||||
elseif(CPU_TYPE STREQUAL "arm64" OR CPU_TYPE STREQUAL "arm")
|
||||
|
||||
# If Neon instructions are not explicitly enabled at compile time (e.g. using
|
||||
# -mfpu=neon) with an AArch32 Linux or Android build, then the AArch32 SIMD
|
||||
# dispatcher will parse /proc/cpuinfo to determine whether the Neon SIMD
|
||||
# extensions can be enabled at run time. In order to support all AArch32 CPUs
|
||||
# using the same code base, i.e. to support run-time FPU and Neon
|
||||
# auto-detection, it is necessary to compile the scalar C source code using
|
||||
# -mfloat-abi=soft (which is usually the default) but compile the intrinsics
|
||||
# implementation of the Neon SIMD extensions using -mfloat-abi=softfp. The
|
||||
# following test determines whether -mfloat-abi=softfp should be explicitly
|
||||
# added to the compile flags for the intrinsics implementation of the Neon SIMD
|
||||
# extensions.
|
||||
|
||||
if(BITS EQUAL 32)
|
||||
check_c_source_compiles("
|
||||
#if defined(__ARM_NEON__) || (!defined(__linux__) && !defined(ANDROID) && !defined(__ANDROID__))
|
||||
#error \"Neon run-time auto-detection will not be used\"
|
||||
#endif
|
||||
#if __ARM_PCS_VFP == 1
|
||||
#error \"float ABI = hard\"
|
||||
#endif
|
||||
#if __SOFTFP__ != 1
|
||||
#error \"float ABI = softfp\"
|
||||
#endif
|
||||
int main(void) { return 0; }" NEED_SOFTFP_FOR_INTRINSICS)
|
||||
if(NEED_SOFTFP_FOR_INTRINSICS)
|
||||
set(SOFTFP_FLAG -mfloat-abi=softfp)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(BITS EQUAL 32)
|
||||
set(CMAKE_REQUIRED_FLAGS "-mfpu=neon ${SOFTFP_FLAG}")
|
||||
check_c_source_compiles("
|
||||
#include <arm_neon.h>
|
||||
int main(int argc, char **argv) {
|
||||
uint16x8_t input = vdupq_n_u16((uint16_t)argc);
|
||||
uint8x8_t output = vmovn_u16(input);
|
||||
return (int)output[0];
|
||||
}" HAVE_NEON)
|
||||
if(NOT HAVE_NEON)
|
||||
simd_fail("SIMD extensions not available for this architecture")
|
||||
return()
|
||||
endif()
|
||||
endif()
|
||||
check_c_source_compiles("
|
||||
#include <arm_neon.h>
|
||||
int main(int argc, char **argv) {
|
||||
int16_t input[] = {
|
||||
(int16_t)argc, (int16_t)argc, (int16_t)argc, (int16_t)argc,
|
||||
(int16_t)argc, (int16_t)argc, (int16_t)argc, (int16_t)argc,
|
||||
(int16_t)argc, (int16_t)argc, (int16_t)argc, (int16_t)argc
|
||||
};
|
||||
int16x4x3_t output = vld1_s16_x3(input);
|
||||
vst3_s16(input, output);
|
||||
return (int)input[0];
|
||||
}" HAVE_VLD1_S16_X3)
|
||||
check_c_source_compiles("
|
||||
#include <arm_neon.h>
|
||||
int main(int argc, char **argv) {
|
||||
uint16_t input[] = {
|
||||
(uint16_t)argc, (uint16_t)argc, (uint16_t)argc, (uint16_t)argc,
|
||||
(uint16_t)argc, (uint16_t)argc, (uint16_t)argc, (uint16_t)argc
|
||||
};
|
||||
uint16x4x2_t output = vld1_u16_x2(input);
|
||||
vst2_u16(input, output);
|
||||
return (int)input[0];
|
||||
}" HAVE_VLD1_U16_X2)
|
||||
check_c_source_compiles("
|
||||
#include <arm_neon.h>
|
||||
int main(int argc, char **argv) {
|
||||
uint8_t input[] = {
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc,
|
||||
(uint8_t)argc, (uint8_t)argc, (uint8_t)argc, (uint8_t)argc
|
||||
};
|
||||
uint8x16x4_t output = vld1q_u8_x4(input);
|
||||
vst4q_u8(input, output);
|
||||
return (int)input[0];
|
||||
}" HAVE_VLD1Q_U8_X4)
|
||||
if(BITS EQUAL 32)
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
endif()
|
||||
configure_file(arm/neon-compat.h.in arm/neon-compat.h @ONLY)
|
||||
include_directories(${CMAKE_CURRENT_BINARY_DIR}/arm)
|
||||
|
||||
# GCC 11 and earlier and some older versions of Clang do not have a full or
|
||||
# optimal set of Neon intrinsics, so for performance reasons, when using those
|
||||
# compilers, we default to using the older GAS implementation of the Neon SIMD
|
||||
# extensions for certain algorithms. The presence or absence of the three
|
||||
# intrinsics we tested above is a reasonable proxy for this, except with GCC 10
|
||||
# and 11.
|
||||
if((HAVE_VLD1_S16_X3 AND HAVE_VLD1_U16_X2 AND HAVE_VLD1Q_U8_X4 AND
|
||||
(NOT CMAKE_COMPILER_IS_GNUCC OR
|
||||
CMAKE_C_COMPILER_VERSION VERSION_EQUAL 12.0.0 OR
|
||||
CMAKE_C_COMPILER_VERSION VERSION_GREATER 12.0.0)))
|
||||
set(DEFAULT_NEON_INTRINSICS 1)
|
||||
else()
|
||||
set(DEFAULT_NEON_INTRINSICS 0)
|
||||
endif()
|
||||
option(NEON_INTRINSICS
|
||||
"Because GCC (as of this writing) and some older versions of Clang do not have a full or optimal set of Neon intrinsics, for performance reasons, the default when building libjpeg-turbo with those compilers is to continue using the older GAS implementation of the Neon SIMD extensions for certain algorithms. Setting this option forces the full Neon intrinsics implementation to be used with all compilers. Unsetting this option forces the hybrid GAS/intrinsics implementation to be used with all compilers."
|
||||
${DEFAULT_NEON_INTRINSICS})
|
||||
if(NOT NEON_INTRINSICS)
|
||||
enable_language(ASM)
|
||||
|
||||
set(CMAKE_ASM_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_ASM_FLAGS}")
|
||||
|
||||
# Test whether gas-preprocessor.pl would be needed to build the GAS
|
||||
# implementation of the Neon SIMD extensions. If so, then automatically
|
||||
# enable the full Neon intrinsics implementation.
|
||||
if(CPU_TYPE STREQUAL "arm")
|
||||
file(WRITE ${CMAKE_CURRENT_BINARY_DIR}/gastest.S "
|
||||
.text
|
||||
.fpu neon
|
||||
.arch armv7a
|
||||
.object_arch armv4
|
||||
.arm
|
||||
pld [r0]
|
||||
vmovn.u16 d0, q0")
|
||||
else()
|
||||
file(WRITE ${CMAKE_CURRENT_BINARY_DIR}/gastest.S "
|
||||
.text
|
||||
MYVAR .req x0
|
||||
movi v0.16b, #100
|
||||
mov MYVAR, #100
|
||||
.unreq MYVAR")
|
||||
endif()
|
||||
separate_arguments(CMAKE_ASM_FLAGS_SEP UNIX_COMMAND "${CMAKE_ASM_FLAGS}")
|
||||
execute_process(COMMAND ${CMAKE_ASM_COMPILER} ${CMAKE_ASM_FLAGS_SEP}
|
||||
-x assembler-with-cpp -c ${CMAKE_CURRENT_BINARY_DIR}/gastest.S
|
||||
RESULT_VARIABLE RESULT OUTPUT_VARIABLE OUTPUT ERROR_VARIABLE ERROR)
|
||||
if(NOT RESULT EQUAL 0)
|
||||
message(STATUS "libjpeg-turbo(SIMD): GAS appears to be broken. Using the full Neon SIMD intrinsics implementation.")
|
||||
set(NEON_INTRINSICS 1 CACHE INTERNAL "" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
boolean_number(NEON_INTRINSICS PARENT_SCOPE)
|
||||
if(NEON_INTRINSICS)
|
||||
add_definitions(-DNEON_INTRINSICS)
|
||||
message(STATUS "Use full Neon SIMD intrinsics implementation (NEON_INTRINSICS = ${NEON_INTRINSICS})")
|
||||
else()
|
||||
message(STATUS "Use partial Neon SIMD intrinsics implementation (NEON_INTRINSICS = ${NEON_INTRINSICS})")
|
||||
endif()
|
||||
|
||||
set(SIMD_SOURCES arm/jcgray-neon.c arm/jcphuff-neon.c arm/jcsample-neon.c
|
||||
arm/jdmerge-neon.c arm/jdsample-neon.c arm/jfdctfst-neon.c
|
||||
arm/jidctred-neon.c arm/jquanti-neon.c)
|
||||
if(NEON_INTRINSICS)
|
||||
set(SIMD_SOURCES ${SIMD_SOURCES} arm/jccolor-neon.c arm/jidctint-neon.c)
|
||||
endif()
|
||||
if(NEON_INTRINSICS OR BITS EQUAL 64)
|
||||
set(SIMD_SOURCES ${SIMD_SOURCES} arm/jidctfst-neon.c)
|
||||
endif()
|
||||
if(NEON_INTRINSICS OR BITS EQUAL 32)
|
||||
set(SIMD_SOURCES ${SIMD_SOURCES} arm/aarch${BITS}/jchuff-neon.c
|
||||
arm/jdcolor-neon.c arm/jfdctint-neon.c)
|
||||
endif()
|
||||
if(BITS EQUAL 32)
|
||||
set_source_files_properties(${SIMD_SOURCES} COMPILE_FLAGS "-mfpu=neon ${SOFTFP_FLAG}")
|
||||
endif()
|
||||
if(NOT NEON_INTRINSICS)
|
||||
string(TOUPPER ${CMAKE_BUILD_TYPE} CMAKE_BUILD_TYPE_UC)
|
||||
set(EFFECTIVE_ASM_FLAGS "${CMAKE_ASM_FLAGS} ${CMAKE_ASM_FLAGS_${CMAKE_BUILD_TYPE_UC}}")
|
||||
message(STATUS "CMAKE_ASM_FLAGS = ${EFFECTIVE_ASM_FLAGS}")
|
||||
|
||||
set(SIMD_SOURCES ${SIMD_SOURCES} arm/aarch${BITS}/jsimd_neon.S)
|
||||
endif()
|
||||
|
||||
add_library(jsimd OBJECT ${SIMD_SOURCES} arm/aarch${BITS}/jsimd.c)
|
||||
|
||||
if(CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED)
|
||||
set_target_properties(jsimd PROPERTIES POSITION_INDEPENDENT_CODE 1)
|
||||
endif()
|
||||
|
||||
|
||||
###############################################################################
|
||||
# MIPS (GAS)
|
||||
###############################################################################
|
||||
|
||||
elseif(CPU_TYPE STREQUAL "mips" OR CPU_TYPE STREQUAL "mipsel")
|
||||
|
||||
enable_language(ASM)
|
||||
|
||||
string(TOUPPER ${CMAKE_BUILD_TYPE} CMAKE_BUILD_TYPE_UC)
|
||||
set(EFFECTIVE_ASM_FLAGS "${CMAKE_ASM_FLAGS} ${CMAKE_ASM_FLAGS_${CMAKE_BUILD_TYPE_UC}}")
|
||||
message(STATUS "CMAKE_ASM_FLAGS = ${EFFECTIVE_ASM_FLAGS}")
|
||||
|
||||
set(CMAKE_REQUIRED_FLAGS -mdspr2)
|
||||
|
||||
check_c_source_compiles("
|
||||
#if !(defined(__mips__) && __mips_isa_rev >= 2)
|
||||
#error MIPS DSPr2 is currently only available on MIPS32r2 platforms.
|
||||
#endif
|
||||
int main(void) {
|
||||
int c = 0, a = 0, b = 0;
|
||||
__asm__ __volatile__ (
|
||||
\"precr.qb.ph %[c], %[a], %[b]\"
|
||||
: [c] \"=r\" (c)
|
||||
: [a] \"r\" (a), [b] \"r\" (b)
|
||||
);
|
||||
return c;
|
||||
}" HAVE_DSPR2)
|
||||
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
|
||||
if(NOT HAVE_DSPR2)
|
||||
simd_fail("SIMD extensions not available for this CPU")
|
||||
return()
|
||||
endif()
|
||||
|
||||
add_library(jsimd OBJECT mips/jsimd_dspr2.S mips/jsimd.c)
|
||||
|
||||
if(CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED)
|
||||
set_target_properties(jsimd PROPERTIES POSITION_INDEPENDENT_CODE 1)
|
||||
endif()
|
||||
|
||||
###############################################################################
|
||||
# MIPS64 (Intrinsics)
|
||||
###############################################################################
|
||||
|
||||
elseif(CPU_TYPE STREQUAL "loongson" OR CPU_TYPE MATCHES "^mips64")
|
||||
|
||||
set(CMAKE_REQUIRED_FLAGS -Wa,-mloongson-mmi,-mloongson-ext)
|
||||
|
||||
check_c_source_compiles("
|
||||
int main(void) {
|
||||
int c = 0, a = 0, b = 0;
|
||||
asm (
|
||||
\"paddb %0, %1, %2\"
|
||||
: \"=f\" (c)
|
||||
: \"f\" (a), \"f\" (b)
|
||||
);
|
||||
return c;
|
||||
}" HAVE_MMI)
|
||||
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
|
||||
if(NOT HAVE_MMI)
|
||||
simd_fail("SIMD extensions not available for this CPU")
|
||||
return()
|
||||
endif()
|
||||
|
||||
set(SIMD_SOURCES mips64/jccolor-mmi.c mips64/jcgray-mmi.c mips64/jcsample-mmi.c
|
||||
mips64/jdcolor-mmi.c mips64/jdmerge-mmi.c mips64/jdsample-mmi.c
|
||||
mips64/jfdctfst-mmi.c mips64/jfdctint-mmi.c mips64/jidctfst-mmi.c
|
||||
mips64/jidctint-mmi.c mips64/jquanti-mmi.c)
|
||||
|
||||
if(CMAKE_COMPILER_IS_GNUCC)
|
||||
foreach(file ${SIMD_SOURCES})
|
||||
set_property(SOURCE ${file} APPEND_STRING PROPERTY COMPILE_FLAGS
|
||||
" -fno-strict-aliasing")
|
||||
endforeach()
|
||||
endif()
|
||||
foreach(file ${SIMD_SOURCES})
|
||||
set_property(SOURCE ${file} APPEND_STRING PROPERTY COMPILE_FLAGS
|
||||
" -Wa,-mloongson-mmi,-mloongson-ext")
|
||||
endforeach()
|
||||
|
||||
add_library(jsimd OBJECT ${SIMD_SOURCES} mips64/jsimd.c)
|
||||
|
||||
if(CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED)
|
||||
set_target_properties(jsimd PROPERTIES POSITION_INDEPENDENT_CODE 1)
|
||||
endif()
|
||||
|
||||
###############################################################################
|
||||
# PowerPC (Intrinsics)
|
||||
###############################################################################
|
||||
|
||||
elseif(CPU_TYPE STREQUAL "powerpc")
|
||||
|
||||
set(CMAKE_REQUIRED_FLAGS -maltivec)
|
||||
|
||||
check_c_source_compiles("
|
||||
#include <altivec.h>
|
||||
int main(void) {
|
||||
__vector int vi = { 0, 0, 0, 0 };
|
||||
int i[4];
|
||||
vec_st(vi, 0, i);
|
||||
return i[0];
|
||||
}" HAVE_ALTIVEC)
|
||||
|
||||
unset(CMAKE_REQUIRED_FLAGS)
|
||||
|
||||
if(NOT HAVE_ALTIVEC)
|
||||
simd_fail("SIMD extensions not available for this CPU (PowerPC SPE)")
|
||||
return()
|
||||
endif()
|
||||
|
||||
set(SIMD_SOURCES powerpc/jccolor-altivec.c powerpc/jcgray-altivec.c
|
||||
powerpc/jcsample-altivec.c powerpc/jdcolor-altivec.c
|
||||
powerpc/jdmerge-altivec.c powerpc/jdsample-altivec.c
|
||||
powerpc/jfdctfst-altivec.c powerpc/jfdctint-altivec.c
|
||||
powerpc/jidctfst-altivec.c powerpc/jidctint-altivec.c
|
||||
powerpc/jquanti-altivec.c)
|
||||
|
||||
set_source_files_properties(${SIMD_SOURCES} PROPERTIES
|
||||
COMPILE_FLAGS -maltivec)
|
||||
|
||||
add_library(jsimd OBJECT ${SIMD_SOURCES} powerpc/jsimd.c)
|
||||
|
||||
if(CMAKE_POSITION_INDEPENDENT_CODE OR ENABLE_SHARED)
|
||||
set_target_properties(jsimd PROPERTIES POSITION_INDEPENDENT_CODE 1)
|
||||
endif()
|
||||
|
||||
|
||||
###############################################################################
|
||||
# None
|
||||
###############################################################################
|
||||
|
||||
else()
|
||||
|
||||
simd_fail("SIMD extensions not available for this CPU (${CMAKE_SYSTEM_PROCESSOR})")
|
||||
|
||||
endif() # CPU_TYPE
|
||||
@@ -0,0 +1,148 @@
|
||||
/*
|
||||
* jccolext-neon.c - colorspace conversion (32-bit Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
/* This file is included by jccolor-neon.c */
|
||||
|
||||
|
||||
/* RGB -> YCbCr conversion is defined by the following equations:
|
||||
* Y = 0.29900 * R + 0.58700 * G + 0.11400 * B
|
||||
* Cb = -0.16874 * R - 0.33126 * G + 0.50000 * B + 128
|
||||
* Cr = 0.50000 * R - 0.41869 * G - 0.08131 * B + 128
|
||||
*
|
||||
* Avoid floating point arithmetic by using shifted integer constants:
|
||||
* 0.29899597 = 19595 * 2^-16
|
||||
* 0.58700561 = 38470 * 2^-16
|
||||
* 0.11399841 = 7471 * 2^-16
|
||||
* 0.16874695 = 11059 * 2^-16
|
||||
* 0.33125305 = 21709 * 2^-16
|
||||
* 0.50000000 = 32768 * 2^-16
|
||||
* 0.41868592 = 27439 * 2^-16
|
||||
* 0.08131409 = 5329 * 2^-16
|
||||
* These constants are defined in jccolor-neon.c
|
||||
*
|
||||
* We add the fixed-point equivalent of 0.5 to Cb and Cr, which effectively
|
||||
* rounds up or down the result via integer truncation.
|
||||
*/
|
||||
|
||||
void jsimd_rgb_ycc_convert_neon(JDIMENSION image_width, JSAMPARRAY input_buf,
|
||||
JSAMPIMAGE output_buf, JDIMENSION output_row,
|
||||
int num_rows)
|
||||
{
|
||||
/* Pointer to RGB(X/A) input data */
|
||||
JSAMPROW inptr;
|
||||
/* Pointers to Y, Cb, and Cr output data */
|
||||
JSAMPROW outptr0, outptr1, outptr2;
|
||||
/* Allocate temporary buffer for final (image_width % 8) pixels in row. */
|
||||
ALIGN(16) uint8_t tmp_buf[8 * RGB_PIXELSIZE];
|
||||
|
||||
/* Set up conversion constants. */
|
||||
#ifdef HAVE_VLD1_U16_X2
|
||||
const uint16x4x2_t consts = vld1_u16_x2(jsimd_rgb_ycc_neon_consts);
|
||||
#else
|
||||
/* GCC does not currently support the intrinsic vld1_<type>_x2(). */
|
||||
const uint16x4_t consts1 = vld1_u16(jsimd_rgb_ycc_neon_consts);
|
||||
const uint16x4_t consts2 = vld1_u16(jsimd_rgb_ycc_neon_consts + 4);
|
||||
const uint16x4x2_t consts = { { consts1, consts2 } };
|
||||
#endif
|
||||
const uint32x4_t scaled_128_5 = vdupq_n_u32((128 << 16) + 32767);
|
||||
|
||||
while (--num_rows >= 0) {
|
||||
inptr = *input_buf++;
|
||||
outptr0 = output_buf[0][output_row];
|
||||
outptr1 = output_buf[1][output_row];
|
||||
outptr2 = output_buf[2][output_row];
|
||||
output_row++;
|
||||
|
||||
int cols_remaining = image_width;
|
||||
for (; cols_remaining > 0; cols_remaining -= 8) {
|
||||
|
||||
/* To prevent buffer overread by the vector load instructions, the last
|
||||
* (image_width % 8) columns of data are first memcopied to a temporary
|
||||
* buffer large enough to accommodate the vector load.
|
||||
*/
|
||||
if (cols_remaining < 8) {
|
||||
memcpy(tmp_buf, inptr, cols_remaining * RGB_PIXELSIZE);
|
||||
inptr = tmp_buf;
|
||||
}
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x8x4_t input_pixels = vld4_u8(inptr);
|
||||
#else
|
||||
uint8x8x3_t input_pixels = vld3_u8(inptr);
|
||||
#endif
|
||||
uint16x8_t r = vmovl_u8(input_pixels.val[RGB_RED]);
|
||||
uint16x8_t g = vmovl_u8(input_pixels.val[RGB_GREEN]);
|
||||
uint16x8_t b = vmovl_u8(input_pixels.val[RGB_BLUE]);
|
||||
|
||||
/* Compute Y = 0.29900 * R + 0.58700 * G + 0.11400 * B */
|
||||
uint32x4_t y_low = vmull_lane_u16(vget_low_u16(r), consts.val[0], 0);
|
||||
y_low = vmlal_lane_u16(y_low, vget_low_u16(g), consts.val[0], 1);
|
||||
y_low = vmlal_lane_u16(y_low, vget_low_u16(b), consts.val[0], 2);
|
||||
uint32x4_t y_high = vmull_lane_u16(vget_high_u16(r), consts.val[0], 0);
|
||||
y_high = vmlal_lane_u16(y_high, vget_high_u16(g), consts.val[0], 1);
|
||||
y_high = vmlal_lane_u16(y_high, vget_high_u16(b), consts.val[0], 2);
|
||||
|
||||
/* Compute Cb = -0.16874 * R - 0.33126 * G + 0.50000 * B + 128 */
|
||||
uint32x4_t cb_low = scaled_128_5;
|
||||
cb_low = vmlsl_lane_u16(cb_low, vget_low_u16(r), consts.val[0], 3);
|
||||
cb_low = vmlsl_lane_u16(cb_low, vget_low_u16(g), consts.val[1], 0);
|
||||
cb_low = vmlal_lane_u16(cb_low, vget_low_u16(b), consts.val[1], 1);
|
||||
uint32x4_t cb_high = scaled_128_5;
|
||||
cb_high = vmlsl_lane_u16(cb_high, vget_high_u16(r), consts.val[0], 3);
|
||||
cb_high = vmlsl_lane_u16(cb_high, vget_high_u16(g), consts.val[1], 0);
|
||||
cb_high = vmlal_lane_u16(cb_high, vget_high_u16(b), consts.val[1], 1);
|
||||
|
||||
/* Compute Cr = 0.50000 * R - 0.41869 * G - 0.08131 * B + 128 */
|
||||
uint32x4_t cr_low = scaled_128_5;
|
||||
cr_low = vmlal_lane_u16(cr_low, vget_low_u16(r), consts.val[1], 1);
|
||||
cr_low = vmlsl_lane_u16(cr_low, vget_low_u16(g), consts.val[1], 2);
|
||||
cr_low = vmlsl_lane_u16(cr_low, vget_low_u16(b), consts.val[1], 3);
|
||||
uint32x4_t cr_high = scaled_128_5;
|
||||
cr_high = vmlal_lane_u16(cr_high, vget_high_u16(r), consts.val[1], 1);
|
||||
cr_high = vmlsl_lane_u16(cr_high, vget_high_u16(g), consts.val[1], 2);
|
||||
cr_high = vmlsl_lane_u16(cr_high, vget_high_u16(b), consts.val[1], 3);
|
||||
|
||||
/* Descale Y values (rounding right shift) and narrow to 16-bit. */
|
||||
uint16x8_t y_u16 = vcombine_u16(vrshrn_n_u32(y_low, 16),
|
||||
vrshrn_n_u32(y_high, 16));
|
||||
/* Descale Cb values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cb_u16 = vcombine_u16(vshrn_n_u32(cb_low, 16),
|
||||
vshrn_n_u32(cb_high, 16));
|
||||
/* Descale Cr values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cr_u16 = vcombine_u16(vshrn_n_u32(cr_low, 16),
|
||||
vshrn_n_u32(cr_high, 16));
|
||||
/* Narrow Y, Cb, and Cr values to 8-bit and store to memory. Buffer
|
||||
* overwrite is permitted up to the next multiple of ALIGN_SIZE bytes.
|
||||
*/
|
||||
vst1_u8(outptr0, vmovn_u16(y_u16));
|
||||
vst1_u8(outptr1, vmovn_u16(cb_u16));
|
||||
vst1_u8(outptr2, vmovn_u16(cr_u16));
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr += (8 * RGB_PIXELSIZE);
|
||||
outptr0 += 8;
|
||||
outptr1 += 8;
|
||||
outptr2 += 8;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
/*
|
||||
* jchuff-neon.c - Huffman entropy encoding (32-bit Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*
|
||||
* NOTE: All referenced figures are from
|
||||
* Recommendation ITU-T T.81 (1992) | ISO/IEC 10918-1:1994.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../../jinclude.h"
|
||||
#include "../../../jpeglib.h"
|
||||
#include "../../../jsimd.h"
|
||||
#include "../../../jdct.h"
|
||||
#include "../../../jsimddct.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../jchuff.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <limits.h>
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
JOCTET *jsimd_huff_encode_one_block_neon(void *state, JOCTET *buffer,
|
||||
JCOEFPTR block, int last_dc_val,
|
||||
c_derived_tbl *dctbl,
|
||||
c_derived_tbl *actbl)
|
||||
{
|
||||
uint8_t block_nbits[DCTSIZE2];
|
||||
uint16_t block_diff[DCTSIZE2];
|
||||
|
||||
/* Load rows of coefficients from DCT block in zig-zag order. */
|
||||
|
||||
/* Compute DC coefficient difference value. (F.1.1.5.1) */
|
||||
int16x8_t row0 = vdupq_n_s16(block[0] - last_dc_val);
|
||||
row0 = vld1q_lane_s16(block + 1, row0, 1);
|
||||
row0 = vld1q_lane_s16(block + 8, row0, 2);
|
||||
row0 = vld1q_lane_s16(block + 16, row0, 3);
|
||||
row0 = vld1q_lane_s16(block + 9, row0, 4);
|
||||
row0 = vld1q_lane_s16(block + 2, row0, 5);
|
||||
row0 = vld1q_lane_s16(block + 3, row0, 6);
|
||||
row0 = vld1q_lane_s16(block + 10, row0, 7);
|
||||
|
||||
int16x8_t row1 = vld1q_dup_s16(block + 17);
|
||||
row1 = vld1q_lane_s16(block + 24, row1, 1);
|
||||
row1 = vld1q_lane_s16(block + 32, row1, 2);
|
||||
row1 = vld1q_lane_s16(block + 25, row1, 3);
|
||||
row1 = vld1q_lane_s16(block + 18, row1, 4);
|
||||
row1 = vld1q_lane_s16(block + 11, row1, 5);
|
||||
row1 = vld1q_lane_s16(block + 4, row1, 6);
|
||||
row1 = vld1q_lane_s16(block + 5, row1, 7);
|
||||
|
||||
int16x8_t row2 = vld1q_dup_s16(block + 12);
|
||||
row2 = vld1q_lane_s16(block + 19, row2, 1);
|
||||
row2 = vld1q_lane_s16(block + 26, row2, 2);
|
||||
row2 = vld1q_lane_s16(block + 33, row2, 3);
|
||||
row2 = vld1q_lane_s16(block + 40, row2, 4);
|
||||
row2 = vld1q_lane_s16(block + 48, row2, 5);
|
||||
row2 = vld1q_lane_s16(block + 41, row2, 6);
|
||||
row2 = vld1q_lane_s16(block + 34, row2, 7);
|
||||
|
||||
int16x8_t row3 = vld1q_dup_s16(block + 27);
|
||||
row3 = vld1q_lane_s16(block + 20, row3, 1);
|
||||
row3 = vld1q_lane_s16(block + 13, row3, 2);
|
||||
row3 = vld1q_lane_s16(block + 6, row3, 3);
|
||||
row3 = vld1q_lane_s16(block + 7, row3, 4);
|
||||
row3 = vld1q_lane_s16(block + 14, row3, 5);
|
||||
row3 = vld1q_lane_s16(block + 21, row3, 6);
|
||||
row3 = vld1q_lane_s16(block + 28, row3, 7);
|
||||
|
||||
int16x8_t abs_row0 = vabsq_s16(row0);
|
||||
int16x8_t abs_row1 = vabsq_s16(row1);
|
||||
int16x8_t abs_row2 = vabsq_s16(row2);
|
||||
int16x8_t abs_row3 = vabsq_s16(row3);
|
||||
|
||||
int16x8_t row0_lz = vclzq_s16(abs_row0);
|
||||
int16x8_t row1_lz = vclzq_s16(abs_row1);
|
||||
int16x8_t row2_lz = vclzq_s16(abs_row2);
|
||||
int16x8_t row3_lz = vclzq_s16(abs_row3);
|
||||
|
||||
/* Compute number of bits required to represent each coefficient. */
|
||||
uint8x8_t row0_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row0_lz)));
|
||||
uint8x8_t row1_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row1_lz)));
|
||||
uint8x8_t row2_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row2_lz)));
|
||||
uint8x8_t row3_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row3_lz)));
|
||||
|
||||
vst1_u8(block_nbits + 0 * DCTSIZE, row0_nbits);
|
||||
vst1_u8(block_nbits + 1 * DCTSIZE, row1_nbits);
|
||||
vst1_u8(block_nbits + 2 * DCTSIZE, row2_nbits);
|
||||
vst1_u8(block_nbits + 3 * DCTSIZE, row3_nbits);
|
||||
|
||||
uint16x8_t row0_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row0, 15)),
|
||||
vnegq_s16(row0_lz));
|
||||
uint16x8_t row1_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row1, 15)),
|
||||
vnegq_s16(row1_lz));
|
||||
uint16x8_t row2_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row2, 15)),
|
||||
vnegq_s16(row2_lz));
|
||||
uint16x8_t row3_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row3, 15)),
|
||||
vnegq_s16(row3_lz));
|
||||
|
||||
uint16x8_t row0_diff = veorq_u16(vreinterpretq_u16_s16(abs_row0), row0_mask);
|
||||
uint16x8_t row1_diff = veorq_u16(vreinterpretq_u16_s16(abs_row1), row1_mask);
|
||||
uint16x8_t row2_diff = veorq_u16(vreinterpretq_u16_s16(abs_row2), row2_mask);
|
||||
uint16x8_t row3_diff = veorq_u16(vreinterpretq_u16_s16(abs_row3), row3_mask);
|
||||
|
||||
/* Store diff values for rows 0, 1, 2, and 3. */
|
||||
vst1q_u16(block_diff + 0 * DCTSIZE, row0_diff);
|
||||
vst1q_u16(block_diff + 1 * DCTSIZE, row1_diff);
|
||||
vst1q_u16(block_diff + 2 * DCTSIZE, row2_diff);
|
||||
vst1q_u16(block_diff + 3 * DCTSIZE, row3_diff);
|
||||
|
||||
/* Load last four rows of coefficients from DCT block in zig-zag order. */
|
||||
int16x8_t row4 = vld1q_dup_s16(block + 35);
|
||||
row4 = vld1q_lane_s16(block + 42, row4, 1);
|
||||
row4 = vld1q_lane_s16(block + 49, row4, 2);
|
||||
row4 = vld1q_lane_s16(block + 56, row4, 3);
|
||||
row4 = vld1q_lane_s16(block + 57, row4, 4);
|
||||
row4 = vld1q_lane_s16(block + 50, row4, 5);
|
||||
row4 = vld1q_lane_s16(block + 43, row4, 6);
|
||||
row4 = vld1q_lane_s16(block + 36, row4, 7);
|
||||
|
||||
int16x8_t row5 = vld1q_dup_s16(block + 29);
|
||||
row5 = vld1q_lane_s16(block + 22, row5, 1);
|
||||
row5 = vld1q_lane_s16(block + 15, row5, 2);
|
||||
row5 = vld1q_lane_s16(block + 23, row5, 3);
|
||||
row5 = vld1q_lane_s16(block + 30, row5, 4);
|
||||
row5 = vld1q_lane_s16(block + 37, row5, 5);
|
||||
row5 = vld1q_lane_s16(block + 44, row5, 6);
|
||||
row5 = vld1q_lane_s16(block + 51, row5, 7);
|
||||
|
||||
int16x8_t row6 = vld1q_dup_s16(block + 58);
|
||||
row6 = vld1q_lane_s16(block + 59, row6, 1);
|
||||
row6 = vld1q_lane_s16(block + 52, row6, 2);
|
||||
row6 = vld1q_lane_s16(block + 45, row6, 3);
|
||||
row6 = vld1q_lane_s16(block + 38, row6, 4);
|
||||
row6 = vld1q_lane_s16(block + 31, row6, 5);
|
||||
row6 = vld1q_lane_s16(block + 39, row6, 6);
|
||||
row6 = vld1q_lane_s16(block + 46, row6, 7);
|
||||
|
||||
int16x8_t row7 = vld1q_dup_s16(block + 53);
|
||||
row7 = vld1q_lane_s16(block + 60, row7, 1);
|
||||
row7 = vld1q_lane_s16(block + 61, row7, 2);
|
||||
row7 = vld1q_lane_s16(block + 54, row7, 3);
|
||||
row7 = vld1q_lane_s16(block + 47, row7, 4);
|
||||
row7 = vld1q_lane_s16(block + 55, row7, 5);
|
||||
row7 = vld1q_lane_s16(block + 62, row7, 6);
|
||||
row7 = vld1q_lane_s16(block + 63, row7, 7);
|
||||
|
||||
int16x8_t abs_row4 = vabsq_s16(row4);
|
||||
int16x8_t abs_row5 = vabsq_s16(row5);
|
||||
int16x8_t abs_row6 = vabsq_s16(row6);
|
||||
int16x8_t abs_row7 = vabsq_s16(row7);
|
||||
|
||||
int16x8_t row4_lz = vclzq_s16(abs_row4);
|
||||
int16x8_t row5_lz = vclzq_s16(abs_row5);
|
||||
int16x8_t row6_lz = vclzq_s16(abs_row6);
|
||||
int16x8_t row7_lz = vclzq_s16(abs_row7);
|
||||
|
||||
/* Compute number of bits required to represent each coefficient. */
|
||||
uint8x8_t row4_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row4_lz)));
|
||||
uint8x8_t row5_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row5_lz)));
|
||||
uint8x8_t row6_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row6_lz)));
|
||||
uint8x8_t row7_nbits = vsub_u8(vdup_n_u8(16),
|
||||
vmovn_u16(vreinterpretq_u16_s16(row7_lz)));
|
||||
|
||||
vst1_u8(block_nbits + 4 * DCTSIZE, row4_nbits);
|
||||
vst1_u8(block_nbits + 5 * DCTSIZE, row5_nbits);
|
||||
vst1_u8(block_nbits + 6 * DCTSIZE, row6_nbits);
|
||||
vst1_u8(block_nbits + 7 * DCTSIZE, row7_nbits);
|
||||
|
||||
uint16x8_t row4_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row4, 15)),
|
||||
vnegq_s16(row4_lz));
|
||||
uint16x8_t row5_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row5, 15)),
|
||||
vnegq_s16(row5_lz));
|
||||
uint16x8_t row6_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row6, 15)),
|
||||
vnegq_s16(row6_lz));
|
||||
uint16x8_t row7_mask =
|
||||
vshlq_u16(vreinterpretq_u16_s16(vshrq_n_s16(row7, 15)),
|
||||
vnegq_s16(row7_lz));
|
||||
|
||||
uint16x8_t row4_diff = veorq_u16(vreinterpretq_u16_s16(abs_row4), row4_mask);
|
||||
uint16x8_t row5_diff = veorq_u16(vreinterpretq_u16_s16(abs_row5), row5_mask);
|
||||
uint16x8_t row6_diff = veorq_u16(vreinterpretq_u16_s16(abs_row6), row6_mask);
|
||||
uint16x8_t row7_diff = veorq_u16(vreinterpretq_u16_s16(abs_row7), row7_mask);
|
||||
|
||||
/* Store diff values for rows 4, 5, 6, and 7. */
|
||||
vst1q_u16(block_diff + 4 * DCTSIZE, row4_diff);
|
||||
vst1q_u16(block_diff + 5 * DCTSIZE, row5_diff);
|
||||
vst1q_u16(block_diff + 6 * DCTSIZE, row6_diff);
|
||||
vst1q_u16(block_diff + 7 * DCTSIZE, row7_diff);
|
||||
|
||||
/* Construct bitmap to accelerate encoding of AC coefficients. A set bit
|
||||
* means that the corresponding coefficient != 0.
|
||||
*/
|
||||
uint8x8_t row0_nbits_gt0 = vcgt_u8(row0_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row1_nbits_gt0 = vcgt_u8(row1_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row2_nbits_gt0 = vcgt_u8(row2_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row3_nbits_gt0 = vcgt_u8(row3_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row4_nbits_gt0 = vcgt_u8(row4_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row5_nbits_gt0 = vcgt_u8(row5_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row6_nbits_gt0 = vcgt_u8(row6_nbits, vdup_n_u8(0));
|
||||
uint8x8_t row7_nbits_gt0 = vcgt_u8(row7_nbits, vdup_n_u8(0));
|
||||
|
||||
/* { 0x80, 0x40, 0x20, 0x10, 0x08, 0x04, 0x02, 0x01 } */
|
||||
const uint8x8_t bitmap_mask =
|
||||
vreinterpret_u8_u64(vmov_n_u64(0x0102040810204080));
|
||||
|
||||
row0_nbits_gt0 = vand_u8(row0_nbits_gt0, bitmap_mask);
|
||||
row1_nbits_gt0 = vand_u8(row1_nbits_gt0, bitmap_mask);
|
||||
row2_nbits_gt0 = vand_u8(row2_nbits_gt0, bitmap_mask);
|
||||
row3_nbits_gt0 = vand_u8(row3_nbits_gt0, bitmap_mask);
|
||||
row4_nbits_gt0 = vand_u8(row4_nbits_gt0, bitmap_mask);
|
||||
row5_nbits_gt0 = vand_u8(row5_nbits_gt0, bitmap_mask);
|
||||
row6_nbits_gt0 = vand_u8(row6_nbits_gt0, bitmap_mask);
|
||||
row7_nbits_gt0 = vand_u8(row7_nbits_gt0, bitmap_mask);
|
||||
|
||||
uint8x8_t bitmap_rows_10 = vpadd_u8(row1_nbits_gt0, row0_nbits_gt0);
|
||||
uint8x8_t bitmap_rows_32 = vpadd_u8(row3_nbits_gt0, row2_nbits_gt0);
|
||||
uint8x8_t bitmap_rows_54 = vpadd_u8(row5_nbits_gt0, row4_nbits_gt0);
|
||||
uint8x8_t bitmap_rows_76 = vpadd_u8(row7_nbits_gt0, row6_nbits_gt0);
|
||||
uint8x8_t bitmap_rows_3210 = vpadd_u8(bitmap_rows_32, bitmap_rows_10);
|
||||
uint8x8_t bitmap_rows_7654 = vpadd_u8(bitmap_rows_76, bitmap_rows_54);
|
||||
uint8x8_t bitmap = vpadd_u8(bitmap_rows_7654, bitmap_rows_3210);
|
||||
|
||||
/* Shift left to remove DC bit. */
|
||||
bitmap = vreinterpret_u8_u64(vshl_n_u64(vreinterpret_u64_u8(bitmap), 1));
|
||||
/* Move bitmap to 32-bit scalar registers. */
|
||||
uint32_t bitmap_1_32 = vget_lane_u32(vreinterpret_u32_u8(bitmap), 1);
|
||||
uint32_t bitmap_33_63 = vget_lane_u32(vreinterpret_u32_u8(bitmap), 0);
|
||||
|
||||
/* Set up state and bit buffer for output bitstream. */
|
||||
working_state *state_ptr = (working_state *)state;
|
||||
int free_bits = state_ptr->cur.free_bits;
|
||||
size_t put_buffer = state_ptr->cur.put_buffer;
|
||||
|
||||
/* Encode DC coefficient. */
|
||||
|
||||
unsigned int nbits = block_nbits[0];
|
||||
/* Emit Huffman-coded symbol and additional diff bits. */
|
||||
unsigned int diff = block_diff[0];
|
||||
PUT_CODE(dctbl->ehufco[nbits], dctbl->ehufsi[nbits], diff)
|
||||
|
||||
/* Encode AC coefficients. */
|
||||
|
||||
unsigned int r = 0; /* r = run length of zeros */
|
||||
unsigned int i = 1; /* i = number of coefficients encoded */
|
||||
/* Code and size information for a run length of 16 zero coefficients */
|
||||
const unsigned int code_0xf0 = actbl->ehufco[0xf0];
|
||||
const unsigned int size_0xf0 = actbl->ehufsi[0xf0];
|
||||
|
||||
while (bitmap_1_32 != 0) {
|
||||
r = BUILTIN_CLZ(bitmap_1_32);
|
||||
i += r;
|
||||
bitmap_1_32 <<= r;
|
||||
nbits = block_nbits[i];
|
||||
diff = block_diff[i];
|
||||
while (r > 15) {
|
||||
/* If run length > 15, emit special run-length-16 codes. */
|
||||
PUT_BITS(code_0xf0, size_0xf0)
|
||||
r -= 16;
|
||||
}
|
||||
/* Emit Huffman symbol for run length / number of bits. (F.1.2.2.1) */
|
||||
unsigned int rs = (r << 4) + nbits;
|
||||
PUT_CODE(actbl->ehufco[rs], actbl->ehufsi[rs], diff)
|
||||
i++;
|
||||
bitmap_1_32 <<= 1;
|
||||
}
|
||||
|
||||
r = 33 - i;
|
||||
i = 33;
|
||||
|
||||
while (bitmap_33_63 != 0) {
|
||||
unsigned int leading_zeros = BUILTIN_CLZ(bitmap_33_63);
|
||||
r += leading_zeros;
|
||||
i += leading_zeros;
|
||||
bitmap_33_63 <<= leading_zeros;
|
||||
nbits = block_nbits[i];
|
||||
diff = block_diff[i];
|
||||
while (r > 15) {
|
||||
/* If run length > 15, emit special run-length-16 codes. */
|
||||
PUT_BITS(code_0xf0, size_0xf0)
|
||||
r -= 16;
|
||||
}
|
||||
/* Emit Huffman symbol for run length / number of bits. (F.1.2.2.1) */
|
||||
unsigned int rs = (r << 4) + nbits;
|
||||
PUT_CODE(actbl->ehufco[rs], actbl->ehufsi[rs], diff)
|
||||
r = 0;
|
||||
i++;
|
||||
bitmap_33_63 <<= 1;
|
||||
}
|
||||
|
||||
/* If the last coefficient(s) were zero, emit an end-of-block (EOB) code.
|
||||
* The value of RS for the EOB code is 0.
|
||||
*/
|
||||
if (i != 64) {
|
||||
PUT_BITS(actbl->ehufco[0], actbl->ehufsi[0])
|
||||
}
|
||||
|
||||
state_ptr->cur.put_buffer = put_buffer;
|
||||
state_ptr->cur.free_bits = free_bits;
|
||||
|
||||
return buffer;
|
||||
}
|
||||
+980
@@ -0,0 +1,980 @@
|
||||
/*
|
||||
* jsimd_arm.c
|
||||
*
|
||||
* Copyright 2009 Pierre Ossman <ossman@cendio.se> for Cendio AB
|
||||
* Copyright (C) 2011, Nokia Corporation and/or its subsidiary(-ies).
|
||||
* Copyright (C) 2009-2011, 2013-2014, 2016, 2018, 2022, D. R. Commander.
|
||||
* Copyright (C) 2015-2016, 2018, Matthieu Darbois.
|
||||
* Copyright (C) 2019, Google LLC.
|
||||
* Copyright (C) 2020, Arm Limited.
|
||||
*
|
||||
* Based on the x86 SIMD extension for IJG JPEG library,
|
||||
* Copyright (C) 1999-2006, MIYASAKA Masaru.
|
||||
* For conditions of distribution and use, see copyright notice in jsimdext.inc
|
||||
*
|
||||
* This file contains the interface between the "normal" portions
|
||||
* of the library and the SIMD implementations when running on a
|
||||
* 32-bit Arm architecture.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../../jinclude.h"
|
||||
#include "../../../jpeglib.h"
|
||||
#include "../../../jsimd.h"
|
||||
#include "../../../jdct.h"
|
||||
#include "../../../jsimddct.h"
|
||||
#include "../../jsimd.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <ctype.h>
|
||||
|
||||
static unsigned int simd_support = ~0;
|
||||
static unsigned int simd_huffman = 1;
|
||||
|
||||
#if !defined(__ARM_NEON__) && (defined(__linux__) || defined(ANDROID) || defined(__ANDROID__))
|
||||
|
||||
#define SOMEWHAT_SANE_PROC_CPUINFO_SIZE_LIMIT (1024 * 1024)
|
||||
|
||||
LOCAL(int)
|
||||
check_feature(char *buffer, char *feature)
|
||||
{
|
||||
char *p;
|
||||
|
||||
if (*feature == 0)
|
||||
return 0;
|
||||
if (strncmp(buffer, "Features", 8) != 0)
|
||||
return 0;
|
||||
buffer += 8;
|
||||
while (isspace(*buffer))
|
||||
buffer++;
|
||||
|
||||
/* Check if 'feature' is present in the buffer as a separate word */
|
||||
while ((p = strstr(buffer, feature))) {
|
||||
if (p > buffer && !isspace(*(p - 1))) {
|
||||
buffer++;
|
||||
continue;
|
||||
}
|
||||
p += strlen(feature);
|
||||
if (*p != 0 && !isspace(*p)) {
|
||||
buffer++;
|
||||
continue;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
LOCAL(int)
|
||||
parse_proc_cpuinfo(int bufsize)
|
||||
{
|
||||
char *buffer = (char *)malloc(bufsize);
|
||||
FILE *fd;
|
||||
|
||||
simd_support = 0;
|
||||
|
||||
if (!buffer)
|
||||
return 0;
|
||||
|
||||
fd = fopen("/proc/cpuinfo", "r");
|
||||
if (fd) {
|
||||
while (fgets(buffer, bufsize, fd)) {
|
||||
if (!strchr(buffer, '\n') && !feof(fd)) {
|
||||
/* "impossible" happened - insufficient size of the buffer! */
|
||||
fclose(fd);
|
||||
free(buffer);
|
||||
return 0;
|
||||
}
|
||||
if (check_feature(buffer, "neon"))
|
||||
simd_support |= JSIMD_NEON;
|
||||
}
|
||||
fclose(fd);
|
||||
}
|
||||
free(buffer);
|
||||
return 1;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
/*
|
||||
* Check what SIMD accelerations are supported.
|
||||
*
|
||||
* FIXME: This code is racy under a multi-threaded environment.
|
||||
*/
|
||||
LOCAL(void)
|
||||
init_simd(void)
|
||||
{
|
||||
#ifndef NO_GETENV
|
||||
char env[2] = { 0 };
|
||||
#endif
|
||||
#if !defined(__ARM_NEON__) && (defined(__linux__) || defined(ANDROID) || defined(__ANDROID__))
|
||||
int bufsize = 1024; /* an initial guess for the line buffer size limit */
|
||||
#endif
|
||||
|
||||
if (simd_support != ~0U)
|
||||
return;
|
||||
|
||||
simd_support = 0;
|
||||
|
||||
#if defined(__ARM_NEON__)
|
||||
simd_support |= JSIMD_NEON;
|
||||
#elif defined(__linux__) || defined(ANDROID) || defined(__ANDROID__)
|
||||
/* We still have a chance to use Neon regardless of globally used
|
||||
* -mcpu/-mfpu options passed to gcc by performing runtime detection via
|
||||
* /proc/cpuinfo parsing on linux/android */
|
||||
while (!parse_proc_cpuinfo(bufsize)) {
|
||||
bufsize *= 2;
|
||||
if (bufsize > SOMEWHAT_SANE_PROC_CPUINFO_SIZE_LIMIT)
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifndef NO_GETENV
|
||||
/* Force different settings through environment variables */
|
||||
if (!GETENV_S(env, 2, "JSIMD_FORCENEON") && !strcmp(env, "1"))
|
||||
simd_support = JSIMD_NEON;
|
||||
if (!GETENV_S(env, 2, "JSIMD_FORCENONE") && !strcmp(env, "1"))
|
||||
simd_support = 0;
|
||||
if (!GETENV_S(env, 2, "JSIMD_NOHUFFENC") && !strcmp(env, "1"))
|
||||
simd_huffman = 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_rgb_ycc(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if ((RGB_PIXELSIZE != 3) && (RGB_PIXELSIZE != 4))
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_rgb_gray(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if ((RGB_PIXELSIZE != 3) && (RGB_PIXELSIZE != 4))
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_ycc_rgb(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if ((RGB_PIXELSIZE != 3) && (RGB_PIXELSIZE != 4))
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_ycc_rgb565(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_rgb_ycc_convert(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
JSAMPIMAGE output_buf, JDIMENSION output_row,
|
||||
int num_rows)
|
||||
{
|
||||
void (*neonfct) (JDIMENSION, JSAMPARRAY, JSAMPIMAGE, JDIMENSION, int);
|
||||
|
||||
switch (cinfo->in_color_space) {
|
||||
case JCS_EXT_RGB:
|
||||
neonfct = jsimd_extrgb_ycc_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_RGBX:
|
||||
case JCS_EXT_RGBA:
|
||||
neonfct = jsimd_extrgbx_ycc_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_BGR:
|
||||
neonfct = jsimd_extbgr_ycc_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_BGRX:
|
||||
case JCS_EXT_BGRA:
|
||||
neonfct = jsimd_extbgrx_ycc_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_XBGR:
|
||||
case JCS_EXT_ABGR:
|
||||
neonfct = jsimd_extxbgr_ycc_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_XRGB:
|
||||
case JCS_EXT_ARGB:
|
||||
neonfct = jsimd_extxrgb_ycc_convert_neon;
|
||||
break;
|
||||
default:
|
||||
neonfct = jsimd_extrgb_ycc_convert_neon;
|
||||
break;
|
||||
}
|
||||
|
||||
neonfct(cinfo->image_width, input_buf, output_buf, output_row, num_rows);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_rgb_gray_convert(j_compress_ptr cinfo, JSAMPARRAY input_buf,
|
||||
JSAMPIMAGE output_buf, JDIMENSION output_row,
|
||||
int num_rows)
|
||||
{
|
||||
void (*neonfct) (JDIMENSION, JSAMPARRAY, JSAMPIMAGE, JDIMENSION, int);
|
||||
|
||||
switch (cinfo->in_color_space) {
|
||||
case JCS_EXT_RGB:
|
||||
neonfct = jsimd_extrgb_gray_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_RGBX:
|
||||
case JCS_EXT_RGBA:
|
||||
neonfct = jsimd_extrgbx_gray_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_BGR:
|
||||
neonfct = jsimd_extbgr_gray_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_BGRX:
|
||||
case JCS_EXT_BGRA:
|
||||
neonfct = jsimd_extbgrx_gray_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_XBGR:
|
||||
case JCS_EXT_ABGR:
|
||||
neonfct = jsimd_extxbgr_gray_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_XRGB:
|
||||
case JCS_EXT_ARGB:
|
||||
neonfct = jsimd_extxrgb_gray_convert_neon;
|
||||
break;
|
||||
default:
|
||||
neonfct = jsimd_extrgb_gray_convert_neon;
|
||||
break;
|
||||
}
|
||||
|
||||
neonfct(cinfo->image_width, input_buf, output_buf, output_row, num_rows);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_ycc_rgb_convert(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
JDIMENSION input_row, JSAMPARRAY output_buf,
|
||||
int num_rows)
|
||||
{
|
||||
void (*neonfct) (JDIMENSION, JSAMPIMAGE, JDIMENSION, JSAMPARRAY, int);
|
||||
|
||||
switch (cinfo->out_color_space) {
|
||||
case JCS_EXT_RGB:
|
||||
neonfct = jsimd_ycc_extrgb_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_RGBX:
|
||||
case JCS_EXT_RGBA:
|
||||
neonfct = jsimd_ycc_extrgbx_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_BGR:
|
||||
neonfct = jsimd_ycc_extbgr_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_BGRX:
|
||||
case JCS_EXT_BGRA:
|
||||
neonfct = jsimd_ycc_extbgrx_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_XBGR:
|
||||
case JCS_EXT_ABGR:
|
||||
neonfct = jsimd_ycc_extxbgr_convert_neon;
|
||||
break;
|
||||
case JCS_EXT_XRGB:
|
||||
case JCS_EXT_ARGB:
|
||||
neonfct = jsimd_ycc_extxrgb_convert_neon;
|
||||
break;
|
||||
default:
|
||||
neonfct = jsimd_ycc_extrgb_convert_neon;
|
||||
break;
|
||||
}
|
||||
|
||||
neonfct(cinfo->output_width, input_buf, input_row, output_buf, num_rows);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_ycc_rgb565_convert(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
JDIMENSION input_row, JSAMPARRAY output_buf,
|
||||
int num_rows)
|
||||
{
|
||||
jsimd_ycc_rgb565_convert_neon(cinfo->output_width, input_buf, input_row,
|
||||
output_buf, num_rows);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v2_downsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v1_downsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v2_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY output_data)
|
||||
{
|
||||
jsimd_h2v2_downsample_neon(cinfo->image_width, cinfo->max_v_samp_factor,
|
||||
compptr->v_samp_factor, compptr->width_in_blocks,
|
||||
input_data, output_data);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v1_downsample(j_compress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY output_data)
|
||||
{
|
||||
jsimd_h2v1_downsample_neon(cinfo->image_width, cinfo->max_v_samp_factor,
|
||||
compptr->v_samp_factor, compptr->width_in_blocks,
|
||||
input_data, output_data);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v2_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v1_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v2_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
jsimd_h2v2_upsample_neon(cinfo->max_v_samp_factor, cinfo->output_width,
|
||||
input_data, output_data_ptr);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v1_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
jsimd_h2v1_upsample_neon(cinfo->max_v_samp_factor, cinfo->output_width,
|
||||
input_data, output_data_ptr);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v2_fancy_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v1_fancy_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h1v2_fancy_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
jsimd_h2v2_fancy_upsample_neon(cinfo->max_v_samp_factor,
|
||||
compptr->downsampled_width, input_data,
|
||||
output_data_ptr);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v1_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
jsimd_h2v1_fancy_upsample_neon(cinfo->max_v_samp_factor,
|
||||
compptr->downsampled_width, input_data,
|
||||
output_data_ptr);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h1v2_fancy_upsample(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JSAMPARRAY input_data, JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
jsimd_h1v2_fancy_upsample_neon(cinfo->max_v_samp_factor,
|
||||
compptr->downsampled_width, input_data,
|
||||
output_data_ptr);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v2_merged_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_h2v1_merged_upsample(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v2_merged_upsample(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
JDIMENSION in_row_group_ctr, JSAMPARRAY output_buf)
|
||||
{
|
||||
void (*neonfct) (JDIMENSION, JSAMPIMAGE, JDIMENSION, JSAMPARRAY);
|
||||
|
||||
switch (cinfo->out_color_space) {
|
||||
case JCS_EXT_RGB:
|
||||
neonfct = jsimd_h2v2_extrgb_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_RGBX:
|
||||
case JCS_EXT_RGBA:
|
||||
neonfct = jsimd_h2v2_extrgbx_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_BGR:
|
||||
neonfct = jsimd_h2v2_extbgr_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_BGRX:
|
||||
case JCS_EXT_BGRA:
|
||||
neonfct = jsimd_h2v2_extbgrx_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_XBGR:
|
||||
case JCS_EXT_ABGR:
|
||||
neonfct = jsimd_h2v2_extxbgr_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_XRGB:
|
||||
case JCS_EXT_ARGB:
|
||||
neonfct = jsimd_h2v2_extxrgb_merged_upsample_neon;
|
||||
break;
|
||||
default:
|
||||
neonfct = jsimd_h2v2_extrgb_merged_upsample_neon;
|
||||
break;
|
||||
}
|
||||
|
||||
neonfct(cinfo->output_width, input_buf, in_row_group_ctr, output_buf);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_h2v1_merged_upsample(j_decompress_ptr cinfo, JSAMPIMAGE input_buf,
|
||||
JDIMENSION in_row_group_ctr, JSAMPARRAY output_buf)
|
||||
{
|
||||
void (*neonfct) (JDIMENSION, JSAMPIMAGE, JDIMENSION, JSAMPARRAY);
|
||||
|
||||
switch (cinfo->out_color_space) {
|
||||
case JCS_EXT_RGB:
|
||||
neonfct = jsimd_h2v1_extrgb_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_RGBX:
|
||||
case JCS_EXT_RGBA:
|
||||
neonfct = jsimd_h2v1_extrgbx_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_BGR:
|
||||
neonfct = jsimd_h2v1_extbgr_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_BGRX:
|
||||
case JCS_EXT_BGRA:
|
||||
neonfct = jsimd_h2v1_extbgrx_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_XBGR:
|
||||
case JCS_EXT_ABGR:
|
||||
neonfct = jsimd_h2v1_extxbgr_merged_upsample_neon;
|
||||
break;
|
||||
case JCS_EXT_XRGB:
|
||||
case JCS_EXT_ARGB:
|
||||
neonfct = jsimd_h2v1_extxrgb_merged_upsample_neon;
|
||||
break;
|
||||
default:
|
||||
neonfct = jsimd_h2v1_extrgb_merged_upsample_neon;
|
||||
break;
|
||||
}
|
||||
|
||||
neonfct(cinfo->output_width, input_buf, in_row_group_ctr, output_buf);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_convsamp(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if (sizeof(DCTELEM) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_convsamp_float(void)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_convsamp(JSAMPARRAY sample_data, JDIMENSION start_col,
|
||||
DCTELEM *workspace)
|
||||
{
|
||||
jsimd_convsamp_neon(sample_data, start_col, workspace);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_convsamp_float(JSAMPARRAY sample_data, JDIMENSION start_col,
|
||||
FAST_FLOAT *workspace)
|
||||
{
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_fdct_islow(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(DCTELEM) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_fdct_ifast(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(DCTELEM) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_fdct_float(void)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_fdct_islow(DCTELEM *data)
|
||||
{
|
||||
jsimd_fdct_islow_neon(data);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_fdct_ifast(DCTELEM *data)
|
||||
{
|
||||
jsimd_fdct_ifast_neon(data);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_fdct_float(FAST_FLOAT *data)
|
||||
{
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_quantize(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
if (sizeof(DCTELEM) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_quantize_float(void)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_quantize(JCOEFPTR coef_block, DCTELEM *divisors, DCTELEM *workspace)
|
||||
{
|
||||
jsimd_quantize_neon(coef_block, divisors, workspace);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_quantize_float(JCOEFPTR coef_block, FAST_FLOAT *divisors,
|
||||
FAST_FLOAT *workspace)
|
||||
{
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_idct_2x2(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if (sizeof(ISLOW_MULT_TYPE) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_idct_4x4(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if (sizeof(ISLOW_MULT_TYPE) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_idct_2x2(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JCOEFPTR coef_block, JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col)
|
||||
{
|
||||
jsimd_idct_2x2_neon(compptr->dct_table, coef_block, output_buf, output_col);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_idct_4x4(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JCOEFPTR coef_block, JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col)
|
||||
{
|
||||
jsimd_idct_4x4_neon(compptr->dct_table, coef_block, output_buf, output_col);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_idct_islow(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if (sizeof(ISLOW_MULT_TYPE) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_idct_ifast(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
/* The code is optimised for these values only */
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
if (BITS_IN_JSAMPLE != 8)
|
||||
return 0;
|
||||
if (sizeof(JDIMENSION) != 4)
|
||||
return 0;
|
||||
if (sizeof(IFAST_MULT_TYPE) != 2)
|
||||
return 0;
|
||||
if (IFAST_SCALE_BITS != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_idct_float(void)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_idct_islow(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JCOEFPTR coef_block, JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col)
|
||||
{
|
||||
jsimd_idct_islow_neon(compptr->dct_table, coef_block, output_buf,
|
||||
output_col);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_idct_ifast(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JCOEFPTR coef_block, JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col)
|
||||
{
|
||||
jsimd_idct_ifast_neon(compptr->dct_table, coef_block, output_buf,
|
||||
output_col);
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_idct_float(j_decompress_ptr cinfo, jpeg_component_info *compptr,
|
||||
JCOEFPTR coef_block, JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col)
|
||||
{
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_huff_encode_one_block(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON && simd_huffman)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(JOCTET *)
|
||||
jsimd_huff_encode_one_block(void *state, JOCTET *buffer, JCOEFPTR block,
|
||||
int last_dc_val, c_derived_tbl *dctbl,
|
||||
c_derived_tbl *actbl)
|
||||
{
|
||||
return jsimd_huff_encode_one_block_neon(state, buffer, block, last_dc_val,
|
||||
dctbl, actbl);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_encode_mcu_AC_first_prepare(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(void)
|
||||
jsimd_encode_mcu_AC_first_prepare(const JCOEF *block,
|
||||
const int *jpeg_natural_order_start, int Sl,
|
||||
int Al, JCOEF *values, size_t *zerobits)
|
||||
{
|
||||
jsimd_encode_mcu_AC_first_prepare_neon(block, jpeg_natural_order_start,
|
||||
Sl, Al, values, zerobits);
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_can_encode_mcu_AC_refine_prepare(void)
|
||||
{
|
||||
init_simd();
|
||||
|
||||
if (DCTSIZE != 8)
|
||||
return 0;
|
||||
if (sizeof(JCOEF) != 2)
|
||||
return 0;
|
||||
|
||||
if (simd_support & JSIMD_NEON)
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
GLOBAL(int)
|
||||
jsimd_encode_mcu_AC_refine_prepare(const JCOEF *block,
|
||||
const int *jpeg_natural_order_start, int Sl,
|
||||
int Al, JCOEF *absvalues, size_t *bits)
|
||||
{
|
||||
return jsimd_encode_mcu_AC_refine_prepare_neon(block,
|
||||
jpeg_natural_order_start, Sl,
|
||||
Al, absvalues, bits);
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,316 @@
|
||||
/*
|
||||
* jccolext-neon.c - colorspace conversion (64-bit Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
/* This file is included by jccolor-neon.c */
|
||||
|
||||
|
||||
/* RGB -> YCbCr conversion is defined by the following equations:
|
||||
* Y = 0.29900 * R + 0.58700 * G + 0.11400 * B
|
||||
* Cb = -0.16874 * R - 0.33126 * G + 0.50000 * B + 128
|
||||
* Cr = 0.50000 * R - 0.41869 * G - 0.08131 * B + 128
|
||||
*
|
||||
* Avoid floating point arithmetic by using shifted integer constants:
|
||||
* 0.29899597 = 19595 * 2^-16
|
||||
* 0.58700561 = 38470 * 2^-16
|
||||
* 0.11399841 = 7471 * 2^-16
|
||||
* 0.16874695 = 11059 * 2^-16
|
||||
* 0.33125305 = 21709 * 2^-16
|
||||
* 0.50000000 = 32768 * 2^-16
|
||||
* 0.41868592 = 27439 * 2^-16
|
||||
* 0.08131409 = 5329 * 2^-16
|
||||
* These constants are defined in jccolor-neon.c
|
||||
*
|
||||
* We add the fixed-point equivalent of 0.5 to Cb and Cr, which effectively
|
||||
* rounds up or down the result via integer truncation.
|
||||
*/
|
||||
|
||||
void jsimd_rgb_ycc_convert_neon(JDIMENSION image_width, JSAMPARRAY input_buf,
|
||||
JSAMPIMAGE output_buf, JDIMENSION output_row,
|
||||
int num_rows)
|
||||
{
|
||||
/* Pointer to RGB(X/A) input data */
|
||||
JSAMPROW inptr;
|
||||
/* Pointers to Y, Cb, and Cr output data */
|
||||
JSAMPROW outptr0, outptr1, outptr2;
|
||||
/* Allocate temporary buffer for final (image_width % 16) pixels in row. */
|
||||
ALIGN(16) uint8_t tmp_buf[16 * RGB_PIXELSIZE];
|
||||
|
||||
/* Set up conversion constants. */
|
||||
const uint16x8_t consts = vld1q_u16(jsimd_rgb_ycc_neon_consts);
|
||||
const uint32x4_t scaled_128_5 = vdupq_n_u32((128 << 16) + 32767);
|
||||
|
||||
while (--num_rows >= 0) {
|
||||
inptr = *input_buf++;
|
||||
outptr0 = output_buf[0][output_row];
|
||||
outptr1 = output_buf[1][output_row];
|
||||
outptr2 = output_buf[2][output_row];
|
||||
output_row++;
|
||||
|
||||
int cols_remaining = image_width;
|
||||
for (; cols_remaining >= 16; cols_remaining -= 16) {
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x16x4_t input_pixels = vld4q_u8(inptr);
|
||||
#else
|
||||
uint8x16x3_t input_pixels = vld3q_u8(inptr);
|
||||
#endif
|
||||
uint16x8_t r_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_RED]));
|
||||
uint16x8_t g_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_GREEN]));
|
||||
uint16x8_t b_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_BLUE]));
|
||||
uint16x8_t r_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_RED]));
|
||||
uint16x8_t g_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_GREEN]));
|
||||
uint16x8_t b_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_BLUE]));
|
||||
|
||||
/* Compute Y = 0.29900 * R + 0.58700 * G + 0.11400 * B */
|
||||
uint32x4_t y_ll = vmull_laneq_u16(vget_low_u16(r_l), consts, 0);
|
||||
y_ll = vmlal_laneq_u16(y_ll, vget_low_u16(g_l), consts, 1);
|
||||
y_ll = vmlal_laneq_u16(y_ll, vget_low_u16(b_l), consts, 2);
|
||||
uint32x4_t y_lh = vmull_laneq_u16(vget_high_u16(r_l), consts, 0);
|
||||
y_lh = vmlal_laneq_u16(y_lh, vget_high_u16(g_l), consts, 1);
|
||||
y_lh = vmlal_laneq_u16(y_lh, vget_high_u16(b_l), consts, 2);
|
||||
uint32x4_t y_hl = vmull_laneq_u16(vget_low_u16(r_h), consts, 0);
|
||||
y_hl = vmlal_laneq_u16(y_hl, vget_low_u16(g_h), consts, 1);
|
||||
y_hl = vmlal_laneq_u16(y_hl, vget_low_u16(b_h), consts, 2);
|
||||
uint32x4_t y_hh = vmull_laneq_u16(vget_high_u16(r_h), consts, 0);
|
||||
y_hh = vmlal_laneq_u16(y_hh, vget_high_u16(g_h), consts, 1);
|
||||
y_hh = vmlal_laneq_u16(y_hh, vget_high_u16(b_h), consts, 2);
|
||||
|
||||
/* Compute Cb = -0.16874 * R - 0.33126 * G + 0.50000 * B + 128 */
|
||||
uint32x4_t cb_ll = scaled_128_5;
|
||||
cb_ll = vmlsl_laneq_u16(cb_ll, vget_low_u16(r_l), consts, 3);
|
||||
cb_ll = vmlsl_laneq_u16(cb_ll, vget_low_u16(g_l), consts, 4);
|
||||
cb_ll = vmlal_laneq_u16(cb_ll, vget_low_u16(b_l), consts, 5);
|
||||
uint32x4_t cb_lh = scaled_128_5;
|
||||
cb_lh = vmlsl_laneq_u16(cb_lh, vget_high_u16(r_l), consts, 3);
|
||||
cb_lh = vmlsl_laneq_u16(cb_lh, vget_high_u16(g_l), consts, 4);
|
||||
cb_lh = vmlal_laneq_u16(cb_lh, vget_high_u16(b_l), consts, 5);
|
||||
uint32x4_t cb_hl = scaled_128_5;
|
||||
cb_hl = vmlsl_laneq_u16(cb_hl, vget_low_u16(r_h), consts, 3);
|
||||
cb_hl = vmlsl_laneq_u16(cb_hl, vget_low_u16(g_h), consts, 4);
|
||||
cb_hl = vmlal_laneq_u16(cb_hl, vget_low_u16(b_h), consts, 5);
|
||||
uint32x4_t cb_hh = scaled_128_5;
|
||||
cb_hh = vmlsl_laneq_u16(cb_hh, vget_high_u16(r_h), consts, 3);
|
||||
cb_hh = vmlsl_laneq_u16(cb_hh, vget_high_u16(g_h), consts, 4);
|
||||
cb_hh = vmlal_laneq_u16(cb_hh, vget_high_u16(b_h), consts, 5);
|
||||
|
||||
/* Compute Cr = 0.50000 * R - 0.41869 * G - 0.08131 * B + 128 */
|
||||
uint32x4_t cr_ll = scaled_128_5;
|
||||
cr_ll = vmlal_laneq_u16(cr_ll, vget_low_u16(r_l), consts, 5);
|
||||
cr_ll = vmlsl_laneq_u16(cr_ll, vget_low_u16(g_l), consts, 6);
|
||||
cr_ll = vmlsl_laneq_u16(cr_ll, vget_low_u16(b_l), consts, 7);
|
||||
uint32x4_t cr_lh = scaled_128_5;
|
||||
cr_lh = vmlal_laneq_u16(cr_lh, vget_high_u16(r_l), consts, 5);
|
||||
cr_lh = vmlsl_laneq_u16(cr_lh, vget_high_u16(g_l), consts, 6);
|
||||
cr_lh = vmlsl_laneq_u16(cr_lh, vget_high_u16(b_l), consts, 7);
|
||||
uint32x4_t cr_hl = scaled_128_5;
|
||||
cr_hl = vmlal_laneq_u16(cr_hl, vget_low_u16(r_h), consts, 5);
|
||||
cr_hl = vmlsl_laneq_u16(cr_hl, vget_low_u16(g_h), consts, 6);
|
||||
cr_hl = vmlsl_laneq_u16(cr_hl, vget_low_u16(b_h), consts, 7);
|
||||
uint32x4_t cr_hh = scaled_128_5;
|
||||
cr_hh = vmlal_laneq_u16(cr_hh, vget_high_u16(r_h), consts, 5);
|
||||
cr_hh = vmlsl_laneq_u16(cr_hh, vget_high_u16(g_h), consts, 6);
|
||||
cr_hh = vmlsl_laneq_u16(cr_hh, vget_high_u16(b_h), consts, 7);
|
||||
|
||||
/* Descale Y values (rounding right shift) and narrow to 16-bit. */
|
||||
uint16x8_t y_l = vcombine_u16(vrshrn_n_u32(y_ll, 16),
|
||||
vrshrn_n_u32(y_lh, 16));
|
||||
uint16x8_t y_h = vcombine_u16(vrshrn_n_u32(y_hl, 16),
|
||||
vrshrn_n_u32(y_hh, 16));
|
||||
/* Descale Cb values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cb_l = vcombine_u16(vshrn_n_u32(cb_ll, 16),
|
||||
vshrn_n_u32(cb_lh, 16));
|
||||
uint16x8_t cb_h = vcombine_u16(vshrn_n_u32(cb_hl, 16),
|
||||
vshrn_n_u32(cb_hh, 16));
|
||||
/* Descale Cr values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cr_l = vcombine_u16(vshrn_n_u32(cr_ll, 16),
|
||||
vshrn_n_u32(cr_lh, 16));
|
||||
uint16x8_t cr_h = vcombine_u16(vshrn_n_u32(cr_hl, 16),
|
||||
vshrn_n_u32(cr_hh, 16));
|
||||
/* Narrow Y, Cb, and Cr values to 8-bit and store to memory. Buffer
|
||||
* overwrite is permitted up to the next multiple of ALIGN_SIZE bytes.
|
||||
*/
|
||||
vst1q_u8(outptr0, vcombine_u8(vmovn_u16(y_l), vmovn_u16(y_h)));
|
||||
vst1q_u8(outptr1, vcombine_u8(vmovn_u16(cb_l), vmovn_u16(cb_h)));
|
||||
vst1q_u8(outptr2, vcombine_u8(vmovn_u16(cr_l), vmovn_u16(cr_h)));
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr += (16 * RGB_PIXELSIZE);
|
||||
outptr0 += 16;
|
||||
outptr1 += 16;
|
||||
outptr2 += 16;
|
||||
}
|
||||
|
||||
if (cols_remaining > 8) {
|
||||
/* To prevent buffer overread by the vector load instructions, the last
|
||||
* (image_width % 16) columns of data are first memcopied to a temporary
|
||||
* buffer large enough to accommodate the vector load.
|
||||
*/
|
||||
memcpy(tmp_buf, inptr, cols_remaining * RGB_PIXELSIZE);
|
||||
inptr = tmp_buf;
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x16x4_t input_pixels = vld4q_u8(inptr);
|
||||
#else
|
||||
uint8x16x3_t input_pixels = vld3q_u8(inptr);
|
||||
#endif
|
||||
uint16x8_t r_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_RED]));
|
||||
uint16x8_t g_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_GREEN]));
|
||||
uint16x8_t b_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_BLUE]));
|
||||
uint16x8_t r_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_RED]));
|
||||
uint16x8_t g_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_GREEN]));
|
||||
uint16x8_t b_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_BLUE]));
|
||||
|
||||
/* Compute Y = 0.29900 * R + 0.58700 * G + 0.11400 * B */
|
||||
uint32x4_t y_ll = vmull_laneq_u16(vget_low_u16(r_l), consts, 0);
|
||||
y_ll = vmlal_laneq_u16(y_ll, vget_low_u16(g_l), consts, 1);
|
||||
y_ll = vmlal_laneq_u16(y_ll, vget_low_u16(b_l), consts, 2);
|
||||
uint32x4_t y_lh = vmull_laneq_u16(vget_high_u16(r_l), consts, 0);
|
||||
y_lh = vmlal_laneq_u16(y_lh, vget_high_u16(g_l), consts, 1);
|
||||
y_lh = vmlal_laneq_u16(y_lh, vget_high_u16(b_l), consts, 2);
|
||||
uint32x4_t y_hl = vmull_laneq_u16(vget_low_u16(r_h), consts, 0);
|
||||
y_hl = vmlal_laneq_u16(y_hl, vget_low_u16(g_h), consts, 1);
|
||||
y_hl = vmlal_laneq_u16(y_hl, vget_low_u16(b_h), consts, 2);
|
||||
uint32x4_t y_hh = vmull_laneq_u16(vget_high_u16(r_h), consts, 0);
|
||||
y_hh = vmlal_laneq_u16(y_hh, vget_high_u16(g_h), consts, 1);
|
||||
y_hh = vmlal_laneq_u16(y_hh, vget_high_u16(b_h), consts, 2);
|
||||
|
||||
/* Compute Cb = -0.16874 * R - 0.33126 * G + 0.50000 * B + 128 */
|
||||
uint32x4_t cb_ll = scaled_128_5;
|
||||
cb_ll = vmlsl_laneq_u16(cb_ll, vget_low_u16(r_l), consts, 3);
|
||||
cb_ll = vmlsl_laneq_u16(cb_ll, vget_low_u16(g_l), consts, 4);
|
||||
cb_ll = vmlal_laneq_u16(cb_ll, vget_low_u16(b_l), consts, 5);
|
||||
uint32x4_t cb_lh = scaled_128_5;
|
||||
cb_lh = vmlsl_laneq_u16(cb_lh, vget_high_u16(r_l), consts, 3);
|
||||
cb_lh = vmlsl_laneq_u16(cb_lh, vget_high_u16(g_l), consts, 4);
|
||||
cb_lh = vmlal_laneq_u16(cb_lh, vget_high_u16(b_l), consts, 5);
|
||||
uint32x4_t cb_hl = scaled_128_5;
|
||||
cb_hl = vmlsl_laneq_u16(cb_hl, vget_low_u16(r_h), consts, 3);
|
||||
cb_hl = vmlsl_laneq_u16(cb_hl, vget_low_u16(g_h), consts, 4);
|
||||
cb_hl = vmlal_laneq_u16(cb_hl, vget_low_u16(b_h), consts, 5);
|
||||
uint32x4_t cb_hh = scaled_128_5;
|
||||
cb_hh = vmlsl_laneq_u16(cb_hh, vget_high_u16(r_h), consts, 3);
|
||||
cb_hh = vmlsl_laneq_u16(cb_hh, vget_high_u16(g_h), consts, 4);
|
||||
cb_hh = vmlal_laneq_u16(cb_hh, vget_high_u16(b_h), consts, 5);
|
||||
|
||||
/* Compute Cr = 0.50000 * R - 0.41869 * G - 0.08131 * B + 128 */
|
||||
uint32x4_t cr_ll = scaled_128_5;
|
||||
cr_ll = vmlal_laneq_u16(cr_ll, vget_low_u16(r_l), consts, 5);
|
||||
cr_ll = vmlsl_laneq_u16(cr_ll, vget_low_u16(g_l), consts, 6);
|
||||
cr_ll = vmlsl_laneq_u16(cr_ll, vget_low_u16(b_l), consts, 7);
|
||||
uint32x4_t cr_lh = scaled_128_5;
|
||||
cr_lh = vmlal_laneq_u16(cr_lh, vget_high_u16(r_l), consts, 5);
|
||||
cr_lh = vmlsl_laneq_u16(cr_lh, vget_high_u16(g_l), consts, 6);
|
||||
cr_lh = vmlsl_laneq_u16(cr_lh, vget_high_u16(b_l), consts, 7);
|
||||
uint32x4_t cr_hl = scaled_128_5;
|
||||
cr_hl = vmlal_laneq_u16(cr_hl, vget_low_u16(r_h), consts, 5);
|
||||
cr_hl = vmlsl_laneq_u16(cr_hl, vget_low_u16(g_h), consts, 6);
|
||||
cr_hl = vmlsl_laneq_u16(cr_hl, vget_low_u16(b_h), consts, 7);
|
||||
uint32x4_t cr_hh = scaled_128_5;
|
||||
cr_hh = vmlal_laneq_u16(cr_hh, vget_high_u16(r_h), consts, 5);
|
||||
cr_hh = vmlsl_laneq_u16(cr_hh, vget_high_u16(g_h), consts, 6);
|
||||
cr_hh = vmlsl_laneq_u16(cr_hh, vget_high_u16(b_h), consts, 7);
|
||||
|
||||
/* Descale Y values (rounding right shift) and narrow to 16-bit. */
|
||||
uint16x8_t y_l = vcombine_u16(vrshrn_n_u32(y_ll, 16),
|
||||
vrshrn_n_u32(y_lh, 16));
|
||||
uint16x8_t y_h = vcombine_u16(vrshrn_n_u32(y_hl, 16),
|
||||
vrshrn_n_u32(y_hh, 16));
|
||||
/* Descale Cb values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cb_l = vcombine_u16(vshrn_n_u32(cb_ll, 16),
|
||||
vshrn_n_u32(cb_lh, 16));
|
||||
uint16x8_t cb_h = vcombine_u16(vshrn_n_u32(cb_hl, 16),
|
||||
vshrn_n_u32(cb_hh, 16));
|
||||
/* Descale Cr values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cr_l = vcombine_u16(vshrn_n_u32(cr_ll, 16),
|
||||
vshrn_n_u32(cr_lh, 16));
|
||||
uint16x8_t cr_h = vcombine_u16(vshrn_n_u32(cr_hl, 16),
|
||||
vshrn_n_u32(cr_hh, 16));
|
||||
/* Narrow Y, Cb, and Cr values to 8-bit and store to memory. Buffer
|
||||
* overwrite is permitted up to the next multiple of ALIGN_SIZE bytes.
|
||||
*/
|
||||
vst1q_u8(outptr0, vcombine_u8(vmovn_u16(y_l), vmovn_u16(y_h)));
|
||||
vst1q_u8(outptr1, vcombine_u8(vmovn_u16(cb_l), vmovn_u16(cb_h)));
|
||||
vst1q_u8(outptr2, vcombine_u8(vmovn_u16(cr_l), vmovn_u16(cr_h)));
|
||||
|
||||
} else if (cols_remaining > 0) {
|
||||
/* To prevent buffer overread by the vector load instructions, the last
|
||||
* (image_width % 8) columns of data are first memcopied to a temporary
|
||||
* buffer large enough to accommodate the vector load.
|
||||
*/
|
||||
memcpy(tmp_buf, inptr, cols_remaining * RGB_PIXELSIZE);
|
||||
inptr = tmp_buf;
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x8x4_t input_pixels = vld4_u8(inptr);
|
||||
#else
|
||||
uint8x8x3_t input_pixels = vld3_u8(inptr);
|
||||
#endif
|
||||
uint16x8_t r = vmovl_u8(input_pixels.val[RGB_RED]);
|
||||
uint16x8_t g = vmovl_u8(input_pixels.val[RGB_GREEN]);
|
||||
uint16x8_t b = vmovl_u8(input_pixels.val[RGB_BLUE]);
|
||||
|
||||
/* Compute Y = 0.29900 * R + 0.58700 * G + 0.11400 * B */
|
||||
uint32x4_t y_l = vmull_laneq_u16(vget_low_u16(r), consts, 0);
|
||||
y_l = vmlal_laneq_u16(y_l, vget_low_u16(g), consts, 1);
|
||||
y_l = vmlal_laneq_u16(y_l, vget_low_u16(b), consts, 2);
|
||||
uint32x4_t y_h = vmull_laneq_u16(vget_high_u16(r), consts, 0);
|
||||
y_h = vmlal_laneq_u16(y_h, vget_high_u16(g), consts, 1);
|
||||
y_h = vmlal_laneq_u16(y_h, vget_high_u16(b), consts, 2);
|
||||
|
||||
/* Compute Cb = -0.16874 * R - 0.33126 * G + 0.50000 * B + 128 */
|
||||
uint32x4_t cb_l = scaled_128_5;
|
||||
cb_l = vmlsl_laneq_u16(cb_l, vget_low_u16(r), consts, 3);
|
||||
cb_l = vmlsl_laneq_u16(cb_l, vget_low_u16(g), consts, 4);
|
||||
cb_l = vmlal_laneq_u16(cb_l, vget_low_u16(b), consts, 5);
|
||||
uint32x4_t cb_h = scaled_128_5;
|
||||
cb_h = vmlsl_laneq_u16(cb_h, vget_high_u16(r), consts, 3);
|
||||
cb_h = vmlsl_laneq_u16(cb_h, vget_high_u16(g), consts, 4);
|
||||
cb_h = vmlal_laneq_u16(cb_h, vget_high_u16(b), consts, 5);
|
||||
|
||||
/* Compute Cr = 0.50000 * R - 0.41869 * G - 0.08131 * B + 128 */
|
||||
uint32x4_t cr_l = scaled_128_5;
|
||||
cr_l = vmlal_laneq_u16(cr_l, vget_low_u16(r), consts, 5);
|
||||
cr_l = vmlsl_laneq_u16(cr_l, vget_low_u16(g), consts, 6);
|
||||
cr_l = vmlsl_laneq_u16(cr_l, vget_low_u16(b), consts, 7);
|
||||
uint32x4_t cr_h = scaled_128_5;
|
||||
cr_h = vmlal_laneq_u16(cr_h, vget_high_u16(r), consts, 5);
|
||||
cr_h = vmlsl_laneq_u16(cr_h, vget_high_u16(g), consts, 6);
|
||||
cr_h = vmlsl_laneq_u16(cr_h, vget_high_u16(b), consts, 7);
|
||||
|
||||
/* Descale Y values (rounding right shift) and narrow to 16-bit. */
|
||||
uint16x8_t y_u16 = vcombine_u16(vrshrn_n_u32(y_l, 16),
|
||||
vrshrn_n_u32(y_h, 16));
|
||||
/* Descale Cb values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cb_u16 = vcombine_u16(vshrn_n_u32(cb_l, 16),
|
||||
vshrn_n_u32(cb_h, 16));
|
||||
/* Descale Cr values (right shift) and narrow to 16-bit. */
|
||||
uint16x8_t cr_u16 = vcombine_u16(vshrn_n_u32(cr_l, 16),
|
||||
vshrn_n_u32(cr_h, 16));
|
||||
/* Narrow Y, Cb, and Cr values to 8-bit and store to memory. Buffer
|
||||
* overwrite is permitted up to the next multiple of ALIGN_SIZE bytes.
|
||||
*/
|
||||
vst1_u8(outptr0, vmovn_u16(y_u16));
|
||||
vst1_u8(outptr1, vmovn_u16(cb_u16));
|
||||
vst1_u8(outptr2, vmovn_u16(cr_u16));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,411 @@
|
||||
/*
|
||||
* jchuff-neon.c - Huffman entropy encoding (64-bit Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020-2021, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, 2022, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*
|
||||
* NOTE: All referenced figures are from
|
||||
* Recommendation ITU-T T.81 (1992) | ISO/IEC 10918-1:1994.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../../jinclude.h"
|
||||
#include "../../../jpeglib.h"
|
||||
#include "../../../jsimd.h"
|
||||
#include "../../../jdct.h"
|
||||
#include "../../../jsimddct.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../align.h"
|
||||
#include "../jchuff.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <limits.h>
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
ALIGN(16) static const uint8_t jsimd_huff_encode_one_block_consts[] = {
|
||||
0, 1, 2, 3, 16, 17, 32, 33,
|
||||
18, 19, 4, 5, 6, 7, 20, 21,
|
||||
34, 35, 48, 49, 255, 255, 50, 51,
|
||||
36, 37, 22, 23, 8, 9, 10, 11,
|
||||
255, 255, 6, 7, 20, 21, 34, 35,
|
||||
48, 49, 255, 255, 50, 51, 36, 37,
|
||||
54, 55, 40, 41, 26, 27, 12, 13,
|
||||
14, 15, 28, 29, 42, 43, 56, 57,
|
||||
6, 7, 20, 21, 34, 35, 48, 49,
|
||||
50, 51, 36, 37, 22, 23, 8, 9,
|
||||
26, 27, 12, 13, 255, 255, 14, 15,
|
||||
28, 29, 42, 43, 56, 57, 255, 255,
|
||||
52, 53, 54, 55, 40, 41, 26, 27,
|
||||
12, 13, 255, 255, 14, 15, 28, 29,
|
||||
26, 27, 40, 41, 42, 43, 28, 29,
|
||||
14, 15, 30, 31, 44, 45, 46, 47
|
||||
};
|
||||
|
||||
/* The AArch64 implementation of the FLUSH() macro triggers a UBSan misaligned
|
||||
* address warning because the macro sometimes writes a 64-bit value to a
|
||||
* non-64-bit-aligned address. That behavior is technically undefined per
|
||||
* the C specification, but it is supported by the AArch64 architecture and
|
||||
* compilers.
|
||||
*/
|
||||
#if defined(__has_feature)
|
||||
#if __has_feature(undefined_behavior_sanitizer)
|
||||
__attribute__((no_sanitize("alignment")))
|
||||
#endif
|
||||
#endif
|
||||
JOCTET *jsimd_huff_encode_one_block_neon(void *state, JOCTET *buffer,
|
||||
JCOEFPTR block, int last_dc_val,
|
||||
c_derived_tbl *dctbl,
|
||||
c_derived_tbl *actbl)
|
||||
{
|
||||
uint16_t block_diff[DCTSIZE2];
|
||||
|
||||
/* Load lookup table indices for rows of zig-zag ordering. */
|
||||
#ifdef HAVE_VLD1Q_U8_X4
|
||||
const uint8x16x4_t idx_rows_0123 =
|
||||
vld1q_u8_x4(jsimd_huff_encode_one_block_consts + 0 * DCTSIZE);
|
||||
const uint8x16x4_t idx_rows_4567 =
|
||||
vld1q_u8_x4(jsimd_huff_encode_one_block_consts + 8 * DCTSIZE);
|
||||
#else
|
||||
/* GCC does not currently support intrinsics vl1dq_<type>_x4(). */
|
||||
const uint8x16x4_t idx_rows_0123 = { {
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 0 * DCTSIZE),
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 2 * DCTSIZE),
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 4 * DCTSIZE),
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 6 * DCTSIZE)
|
||||
} };
|
||||
const uint8x16x4_t idx_rows_4567 = { {
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 8 * DCTSIZE),
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 10 * DCTSIZE),
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 12 * DCTSIZE),
|
||||
vld1q_u8(jsimd_huff_encode_one_block_consts + 14 * DCTSIZE)
|
||||
} };
|
||||
#endif
|
||||
|
||||
/* Load 8x8 block of DCT coefficients. */
|
||||
#ifdef HAVE_VLD1Q_U8_X4
|
||||
const int8x16x4_t tbl_rows_0123 =
|
||||
vld1q_s8_x4((int8_t *)(block + 0 * DCTSIZE));
|
||||
const int8x16x4_t tbl_rows_4567 =
|
||||
vld1q_s8_x4((int8_t *)(block + 4 * DCTSIZE));
|
||||
#else
|
||||
const int8x16x4_t tbl_rows_0123 = { {
|
||||
vld1q_s8((int8_t *)(block + 0 * DCTSIZE)),
|
||||
vld1q_s8((int8_t *)(block + 1 * DCTSIZE)),
|
||||
vld1q_s8((int8_t *)(block + 2 * DCTSIZE)),
|
||||
vld1q_s8((int8_t *)(block + 3 * DCTSIZE))
|
||||
} };
|
||||
const int8x16x4_t tbl_rows_4567 = { {
|
||||
vld1q_s8((int8_t *)(block + 4 * DCTSIZE)),
|
||||
vld1q_s8((int8_t *)(block + 5 * DCTSIZE)),
|
||||
vld1q_s8((int8_t *)(block + 6 * DCTSIZE)),
|
||||
vld1q_s8((int8_t *)(block + 7 * DCTSIZE))
|
||||
} };
|
||||
#endif
|
||||
|
||||
/* Initialise extra lookup tables. */
|
||||
const int8x16x4_t tbl_rows_2345 = { {
|
||||
tbl_rows_0123.val[2], tbl_rows_0123.val[3],
|
||||
tbl_rows_4567.val[0], tbl_rows_4567.val[1]
|
||||
} };
|
||||
const int8x16x3_t tbl_rows_567 =
|
||||
{ { tbl_rows_4567.val[1], tbl_rows_4567.val[2], tbl_rows_4567.val[3] } };
|
||||
|
||||
/* Shuffle coefficients into zig-zag order. */
|
||||
int16x8_t row0 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_0123, idx_rows_0123.val[0]));
|
||||
int16x8_t row1 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_0123, idx_rows_0123.val[1]));
|
||||
int16x8_t row2 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_2345, idx_rows_0123.val[2]));
|
||||
int16x8_t row3 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_0123, idx_rows_0123.val[3]));
|
||||
int16x8_t row4 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_4567, idx_rows_4567.val[0]));
|
||||
int16x8_t row5 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_2345, idx_rows_4567.val[1]));
|
||||
int16x8_t row6 =
|
||||
vreinterpretq_s16_s8(vqtbl4q_s8(tbl_rows_4567, idx_rows_4567.val[2]));
|
||||
int16x8_t row7 =
|
||||
vreinterpretq_s16_s8(vqtbl3q_s8(tbl_rows_567, idx_rows_4567.val[3]));
|
||||
|
||||
/* Compute DC coefficient difference value (F.1.1.5.1). */
|
||||
row0 = vsetq_lane_s16(block[0] - last_dc_val, row0, 0);
|
||||
/* Initialize AC coefficient lanes not reachable by lookup tables. */
|
||||
row1 =
|
||||
vsetq_lane_s16(vgetq_lane_s16(vreinterpretq_s16_s8(tbl_rows_4567.val[0]),
|
||||
0), row1, 2);
|
||||
row2 =
|
||||
vsetq_lane_s16(vgetq_lane_s16(vreinterpretq_s16_s8(tbl_rows_0123.val[1]),
|
||||
4), row2, 0);
|
||||
row2 =
|
||||
vsetq_lane_s16(vgetq_lane_s16(vreinterpretq_s16_s8(tbl_rows_4567.val[2]),
|
||||
0), row2, 5);
|
||||
row5 =
|
||||
vsetq_lane_s16(vgetq_lane_s16(vreinterpretq_s16_s8(tbl_rows_0123.val[1]),
|
||||
7), row5, 2);
|
||||
row5 =
|
||||
vsetq_lane_s16(vgetq_lane_s16(vreinterpretq_s16_s8(tbl_rows_4567.val[2]),
|
||||
3), row5, 7);
|
||||
row6 =
|
||||
vsetq_lane_s16(vgetq_lane_s16(vreinterpretq_s16_s8(tbl_rows_0123.val[3]),
|
||||
7), row6, 5);
|
||||
|
||||
/* DCT block is now in zig-zag order; start Huffman encoding process. */
|
||||
|
||||
/* Construct bitmap to accelerate encoding of AC coefficients. A set bit
|
||||
* means that the corresponding coefficient != 0.
|
||||
*/
|
||||
uint16x8_t row0_ne_0 = vtstq_s16(row0, row0);
|
||||
uint16x8_t row1_ne_0 = vtstq_s16(row1, row1);
|
||||
uint16x8_t row2_ne_0 = vtstq_s16(row2, row2);
|
||||
uint16x8_t row3_ne_0 = vtstq_s16(row3, row3);
|
||||
uint16x8_t row4_ne_0 = vtstq_s16(row4, row4);
|
||||
uint16x8_t row5_ne_0 = vtstq_s16(row5, row5);
|
||||
uint16x8_t row6_ne_0 = vtstq_s16(row6, row6);
|
||||
uint16x8_t row7_ne_0 = vtstq_s16(row7, row7);
|
||||
|
||||
uint8x16_t row10_ne_0 = vuzp1q_u8(vreinterpretq_u8_u16(row1_ne_0),
|
||||
vreinterpretq_u8_u16(row0_ne_0));
|
||||
uint8x16_t row32_ne_0 = vuzp1q_u8(vreinterpretq_u8_u16(row3_ne_0),
|
||||
vreinterpretq_u8_u16(row2_ne_0));
|
||||
uint8x16_t row54_ne_0 = vuzp1q_u8(vreinterpretq_u8_u16(row5_ne_0),
|
||||
vreinterpretq_u8_u16(row4_ne_0));
|
||||
uint8x16_t row76_ne_0 = vuzp1q_u8(vreinterpretq_u8_u16(row7_ne_0),
|
||||
vreinterpretq_u8_u16(row6_ne_0));
|
||||
|
||||
/* { 0x80, 0x40, 0x20, 0x10, 0x08, 0x04, 0x02, 0x01 } */
|
||||
const uint8x16_t bitmap_mask =
|
||||
vreinterpretq_u8_u64(vdupq_n_u64(0x0102040810204080));
|
||||
|
||||
uint8x16_t bitmap_rows_10 = vandq_u8(row10_ne_0, bitmap_mask);
|
||||
uint8x16_t bitmap_rows_32 = vandq_u8(row32_ne_0, bitmap_mask);
|
||||
uint8x16_t bitmap_rows_54 = vandq_u8(row54_ne_0, bitmap_mask);
|
||||
uint8x16_t bitmap_rows_76 = vandq_u8(row76_ne_0, bitmap_mask);
|
||||
|
||||
uint8x16_t bitmap_rows_3210 = vpaddq_u8(bitmap_rows_32, bitmap_rows_10);
|
||||
uint8x16_t bitmap_rows_7654 = vpaddq_u8(bitmap_rows_76, bitmap_rows_54);
|
||||
uint8x16_t bitmap_rows_76543210 = vpaddq_u8(bitmap_rows_7654,
|
||||
bitmap_rows_3210);
|
||||
uint8x8_t bitmap_all = vpadd_u8(vget_low_u8(bitmap_rows_76543210),
|
||||
vget_high_u8(bitmap_rows_76543210));
|
||||
|
||||
/* Shift left to remove DC bit. */
|
||||
bitmap_all =
|
||||
vreinterpret_u8_u64(vshl_n_u64(vreinterpret_u64_u8(bitmap_all), 1));
|
||||
/* Count bits set (number of non-zero coefficients) in bitmap. */
|
||||
unsigned int non_zero_coefficients = vaddv_u8(vcnt_u8(bitmap_all));
|
||||
/* Move bitmap to 64-bit scalar register. */
|
||||
uint64_t bitmap = vget_lane_u64(vreinterpret_u64_u8(bitmap_all), 0);
|
||||
|
||||
/* Set up state and bit buffer for output bitstream. */
|
||||
working_state *state_ptr = (working_state *)state;
|
||||
int free_bits = state_ptr->cur.free_bits;
|
||||
size_t put_buffer = state_ptr->cur.put_buffer;
|
||||
|
||||
/* Encode DC coefficient. */
|
||||
|
||||
/* For negative coeffs: diff = abs(coeff) -1 = ~abs(coeff) */
|
||||
int16x8_t abs_row0 = vabsq_s16(row0);
|
||||
int16x8_t row0_lz = vclzq_s16(abs_row0);
|
||||
uint16x8_t row0_mask = vshlq_u16(vcltzq_s16(row0), vnegq_s16(row0_lz));
|
||||
uint16x8_t row0_diff = veorq_u16(vreinterpretq_u16_s16(abs_row0), row0_mask);
|
||||
/* Find nbits required to specify sign and amplitude of coefficient. */
|
||||
unsigned int lz = vgetq_lane_u16(vreinterpretq_u16_s16(row0_lz), 0);
|
||||
unsigned int nbits = 16 - lz;
|
||||
/* Emit Huffman-coded symbol and additional diff bits. */
|
||||
unsigned int diff = vgetq_lane_u16(row0_diff, 0);
|
||||
PUT_CODE(dctbl->ehufco[nbits], dctbl->ehufsi[nbits], diff)
|
||||
|
||||
/* Encode AC coefficients. */
|
||||
|
||||
unsigned int r = 0; /* r = run length of zeros */
|
||||
unsigned int i = 1; /* i = number of coefficients encoded */
|
||||
/* Code and size information for a run length of 16 zero coefficients */
|
||||
const unsigned int code_0xf0 = actbl->ehufco[0xf0];
|
||||
const unsigned int size_0xf0 = actbl->ehufsi[0xf0];
|
||||
|
||||
/* The most efficient method of computing nbits and diff depends on the
|
||||
* number of non-zero coefficients. If the bitmap is not too sparse (> 8
|
||||
* non-zero AC coefficients), it is beneficial to do all of the work using
|
||||
* Neon; else we do some of the work using Neon and the rest on demand using
|
||||
* scalar code.
|
||||
*/
|
||||
if (non_zero_coefficients > 8) {
|
||||
uint8_t block_nbits[DCTSIZE2];
|
||||
|
||||
int16x8_t abs_row1 = vabsq_s16(row1);
|
||||
int16x8_t abs_row2 = vabsq_s16(row2);
|
||||
int16x8_t abs_row3 = vabsq_s16(row3);
|
||||
int16x8_t abs_row4 = vabsq_s16(row4);
|
||||
int16x8_t abs_row5 = vabsq_s16(row5);
|
||||
int16x8_t abs_row6 = vabsq_s16(row6);
|
||||
int16x8_t abs_row7 = vabsq_s16(row7);
|
||||
int16x8_t row1_lz = vclzq_s16(abs_row1);
|
||||
int16x8_t row2_lz = vclzq_s16(abs_row2);
|
||||
int16x8_t row3_lz = vclzq_s16(abs_row3);
|
||||
int16x8_t row4_lz = vclzq_s16(abs_row4);
|
||||
int16x8_t row5_lz = vclzq_s16(abs_row5);
|
||||
int16x8_t row6_lz = vclzq_s16(abs_row6);
|
||||
int16x8_t row7_lz = vclzq_s16(abs_row7);
|
||||
/* Narrow leading zero count to 8 bits. */
|
||||
uint8x16_t row01_lz = vuzp1q_u8(vreinterpretq_u8_s16(row0_lz),
|
||||
vreinterpretq_u8_s16(row1_lz));
|
||||
uint8x16_t row23_lz = vuzp1q_u8(vreinterpretq_u8_s16(row2_lz),
|
||||
vreinterpretq_u8_s16(row3_lz));
|
||||
uint8x16_t row45_lz = vuzp1q_u8(vreinterpretq_u8_s16(row4_lz),
|
||||
vreinterpretq_u8_s16(row5_lz));
|
||||
uint8x16_t row67_lz = vuzp1q_u8(vreinterpretq_u8_s16(row6_lz),
|
||||
vreinterpretq_u8_s16(row7_lz));
|
||||
/* Compute nbits needed to specify magnitude of each coefficient. */
|
||||
uint8x16_t row01_nbits = vsubq_u8(vdupq_n_u8(16), row01_lz);
|
||||
uint8x16_t row23_nbits = vsubq_u8(vdupq_n_u8(16), row23_lz);
|
||||
uint8x16_t row45_nbits = vsubq_u8(vdupq_n_u8(16), row45_lz);
|
||||
uint8x16_t row67_nbits = vsubq_u8(vdupq_n_u8(16), row67_lz);
|
||||
/* Store nbits. */
|
||||
vst1q_u8(block_nbits + 0 * DCTSIZE, row01_nbits);
|
||||
vst1q_u8(block_nbits + 2 * DCTSIZE, row23_nbits);
|
||||
vst1q_u8(block_nbits + 4 * DCTSIZE, row45_nbits);
|
||||
vst1q_u8(block_nbits + 6 * DCTSIZE, row67_nbits);
|
||||
/* Mask bits not required to specify sign and amplitude of diff. */
|
||||
uint16x8_t row1_mask = vshlq_u16(vcltzq_s16(row1), vnegq_s16(row1_lz));
|
||||
uint16x8_t row2_mask = vshlq_u16(vcltzq_s16(row2), vnegq_s16(row2_lz));
|
||||
uint16x8_t row3_mask = vshlq_u16(vcltzq_s16(row3), vnegq_s16(row3_lz));
|
||||
uint16x8_t row4_mask = vshlq_u16(vcltzq_s16(row4), vnegq_s16(row4_lz));
|
||||
uint16x8_t row5_mask = vshlq_u16(vcltzq_s16(row5), vnegq_s16(row5_lz));
|
||||
uint16x8_t row6_mask = vshlq_u16(vcltzq_s16(row6), vnegq_s16(row6_lz));
|
||||
uint16x8_t row7_mask = vshlq_u16(vcltzq_s16(row7), vnegq_s16(row7_lz));
|
||||
/* diff = abs(coeff) ^ sign(coeff) [no-op for positive coefficients] */
|
||||
uint16x8_t row1_diff = veorq_u16(vreinterpretq_u16_s16(abs_row1),
|
||||
row1_mask);
|
||||
uint16x8_t row2_diff = veorq_u16(vreinterpretq_u16_s16(abs_row2),
|
||||
row2_mask);
|
||||
uint16x8_t row3_diff = veorq_u16(vreinterpretq_u16_s16(abs_row3),
|
||||
row3_mask);
|
||||
uint16x8_t row4_diff = veorq_u16(vreinterpretq_u16_s16(abs_row4),
|
||||
row4_mask);
|
||||
uint16x8_t row5_diff = veorq_u16(vreinterpretq_u16_s16(abs_row5),
|
||||
row5_mask);
|
||||
uint16x8_t row6_diff = veorq_u16(vreinterpretq_u16_s16(abs_row6),
|
||||
row6_mask);
|
||||
uint16x8_t row7_diff = veorq_u16(vreinterpretq_u16_s16(abs_row7),
|
||||
row7_mask);
|
||||
/* Store diff bits. */
|
||||
vst1q_u16(block_diff + 0 * DCTSIZE, row0_diff);
|
||||
vst1q_u16(block_diff + 1 * DCTSIZE, row1_diff);
|
||||
vst1q_u16(block_diff + 2 * DCTSIZE, row2_diff);
|
||||
vst1q_u16(block_diff + 3 * DCTSIZE, row3_diff);
|
||||
vst1q_u16(block_diff + 4 * DCTSIZE, row4_diff);
|
||||
vst1q_u16(block_diff + 5 * DCTSIZE, row5_diff);
|
||||
vst1q_u16(block_diff + 6 * DCTSIZE, row6_diff);
|
||||
vst1q_u16(block_diff + 7 * DCTSIZE, row7_diff);
|
||||
|
||||
while (bitmap != 0) {
|
||||
r = BUILTIN_CLZLL(bitmap);
|
||||
i += r;
|
||||
bitmap <<= r;
|
||||
nbits = block_nbits[i];
|
||||
diff = block_diff[i];
|
||||
while (r > 15) {
|
||||
/* If run length > 15, emit special run-length-16 codes. */
|
||||
PUT_BITS(code_0xf0, size_0xf0)
|
||||
r -= 16;
|
||||
}
|
||||
/* Emit Huffman symbol for run length / number of bits. (F.1.2.2.1) */
|
||||
unsigned int rs = (r << 4) + nbits;
|
||||
PUT_CODE(actbl->ehufco[rs], actbl->ehufsi[rs], diff)
|
||||
i++;
|
||||
bitmap <<= 1;
|
||||
}
|
||||
} else if (bitmap != 0) {
|
||||
uint16_t block_abs[DCTSIZE2];
|
||||
/* Compute and store absolute value of coefficients. */
|
||||
int16x8_t abs_row1 = vabsq_s16(row1);
|
||||
int16x8_t abs_row2 = vabsq_s16(row2);
|
||||
int16x8_t abs_row3 = vabsq_s16(row3);
|
||||
int16x8_t abs_row4 = vabsq_s16(row4);
|
||||
int16x8_t abs_row5 = vabsq_s16(row5);
|
||||
int16x8_t abs_row6 = vabsq_s16(row6);
|
||||
int16x8_t abs_row7 = vabsq_s16(row7);
|
||||
vst1q_u16(block_abs + 0 * DCTSIZE, vreinterpretq_u16_s16(abs_row0));
|
||||
vst1q_u16(block_abs + 1 * DCTSIZE, vreinterpretq_u16_s16(abs_row1));
|
||||
vst1q_u16(block_abs + 2 * DCTSIZE, vreinterpretq_u16_s16(abs_row2));
|
||||
vst1q_u16(block_abs + 3 * DCTSIZE, vreinterpretq_u16_s16(abs_row3));
|
||||
vst1q_u16(block_abs + 4 * DCTSIZE, vreinterpretq_u16_s16(abs_row4));
|
||||
vst1q_u16(block_abs + 5 * DCTSIZE, vreinterpretq_u16_s16(abs_row5));
|
||||
vst1q_u16(block_abs + 6 * DCTSIZE, vreinterpretq_u16_s16(abs_row6));
|
||||
vst1q_u16(block_abs + 7 * DCTSIZE, vreinterpretq_u16_s16(abs_row7));
|
||||
/* Compute diff bits (without nbits mask) and store. */
|
||||
uint16x8_t row1_diff = veorq_u16(vreinterpretq_u16_s16(abs_row1),
|
||||
vcltzq_s16(row1));
|
||||
uint16x8_t row2_diff = veorq_u16(vreinterpretq_u16_s16(abs_row2),
|
||||
vcltzq_s16(row2));
|
||||
uint16x8_t row3_diff = veorq_u16(vreinterpretq_u16_s16(abs_row3),
|
||||
vcltzq_s16(row3));
|
||||
uint16x8_t row4_diff = veorq_u16(vreinterpretq_u16_s16(abs_row4),
|
||||
vcltzq_s16(row4));
|
||||
uint16x8_t row5_diff = veorq_u16(vreinterpretq_u16_s16(abs_row5),
|
||||
vcltzq_s16(row5));
|
||||
uint16x8_t row6_diff = veorq_u16(vreinterpretq_u16_s16(abs_row6),
|
||||
vcltzq_s16(row6));
|
||||
uint16x8_t row7_diff = veorq_u16(vreinterpretq_u16_s16(abs_row7),
|
||||
vcltzq_s16(row7));
|
||||
vst1q_u16(block_diff + 0 * DCTSIZE, row0_diff);
|
||||
vst1q_u16(block_diff + 1 * DCTSIZE, row1_diff);
|
||||
vst1q_u16(block_diff + 2 * DCTSIZE, row2_diff);
|
||||
vst1q_u16(block_diff + 3 * DCTSIZE, row3_diff);
|
||||
vst1q_u16(block_diff + 4 * DCTSIZE, row4_diff);
|
||||
vst1q_u16(block_diff + 5 * DCTSIZE, row5_diff);
|
||||
vst1q_u16(block_diff + 6 * DCTSIZE, row6_diff);
|
||||
vst1q_u16(block_diff + 7 * DCTSIZE, row7_diff);
|
||||
|
||||
/* Same as above but must mask diff bits and compute nbits on demand. */
|
||||
while (bitmap != 0) {
|
||||
r = BUILTIN_CLZLL(bitmap);
|
||||
i += r;
|
||||
bitmap <<= r;
|
||||
lz = BUILTIN_CLZ(block_abs[i]);
|
||||
nbits = 32 - lz;
|
||||
diff = ((unsigned int)block_diff[i] << lz) >> lz;
|
||||
while (r > 15) {
|
||||
/* If run length > 15, emit special run-length-16 codes. */
|
||||
PUT_BITS(code_0xf0, size_0xf0)
|
||||
r -= 16;
|
||||
}
|
||||
/* Emit Huffman symbol for run length / number of bits. (F.1.2.2.1) */
|
||||
unsigned int rs = (r << 4) + nbits;
|
||||
PUT_CODE(actbl->ehufco[rs], actbl->ehufsi[rs], diff)
|
||||
i++;
|
||||
bitmap <<= 1;
|
||||
}
|
||||
}
|
||||
|
||||
/* If the last coefficient(s) were zero, emit an end-of-block (EOB) code.
|
||||
* The value of RS for the EOB code is 0.
|
||||
*/
|
||||
if (i != 64) {
|
||||
PUT_BITS(actbl->ehufco[0], actbl->ehufsi[0])
|
||||
}
|
||||
|
||||
state_ptr->cur.put_buffer = put_buffer;
|
||||
state_ptr->cur.free_bits = free_bits;
|
||||
|
||||
return buffer;
|
||||
}
|
||||
+1058
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+28
@@ -0,0 +1,28 @@
|
||||
/*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
/* How to obtain memory alignment for structures and variables */
|
||||
#if defined(_MSC_VER)
|
||||
#define ALIGN(alignment) __declspec(align(alignment))
|
||||
#elif defined(__clang__) || defined(__GNUC__)
|
||||
#define ALIGN(alignment) __attribute__((aligned(alignment)))
|
||||
#else
|
||||
#error "Unknown compiler"
|
||||
#endif
|
||||
+160
@@ -0,0 +1,160 @@
|
||||
/*
|
||||
* jccolor-neon.c - colorspace conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* RGB -> YCbCr conversion constants */
|
||||
|
||||
#define F_0_298 19595
|
||||
#define F_0_587 38470
|
||||
#define F_0_113 7471
|
||||
#define F_0_168 11059
|
||||
#define F_0_331 21709
|
||||
#define F_0_500 32768
|
||||
#define F_0_418 27439
|
||||
#define F_0_081 5329
|
||||
|
||||
ALIGN(16) static const uint16_t jsimd_rgb_ycc_neon_consts[] = {
|
||||
F_0_298, F_0_587, F_0_113, F_0_168,
|
||||
F_0_331, F_0_500, F_0_418, F_0_081
|
||||
};
|
||||
|
||||
|
||||
/* Include inline routines for colorspace extensions. */
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
|
||||
#define RGB_RED EXT_RGB_RED
|
||||
#define RGB_GREEN EXT_RGB_GREEN
|
||||
#define RGB_BLUE EXT_RGB_BLUE
|
||||
#define RGB_PIXELSIZE EXT_RGB_PIXELSIZE
|
||||
#define jsimd_rgb_ycc_convert_neon jsimd_extrgb_ycc_convert_neon
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_ycc_convert_neon
|
||||
|
||||
#define RGB_RED EXT_RGBX_RED
|
||||
#define RGB_GREEN EXT_RGBX_GREEN
|
||||
#define RGB_BLUE EXT_RGBX_BLUE
|
||||
#define RGB_PIXELSIZE EXT_RGBX_PIXELSIZE
|
||||
#define jsimd_rgb_ycc_convert_neon jsimd_extrgbx_ycc_convert_neon
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_ycc_convert_neon
|
||||
|
||||
#define RGB_RED EXT_BGR_RED
|
||||
#define RGB_GREEN EXT_BGR_GREEN
|
||||
#define RGB_BLUE EXT_BGR_BLUE
|
||||
#define RGB_PIXELSIZE EXT_BGR_PIXELSIZE
|
||||
#define jsimd_rgb_ycc_convert_neon jsimd_extbgr_ycc_convert_neon
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_ycc_convert_neon
|
||||
|
||||
#define RGB_RED EXT_BGRX_RED
|
||||
#define RGB_GREEN EXT_BGRX_GREEN
|
||||
#define RGB_BLUE EXT_BGRX_BLUE
|
||||
#define RGB_PIXELSIZE EXT_BGRX_PIXELSIZE
|
||||
#define jsimd_rgb_ycc_convert_neon jsimd_extbgrx_ycc_convert_neon
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_ycc_convert_neon
|
||||
|
||||
#define RGB_RED EXT_XBGR_RED
|
||||
#define RGB_GREEN EXT_XBGR_GREEN
|
||||
#define RGB_BLUE EXT_XBGR_BLUE
|
||||
#define RGB_PIXELSIZE EXT_XBGR_PIXELSIZE
|
||||
#define jsimd_rgb_ycc_convert_neon jsimd_extxbgr_ycc_convert_neon
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_ycc_convert_neon
|
||||
|
||||
#define RGB_RED EXT_XRGB_RED
|
||||
#define RGB_GREEN EXT_XRGB_GREEN
|
||||
#define RGB_BLUE EXT_XRGB_BLUE
|
||||
#define RGB_PIXELSIZE EXT_XRGB_PIXELSIZE
|
||||
#define jsimd_rgb_ycc_convert_neon jsimd_extxrgb_ycc_convert_neon
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#include "aarch64/jccolext-neon.c"
|
||||
#else
|
||||
#include "aarch32/jccolext-neon.c"
|
||||
#endif
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_ycc_convert_neon
|
||||
+120
@@ -0,0 +1,120 @@
|
||||
/*
|
||||
* jcgray-neon.c - grayscale colorspace conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* RGB -> Grayscale conversion constants */
|
||||
|
||||
#define F_0_298 19595
|
||||
#define F_0_587 38470
|
||||
#define F_0_113 7471
|
||||
|
||||
|
||||
/* Include inline routines for colorspace extensions. */
|
||||
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
|
||||
#define RGB_RED EXT_RGB_RED
|
||||
#define RGB_GREEN EXT_RGB_GREEN
|
||||
#define RGB_BLUE EXT_RGB_BLUE
|
||||
#define RGB_PIXELSIZE EXT_RGB_PIXELSIZE
|
||||
#define jsimd_rgb_gray_convert_neon jsimd_extrgb_gray_convert_neon
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_gray_convert_neon
|
||||
|
||||
#define RGB_RED EXT_RGBX_RED
|
||||
#define RGB_GREEN EXT_RGBX_GREEN
|
||||
#define RGB_BLUE EXT_RGBX_BLUE
|
||||
#define RGB_PIXELSIZE EXT_RGBX_PIXELSIZE
|
||||
#define jsimd_rgb_gray_convert_neon jsimd_extrgbx_gray_convert_neon
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_gray_convert_neon
|
||||
|
||||
#define RGB_RED EXT_BGR_RED
|
||||
#define RGB_GREEN EXT_BGR_GREEN
|
||||
#define RGB_BLUE EXT_BGR_BLUE
|
||||
#define RGB_PIXELSIZE EXT_BGR_PIXELSIZE
|
||||
#define jsimd_rgb_gray_convert_neon jsimd_extbgr_gray_convert_neon
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_gray_convert_neon
|
||||
|
||||
#define RGB_RED EXT_BGRX_RED
|
||||
#define RGB_GREEN EXT_BGRX_GREEN
|
||||
#define RGB_BLUE EXT_BGRX_BLUE
|
||||
#define RGB_PIXELSIZE EXT_BGRX_PIXELSIZE
|
||||
#define jsimd_rgb_gray_convert_neon jsimd_extbgrx_gray_convert_neon
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_gray_convert_neon
|
||||
|
||||
#define RGB_RED EXT_XBGR_RED
|
||||
#define RGB_GREEN EXT_XBGR_GREEN
|
||||
#define RGB_BLUE EXT_XBGR_BLUE
|
||||
#define RGB_PIXELSIZE EXT_XBGR_PIXELSIZE
|
||||
#define jsimd_rgb_gray_convert_neon jsimd_extxbgr_gray_convert_neon
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_gray_convert_neon
|
||||
|
||||
#define RGB_RED EXT_XRGB_RED
|
||||
#define RGB_GREEN EXT_XRGB_GREEN
|
||||
#define RGB_BLUE EXT_XRGB_BLUE
|
||||
#define RGB_PIXELSIZE EXT_XRGB_PIXELSIZE
|
||||
#define jsimd_rgb_gray_convert_neon jsimd_extxrgb_gray_convert_neon
|
||||
#include "jcgryext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_rgb_gray_convert_neon
|
||||
+106
@@ -0,0 +1,106 @@
|
||||
/*
|
||||
* jcgryext-neon.c - grayscale colorspace conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
/* This file is included by jcgray-neon.c */
|
||||
|
||||
|
||||
/* RGB -> Grayscale conversion is defined by the following equation:
|
||||
* Y = 0.29900 * R + 0.58700 * G + 0.11400 * B
|
||||
*
|
||||
* Avoid floating point arithmetic by using shifted integer constants:
|
||||
* 0.29899597 = 19595 * 2^-16
|
||||
* 0.58700561 = 38470 * 2^-16
|
||||
* 0.11399841 = 7471 * 2^-16
|
||||
* These constants are defined in jcgray-neon.c
|
||||
*
|
||||
* This is the same computation as the RGB -> Y portion of RGB -> YCbCr.
|
||||
*/
|
||||
|
||||
void jsimd_rgb_gray_convert_neon(JDIMENSION image_width, JSAMPARRAY input_buf,
|
||||
JSAMPIMAGE output_buf, JDIMENSION output_row,
|
||||
int num_rows)
|
||||
{
|
||||
JSAMPROW inptr;
|
||||
JSAMPROW outptr;
|
||||
/* Allocate temporary buffer for final (image_width % 16) pixels in row. */
|
||||
ALIGN(16) uint8_t tmp_buf[16 * RGB_PIXELSIZE];
|
||||
|
||||
while (--num_rows >= 0) {
|
||||
inptr = *input_buf++;
|
||||
outptr = output_buf[0][output_row];
|
||||
output_row++;
|
||||
|
||||
int cols_remaining = image_width;
|
||||
for (; cols_remaining > 0; cols_remaining -= 16) {
|
||||
|
||||
/* To prevent buffer overread by the vector load instructions, the last
|
||||
* (image_width % 16) columns of data are first memcopied to a temporary
|
||||
* buffer large enough to accommodate the vector load.
|
||||
*/
|
||||
if (cols_remaining < 16) {
|
||||
memcpy(tmp_buf, inptr, cols_remaining * RGB_PIXELSIZE);
|
||||
inptr = tmp_buf;
|
||||
}
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x16x4_t input_pixels = vld4q_u8(inptr);
|
||||
#else
|
||||
uint8x16x3_t input_pixels = vld3q_u8(inptr);
|
||||
#endif
|
||||
uint16x8_t r_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_RED]));
|
||||
uint16x8_t r_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_RED]));
|
||||
uint16x8_t g_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_GREEN]));
|
||||
uint16x8_t g_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_GREEN]));
|
||||
uint16x8_t b_l = vmovl_u8(vget_low_u8(input_pixels.val[RGB_BLUE]));
|
||||
uint16x8_t b_h = vmovl_u8(vget_high_u8(input_pixels.val[RGB_BLUE]));
|
||||
|
||||
/* Compute Y = 0.29900 * R + 0.58700 * G + 0.11400 * B */
|
||||
uint32x4_t y_ll = vmull_n_u16(vget_low_u16(r_l), F_0_298);
|
||||
uint32x4_t y_lh = vmull_n_u16(vget_high_u16(r_l), F_0_298);
|
||||
uint32x4_t y_hl = vmull_n_u16(vget_low_u16(r_h), F_0_298);
|
||||
uint32x4_t y_hh = vmull_n_u16(vget_high_u16(r_h), F_0_298);
|
||||
y_ll = vmlal_n_u16(y_ll, vget_low_u16(g_l), F_0_587);
|
||||
y_lh = vmlal_n_u16(y_lh, vget_high_u16(g_l), F_0_587);
|
||||
y_hl = vmlal_n_u16(y_hl, vget_low_u16(g_h), F_0_587);
|
||||
y_hh = vmlal_n_u16(y_hh, vget_high_u16(g_h), F_0_587);
|
||||
y_ll = vmlal_n_u16(y_ll, vget_low_u16(b_l), F_0_113);
|
||||
y_lh = vmlal_n_u16(y_lh, vget_high_u16(b_l), F_0_113);
|
||||
y_hl = vmlal_n_u16(y_hl, vget_low_u16(b_h), F_0_113);
|
||||
y_hh = vmlal_n_u16(y_hh, vget_high_u16(b_h), F_0_113);
|
||||
|
||||
/* Descale Y values (rounding right shift) and narrow to 16-bit. */
|
||||
uint16x8_t y_l = vcombine_u16(vrshrn_n_u32(y_ll, 16),
|
||||
vrshrn_n_u32(y_lh, 16));
|
||||
uint16x8_t y_h = vcombine_u16(vrshrn_n_u32(y_hl, 16),
|
||||
vrshrn_n_u32(y_hh, 16));
|
||||
|
||||
/* Narrow Y values to 8-bit and store to memory. Buffer overwrite is
|
||||
* permitted up to the next multiple of ALIGN_SIZE bytes.
|
||||
*/
|
||||
vst1q_u8(outptr, vcombine_u8(vmovn_u16(y_l), vmovn_u16(y_h)));
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr += (16 * RGB_PIXELSIZE);
|
||||
outptr += 16;
|
||||
}
|
||||
}
|
||||
}
|
||||
+131
@@ -0,0 +1,131 @@
|
||||
/*
|
||||
* jchuff.h
|
||||
*
|
||||
* This file was part of the Independent JPEG Group's software:
|
||||
* Copyright (C) 1991-1997, Thomas G. Lane.
|
||||
* libjpeg-turbo Modifications:
|
||||
* Copyright (C) 2009, 2018, 2021, D. R. Commander.
|
||||
* Copyright (C) 2018, Matthias Räncker.
|
||||
* Copyright (C) 2020-2021, Arm Limited.
|
||||
* For conditions of distribution and use, see the accompanying README.ijg
|
||||
* file.
|
||||
*/
|
||||
|
||||
/* Expanded entropy encoder object for Huffman encoding.
|
||||
*
|
||||
* The savable_state subrecord contains fields that change within an MCU,
|
||||
* but must not be updated permanently until we complete the MCU.
|
||||
*/
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
#define BIT_BUF_SIZE 64
|
||||
#else
|
||||
#define BIT_BUF_SIZE 32
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
size_t put_buffer; /* current bit accumulation buffer */
|
||||
int free_bits; /* # of bits available in it */
|
||||
int last_dc_val[MAX_COMPS_IN_SCAN]; /* last DC coef for each component */
|
||||
} savable_state;
|
||||
|
||||
typedef struct {
|
||||
JOCTET *next_output_byte; /* => next byte to write in buffer */
|
||||
size_t free_in_buffer; /* # of byte spaces remaining in buffer */
|
||||
savable_state cur; /* Current bit buffer & DC state */
|
||||
j_compress_ptr cinfo; /* dump_buffer needs access to this */
|
||||
int simd;
|
||||
} working_state;
|
||||
|
||||
/* Outputting bits to the file */
|
||||
|
||||
/* Output byte b and, speculatively, an additional 0 byte. 0xFF must be encoded
|
||||
* as 0xFF 0x00, so the output buffer pointer is advanced by 2 if the byte is
|
||||
* 0xFF. Otherwise, the output buffer pointer is advanced by 1, and the
|
||||
* speculative 0 byte will be overwritten by the next byte.
|
||||
*/
|
||||
#define EMIT_BYTE(b) { \
|
||||
buffer[0] = (JOCTET)(b); \
|
||||
buffer[1] = 0; \
|
||||
buffer -= -2 + ((JOCTET)(b) < 0xFF); \
|
||||
}
|
||||
|
||||
/* Output the entire bit buffer. If there are no 0xFF bytes in it, then write
|
||||
* directly to the output buffer. Otherwise, use the EMIT_BYTE() macro to
|
||||
* encode 0xFF as 0xFF 0x00.
|
||||
*/
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
|
||||
#define FLUSH() { \
|
||||
if (put_buffer & 0x8080808080808080 & ~(put_buffer + 0x0101010101010101)) { \
|
||||
EMIT_BYTE(put_buffer >> 56) \
|
||||
EMIT_BYTE(put_buffer >> 48) \
|
||||
EMIT_BYTE(put_buffer >> 40) \
|
||||
EMIT_BYTE(put_buffer >> 32) \
|
||||
EMIT_BYTE(put_buffer >> 24) \
|
||||
EMIT_BYTE(put_buffer >> 16) \
|
||||
EMIT_BYTE(put_buffer >> 8) \
|
||||
EMIT_BYTE(put_buffer ) \
|
||||
} else { \
|
||||
*((uint64_t *)buffer) = BUILTIN_BSWAP64(put_buffer); \
|
||||
buffer += 8; \
|
||||
} \
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#define SPLAT() { \
|
||||
buffer[0] = (JOCTET)(put_buffer >> 24); \
|
||||
buffer[1] = (JOCTET)(put_buffer >> 16); \
|
||||
buffer[2] = (JOCTET)(put_buffer >> 8); \
|
||||
buffer[3] = (JOCTET)(put_buffer ); \
|
||||
buffer += 4; \
|
||||
}
|
||||
#else
|
||||
#define SPLAT() { \
|
||||
put_buffer = __builtin_bswap32(put_buffer); \
|
||||
__asm__("str %1, [%0], #4" : "+r" (buffer) : "r" (put_buffer)); \
|
||||
}
|
||||
#endif
|
||||
|
||||
#define FLUSH() { \
|
||||
if (put_buffer & 0x80808080 & ~(put_buffer + 0x01010101)) { \
|
||||
EMIT_BYTE(put_buffer >> 24) \
|
||||
EMIT_BYTE(put_buffer >> 16) \
|
||||
EMIT_BYTE(put_buffer >> 8) \
|
||||
EMIT_BYTE(put_buffer ) \
|
||||
} else { \
|
||||
SPLAT(); \
|
||||
} \
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
/* Fill the bit buffer to capacity with the leading bits from code, then output
|
||||
* the bit buffer and put the remaining bits from code into the bit buffer.
|
||||
*/
|
||||
#define PUT_AND_FLUSH(code, size) { \
|
||||
put_buffer = (put_buffer << (size + free_bits)) | (code >> -free_bits); \
|
||||
FLUSH() \
|
||||
free_bits += BIT_BUF_SIZE; \
|
||||
put_buffer = code; \
|
||||
}
|
||||
|
||||
/* Insert code into the bit buffer and output the bit buffer if needed.
|
||||
* NOTE: We can't flush with free_bits == 0, since the left shift in
|
||||
* PUT_AND_FLUSH() would have undefined behavior.
|
||||
*/
|
||||
#define PUT_BITS(code, size) { \
|
||||
free_bits -= size; \
|
||||
if (free_bits < 0) \
|
||||
PUT_AND_FLUSH(code, size) \
|
||||
else \
|
||||
put_buffer = (put_buffer << size) | code; \
|
||||
}
|
||||
|
||||
#define PUT_CODE(code, size, diff) { \
|
||||
diff |= code << nbits; \
|
||||
nbits += size; \
|
||||
PUT_BITS(diff, nbits) \
|
||||
}
|
||||
+622
@@ -0,0 +1,622 @@
|
||||
/*
|
||||
* jcphuff-neon.c - prepare data for progressive Huffman encoding (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020-2021, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "jconfigint.h"
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* Data preparation for encode_mcu_AC_first().
|
||||
*
|
||||
* The equivalent scalar C function (encode_mcu_AC_first_prepare()) can be
|
||||
* found in jcphuff.c.
|
||||
*/
|
||||
|
||||
void jsimd_encode_mcu_AC_first_prepare_neon
|
||||
(const JCOEF *block, const int *jpeg_natural_order_start, int Sl, int Al,
|
||||
JCOEF *values, size_t *zerobits)
|
||||
{
|
||||
JCOEF *values_ptr = values;
|
||||
JCOEF *diff_values_ptr = values + DCTSIZE2;
|
||||
|
||||
/* Rows of coefficients to zero (since they haven't been processed) */
|
||||
int i, rows_to_zero = 8;
|
||||
|
||||
for (i = 0; i < Sl / 16; i++) {
|
||||
int16x8_t coefs1 = vld1q_dup_s16(block + jpeg_natural_order_start[0]);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[1], coefs1, 1);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[2], coefs1, 2);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[3], coefs1, 3);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[4], coefs1, 4);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[5], coefs1, 5);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[6], coefs1, 6);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[7], coefs1, 7);
|
||||
int16x8_t coefs2 = vld1q_dup_s16(block + jpeg_natural_order_start[8]);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[9], coefs2, 1);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[10], coefs2, 2);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[11], coefs2, 3);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[12], coefs2, 4);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[13], coefs2, 5);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[14], coefs2, 6);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[15], coefs2, 7);
|
||||
|
||||
/* Isolate sign of coefficients. */
|
||||
int16x8_t sign_coefs1 = vshrq_n_s16(coefs1, 15);
|
||||
int16x8_t sign_coefs2 = vshrq_n_s16(coefs2, 15);
|
||||
/* Compute absolute value of coefficients and apply point transform Al. */
|
||||
int16x8_t abs_coefs1 = vabsq_s16(coefs1);
|
||||
int16x8_t abs_coefs2 = vabsq_s16(coefs2);
|
||||
coefs1 = vshlq_s16(abs_coefs1, vdupq_n_s16(-Al));
|
||||
coefs2 = vshlq_s16(abs_coefs2, vdupq_n_s16(-Al));
|
||||
|
||||
/* Compute diff values. */
|
||||
int16x8_t diff1 = veorq_s16(coefs1, sign_coefs1);
|
||||
int16x8_t diff2 = veorq_s16(coefs2, sign_coefs2);
|
||||
|
||||
/* Store transformed coefficients and diff values. */
|
||||
vst1q_s16(values_ptr, coefs1);
|
||||
vst1q_s16(values_ptr + DCTSIZE, coefs2);
|
||||
vst1q_s16(diff_values_ptr, diff1);
|
||||
vst1q_s16(diff_values_ptr + DCTSIZE, diff2);
|
||||
values_ptr += 16;
|
||||
diff_values_ptr += 16;
|
||||
jpeg_natural_order_start += 16;
|
||||
rows_to_zero -= 2;
|
||||
}
|
||||
|
||||
/* Same operation but for remaining partial vector */
|
||||
int remaining_coefs = Sl % 16;
|
||||
if (remaining_coefs > 8) {
|
||||
int16x8_t coefs1 = vld1q_dup_s16(block + jpeg_natural_order_start[0]);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[1], coefs1, 1);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[2], coefs1, 2);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[3], coefs1, 3);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[4], coefs1, 4);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[5], coefs1, 5);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[6], coefs1, 6);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[7], coefs1, 7);
|
||||
int16x8_t coefs2 = vdupq_n_s16(0);
|
||||
switch (remaining_coefs) {
|
||||
case 15:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[14], coefs2, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 14:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[13], coefs2, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 13:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[12], coefs2, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 12:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[11], coefs2, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 11:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[10], coefs2, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 10:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[9], coefs2, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 9:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[8], coefs2, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
/* Isolate sign of coefficients. */
|
||||
int16x8_t sign_coefs1 = vshrq_n_s16(coefs1, 15);
|
||||
int16x8_t sign_coefs2 = vshrq_n_s16(coefs2, 15);
|
||||
/* Compute absolute value of coefficients and apply point transform Al. */
|
||||
int16x8_t abs_coefs1 = vabsq_s16(coefs1);
|
||||
int16x8_t abs_coefs2 = vabsq_s16(coefs2);
|
||||
coefs1 = vshlq_s16(abs_coefs1, vdupq_n_s16(-Al));
|
||||
coefs2 = vshlq_s16(abs_coefs2, vdupq_n_s16(-Al));
|
||||
|
||||
/* Compute diff values. */
|
||||
int16x8_t diff1 = veorq_s16(coefs1, sign_coefs1);
|
||||
int16x8_t diff2 = veorq_s16(coefs2, sign_coefs2);
|
||||
|
||||
/* Store transformed coefficients and diff values. */
|
||||
vst1q_s16(values_ptr, coefs1);
|
||||
vst1q_s16(values_ptr + DCTSIZE, coefs2);
|
||||
vst1q_s16(diff_values_ptr, diff1);
|
||||
vst1q_s16(diff_values_ptr + DCTSIZE, diff2);
|
||||
values_ptr += 16;
|
||||
diff_values_ptr += 16;
|
||||
rows_to_zero -= 2;
|
||||
|
||||
} else if (remaining_coefs > 0) {
|
||||
int16x8_t coefs = vdupq_n_s16(0);
|
||||
|
||||
switch (remaining_coefs) {
|
||||
case 8:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[7], coefs, 7);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 7:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[6], coefs, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[5], coefs, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[4], coefs, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[3], coefs, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[2], coefs, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[1], coefs, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[0], coefs, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
/* Isolate sign of coefficients. */
|
||||
int16x8_t sign_coefs = vshrq_n_s16(coefs, 15);
|
||||
/* Compute absolute value of coefficients and apply point transform Al. */
|
||||
int16x8_t abs_coefs = vabsq_s16(coefs);
|
||||
coefs = vshlq_s16(abs_coefs, vdupq_n_s16(-Al));
|
||||
|
||||
/* Compute diff values. */
|
||||
int16x8_t diff = veorq_s16(coefs, sign_coefs);
|
||||
|
||||
/* Store transformed coefficients and diff values. */
|
||||
vst1q_s16(values_ptr, coefs);
|
||||
vst1q_s16(diff_values_ptr, diff);
|
||||
values_ptr += 8;
|
||||
diff_values_ptr += 8;
|
||||
rows_to_zero--;
|
||||
}
|
||||
|
||||
/* Zero remaining memory in the values and diff_values blocks. */
|
||||
for (i = 0; i < rows_to_zero; i++) {
|
||||
vst1q_s16(values_ptr, vdupq_n_s16(0));
|
||||
vst1q_s16(diff_values_ptr, vdupq_n_s16(0));
|
||||
values_ptr += 8;
|
||||
diff_values_ptr += 8;
|
||||
}
|
||||
|
||||
/* Construct zerobits bitmap. A set bit means that the corresponding
|
||||
* coefficient != 0.
|
||||
*/
|
||||
int16x8_t row0 = vld1q_s16(values + 0 * DCTSIZE);
|
||||
int16x8_t row1 = vld1q_s16(values + 1 * DCTSIZE);
|
||||
int16x8_t row2 = vld1q_s16(values + 2 * DCTSIZE);
|
||||
int16x8_t row3 = vld1q_s16(values + 3 * DCTSIZE);
|
||||
int16x8_t row4 = vld1q_s16(values + 4 * DCTSIZE);
|
||||
int16x8_t row5 = vld1q_s16(values + 5 * DCTSIZE);
|
||||
int16x8_t row6 = vld1q_s16(values + 6 * DCTSIZE);
|
||||
int16x8_t row7 = vld1q_s16(values + 7 * DCTSIZE);
|
||||
|
||||
uint8x8_t row0_eq0 = vmovn_u16(vceqq_s16(row0, vdupq_n_s16(0)));
|
||||
uint8x8_t row1_eq0 = vmovn_u16(vceqq_s16(row1, vdupq_n_s16(0)));
|
||||
uint8x8_t row2_eq0 = vmovn_u16(vceqq_s16(row2, vdupq_n_s16(0)));
|
||||
uint8x8_t row3_eq0 = vmovn_u16(vceqq_s16(row3, vdupq_n_s16(0)));
|
||||
uint8x8_t row4_eq0 = vmovn_u16(vceqq_s16(row4, vdupq_n_s16(0)));
|
||||
uint8x8_t row5_eq0 = vmovn_u16(vceqq_s16(row5, vdupq_n_s16(0)));
|
||||
uint8x8_t row6_eq0 = vmovn_u16(vceqq_s16(row6, vdupq_n_s16(0)));
|
||||
uint8x8_t row7_eq0 = vmovn_u16(vceqq_s16(row7, vdupq_n_s16(0)));
|
||||
|
||||
/* { 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80 } */
|
||||
const uint8x8_t bitmap_mask =
|
||||
vreinterpret_u8_u64(vmov_n_u64(0x8040201008040201));
|
||||
|
||||
row0_eq0 = vand_u8(row0_eq0, bitmap_mask);
|
||||
row1_eq0 = vand_u8(row1_eq0, bitmap_mask);
|
||||
row2_eq0 = vand_u8(row2_eq0, bitmap_mask);
|
||||
row3_eq0 = vand_u8(row3_eq0, bitmap_mask);
|
||||
row4_eq0 = vand_u8(row4_eq0, bitmap_mask);
|
||||
row5_eq0 = vand_u8(row5_eq0, bitmap_mask);
|
||||
row6_eq0 = vand_u8(row6_eq0, bitmap_mask);
|
||||
row7_eq0 = vand_u8(row7_eq0, bitmap_mask);
|
||||
|
||||
uint8x8_t bitmap_rows_01 = vpadd_u8(row0_eq0, row1_eq0);
|
||||
uint8x8_t bitmap_rows_23 = vpadd_u8(row2_eq0, row3_eq0);
|
||||
uint8x8_t bitmap_rows_45 = vpadd_u8(row4_eq0, row5_eq0);
|
||||
uint8x8_t bitmap_rows_67 = vpadd_u8(row6_eq0, row7_eq0);
|
||||
uint8x8_t bitmap_rows_0123 = vpadd_u8(bitmap_rows_01, bitmap_rows_23);
|
||||
uint8x8_t bitmap_rows_4567 = vpadd_u8(bitmap_rows_45, bitmap_rows_67);
|
||||
uint8x8_t bitmap_all = vpadd_u8(bitmap_rows_0123, bitmap_rows_4567);
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
/* Move bitmap to a 64-bit scalar register. */
|
||||
uint64_t bitmap = vget_lane_u64(vreinterpret_u64_u8(bitmap_all), 0);
|
||||
/* Store zerobits bitmap. */
|
||||
*zerobits = ~bitmap;
|
||||
#else
|
||||
/* Move bitmap to two 32-bit scalar registers. */
|
||||
uint32_t bitmap0 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 0);
|
||||
uint32_t bitmap1 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 1);
|
||||
/* Store zerobits bitmap. */
|
||||
zerobits[0] = ~bitmap0;
|
||||
zerobits[1] = ~bitmap1;
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
/* Data preparation for encode_mcu_AC_refine().
|
||||
*
|
||||
* The equivalent scalar C function (encode_mcu_AC_refine_prepare()) can be
|
||||
* found in jcphuff.c.
|
||||
*/
|
||||
|
||||
int jsimd_encode_mcu_AC_refine_prepare_neon
|
||||
(const JCOEF *block, const int *jpeg_natural_order_start, int Sl, int Al,
|
||||
JCOEF *absvalues, size_t *bits)
|
||||
{
|
||||
/* Temporary storage buffers for data used to compute the signbits bitmap and
|
||||
* the end-of-block (EOB) position
|
||||
*/
|
||||
uint8_t coef_sign_bits[64];
|
||||
uint8_t coef_eq1_bits[64];
|
||||
|
||||
JCOEF *absvalues_ptr = absvalues;
|
||||
uint8_t *coef_sign_bits_ptr = coef_sign_bits;
|
||||
uint8_t *eq1_bits_ptr = coef_eq1_bits;
|
||||
|
||||
/* Rows of coefficients to zero (since they haven't been processed) */
|
||||
int i, rows_to_zero = 8;
|
||||
|
||||
for (i = 0; i < Sl / 16; i++) {
|
||||
int16x8_t coefs1 = vld1q_dup_s16(block + jpeg_natural_order_start[0]);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[1], coefs1, 1);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[2], coefs1, 2);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[3], coefs1, 3);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[4], coefs1, 4);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[5], coefs1, 5);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[6], coefs1, 6);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[7], coefs1, 7);
|
||||
int16x8_t coefs2 = vld1q_dup_s16(block + jpeg_natural_order_start[8]);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[9], coefs2, 1);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[10], coefs2, 2);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[11], coefs2, 3);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[12], coefs2, 4);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[13], coefs2, 5);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[14], coefs2, 6);
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[15], coefs2, 7);
|
||||
|
||||
/* Compute and store data for signbits bitmap. */
|
||||
uint8x8_t sign_coefs1 =
|
||||
vmovn_u16(vreinterpretq_u16_s16(vshrq_n_s16(coefs1, 15)));
|
||||
uint8x8_t sign_coefs2 =
|
||||
vmovn_u16(vreinterpretq_u16_s16(vshrq_n_s16(coefs2, 15)));
|
||||
vst1_u8(coef_sign_bits_ptr, sign_coefs1);
|
||||
vst1_u8(coef_sign_bits_ptr + DCTSIZE, sign_coefs2);
|
||||
|
||||
/* Compute absolute value of coefficients and apply point transform Al. */
|
||||
int16x8_t abs_coefs1 = vabsq_s16(coefs1);
|
||||
int16x8_t abs_coefs2 = vabsq_s16(coefs2);
|
||||
coefs1 = vshlq_s16(abs_coefs1, vdupq_n_s16(-Al));
|
||||
coefs2 = vshlq_s16(abs_coefs2, vdupq_n_s16(-Al));
|
||||
vst1q_s16(absvalues_ptr, coefs1);
|
||||
vst1q_s16(absvalues_ptr + DCTSIZE, coefs2);
|
||||
|
||||
/* Test whether transformed coefficient values == 1 (used to find EOB
|
||||
* position.)
|
||||
*/
|
||||
uint8x8_t coefs_eq11 = vmovn_u16(vceqq_s16(coefs1, vdupq_n_s16(1)));
|
||||
uint8x8_t coefs_eq12 = vmovn_u16(vceqq_s16(coefs2, vdupq_n_s16(1)));
|
||||
vst1_u8(eq1_bits_ptr, coefs_eq11);
|
||||
vst1_u8(eq1_bits_ptr + DCTSIZE, coefs_eq12);
|
||||
|
||||
absvalues_ptr += 16;
|
||||
coef_sign_bits_ptr += 16;
|
||||
eq1_bits_ptr += 16;
|
||||
jpeg_natural_order_start += 16;
|
||||
rows_to_zero -= 2;
|
||||
}
|
||||
|
||||
/* Same operation but for remaining partial vector */
|
||||
int remaining_coefs = Sl % 16;
|
||||
if (remaining_coefs > 8) {
|
||||
int16x8_t coefs1 = vld1q_dup_s16(block + jpeg_natural_order_start[0]);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[1], coefs1, 1);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[2], coefs1, 2);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[3], coefs1, 3);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[4], coefs1, 4);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[5], coefs1, 5);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[6], coefs1, 6);
|
||||
coefs1 = vld1q_lane_s16(block + jpeg_natural_order_start[7], coefs1, 7);
|
||||
int16x8_t coefs2 = vdupq_n_s16(0);
|
||||
switch (remaining_coefs) {
|
||||
case 15:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[14], coefs2, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 14:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[13], coefs2, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 13:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[12], coefs2, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 12:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[11], coefs2, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 11:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[10], coefs2, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 10:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[9], coefs2, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 9:
|
||||
coefs2 = vld1q_lane_s16(block + jpeg_natural_order_start[8], coefs2, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
/* Compute and store data for signbits bitmap. */
|
||||
uint8x8_t sign_coefs1 =
|
||||
vmovn_u16(vreinterpretq_u16_s16(vshrq_n_s16(coefs1, 15)));
|
||||
uint8x8_t sign_coefs2 =
|
||||
vmovn_u16(vreinterpretq_u16_s16(vshrq_n_s16(coefs2, 15)));
|
||||
vst1_u8(coef_sign_bits_ptr, sign_coefs1);
|
||||
vst1_u8(coef_sign_bits_ptr + DCTSIZE, sign_coefs2);
|
||||
|
||||
/* Compute absolute value of coefficients and apply point transform Al. */
|
||||
int16x8_t abs_coefs1 = vabsq_s16(coefs1);
|
||||
int16x8_t abs_coefs2 = vabsq_s16(coefs2);
|
||||
coefs1 = vshlq_s16(abs_coefs1, vdupq_n_s16(-Al));
|
||||
coefs2 = vshlq_s16(abs_coefs2, vdupq_n_s16(-Al));
|
||||
vst1q_s16(absvalues_ptr, coefs1);
|
||||
vst1q_s16(absvalues_ptr + DCTSIZE, coefs2);
|
||||
|
||||
/* Test whether transformed coefficient values == 1 (used to find EOB
|
||||
* position.)
|
||||
*/
|
||||
uint8x8_t coefs_eq11 = vmovn_u16(vceqq_s16(coefs1, vdupq_n_s16(1)));
|
||||
uint8x8_t coefs_eq12 = vmovn_u16(vceqq_s16(coefs2, vdupq_n_s16(1)));
|
||||
vst1_u8(eq1_bits_ptr, coefs_eq11);
|
||||
vst1_u8(eq1_bits_ptr + DCTSIZE, coefs_eq12);
|
||||
|
||||
absvalues_ptr += 16;
|
||||
coef_sign_bits_ptr += 16;
|
||||
eq1_bits_ptr += 16;
|
||||
jpeg_natural_order_start += 16;
|
||||
rows_to_zero -= 2;
|
||||
|
||||
} else if (remaining_coefs > 0) {
|
||||
int16x8_t coefs = vdupq_n_s16(0);
|
||||
|
||||
switch (remaining_coefs) {
|
||||
case 8:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[7], coefs, 7);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 7:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[6], coefs, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[5], coefs, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[4], coefs, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[3], coefs, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[2], coefs, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[1], coefs, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
coefs = vld1q_lane_s16(block + jpeg_natural_order_start[0], coefs, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
/* Compute and store data for signbits bitmap. */
|
||||
uint8x8_t sign_coefs =
|
||||
vmovn_u16(vreinterpretq_u16_s16(vshrq_n_s16(coefs, 15)));
|
||||
vst1_u8(coef_sign_bits_ptr, sign_coefs);
|
||||
|
||||
/* Compute absolute value of coefficients and apply point transform Al. */
|
||||
int16x8_t abs_coefs = vabsq_s16(coefs);
|
||||
coefs = vshlq_s16(abs_coefs, vdupq_n_s16(-Al));
|
||||
vst1q_s16(absvalues_ptr, coefs);
|
||||
|
||||
/* Test whether transformed coefficient values == 1 (used to find EOB
|
||||
* position.)
|
||||
*/
|
||||
uint8x8_t coefs_eq1 = vmovn_u16(vceqq_s16(coefs, vdupq_n_s16(1)));
|
||||
vst1_u8(eq1_bits_ptr, coefs_eq1);
|
||||
|
||||
absvalues_ptr += 8;
|
||||
coef_sign_bits_ptr += 8;
|
||||
eq1_bits_ptr += 8;
|
||||
rows_to_zero--;
|
||||
}
|
||||
|
||||
/* Zero remaining memory in blocks. */
|
||||
for (i = 0; i < rows_to_zero; i++) {
|
||||
vst1q_s16(absvalues_ptr, vdupq_n_s16(0));
|
||||
vst1_u8(coef_sign_bits_ptr, vdup_n_u8(0));
|
||||
vst1_u8(eq1_bits_ptr, vdup_n_u8(0));
|
||||
absvalues_ptr += 8;
|
||||
coef_sign_bits_ptr += 8;
|
||||
eq1_bits_ptr += 8;
|
||||
}
|
||||
|
||||
/* Construct zerobits bitmap. */
|
||||
int16x8_t abs_row0 = vld1q_s16(absvalues + 0 * DCTSIZE);
|
||||
int16x8_t abs_row1 = vld1q_s16(absvalues + 1 * DCTSIZE);
|
||||
int16x8_t abs_row2 = vld1q_s16(absvalues + 2 * DCTSIZE);
|
||||
int16x8_t abs_row3 = vld1q_s16(absvalues + 3 * DCTSIZE);
|
||||
int16x8_t abs_row4 = vld1q_s16(absvalues + 4 * DCTSIZE);
|
||||
int16x8_t abs_row5 = vld1q_s16(absvalues + 5 * DCTSIZE);
|
||||
int16x8_t abs_row6 = vld1q_s16(absvalues + 6 * DCTSIZE);
|
||||
int16x8_t abs_row7 = vld1q_s16(absvalues + 7 * DCTSIZE);
|
||||
|
||||
uint8x8_t abs_row0_eq0 = vmovn_u16(vceqq_s16(abs_row0, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row1_eq0 = vmovn_u16(vceqq_s16(abs_row1, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row2_eq0 = vmovn_u16(vceqq_s16(abs_row2, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row3_eq0 = vmovn_u16(vceqq_s16(abs_row3, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row4_eq0 = vmovn_u16(vceqq_s16(abs_row4, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row5_eq0 = vmovn_u16(vceqq_s16(abs_row5, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row6_eq0 = vmovn_u16(vceqq_s16(abs_row6, vdupq_n_s16(0)));
|
||||
uint8x8_t abs_row7_eq0 = vmovn_u16(vceqq_s16(abs_row7, vdupq_n_s16(0)));
|
||||
|
||||
/* { 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80 } */
|
||||
const uint8x8_t bitmap_mask =
|
||||
vreinterpret_u8_u64(vmov_n_u64(0x8040201008040201));
|
||||
|
||||
abs_row0_eq0 = vand_u8(abs_row0_eq0, bitmap_mask);
|
||||
abs_row1_eq0 = vand_u8(abs_row1_eq0, bitmap_mask);
|
||||
abs_row2_eq0 = vand_u8(abs_row2_eq0, bitmap_mask);
|
||||
abs_row3_eq0 = vand_u8(abs_row3_eq0, bitmap_mask);
|
||||
abs_row4_eq0 = vand_u8(abs_row4_eq0, bitmap_mask);
|
||||
abs_row5_eq0 = vand_u8(abs_row5_eq0, bitmap_mask);
|
||||
abs_row6_eq0 = vand_u8(abs_row6_eq0, bitmap_mask);
|
||||
abs_row7_eq0 = vand_u8(abs_row7_eq0, bitmap_mask);
|
||||
|
||||
uint8x8_t bitmap_rows_01 = vpadd_u8(abs_row0_eq0, abs_row1_eq0);
|
||||
uint8x8_t bitmap_rows_23 = vpadd_u8(abs_row2_eq0, abs_row3_eq0);
|
||||
uint8x8_t bitmap_rows_45 = vpadd_u8(abs_row4_eq0, abs_row5_eq0);
|
||||
uint8x8_t bitmap_rows_67 = vpadd_u8(abs_row6_eq0, abs_row7_eq0);
|
||||
uint8x8_t bitmap_rows_0123 = vpadd_u8(bitmap_rows_01, bitmap_rows_23);
|
||||
uint8x8_t bitmap_rows_4567 = vpadd_u8(bitmap_rows_45, bitmap_rows_67);
|
||||
uint8x8_t bitmap_all = vpadd_u8(bitmap_rows_0123, bitmap_rows_4567);
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
/* Move bitmap to a 64-bit scalar register. */
|
||||
uint64_t bitmap = vget_lane_u64(vreinterpret_u64_u8(bitmap_all), 0);
|
||||
/* Store zerobits bitmap. */
|
||||
bits[0] = ~bitmap;
|
||||
#else
|
||||
/* Move bitmap to two 32-bit scalar registers. */
|
||||
uint32_t bitmap0 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 0);
|
||||
uint32_t bitmap1 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 1);
|
||||
/* Store zerobits bitmap. */
|
||||
bits[0] = ~bitmap0;
|
||||
bits[1] = ~bitmap1;
|
||||
#endif
|
||||
|
||||
/* Construct signbits bitmap. */
|
||||
uint8x8_t signbits_row0 = vld1_u8(coef_sign_bits + 0 * DCTSIZE);
|
||||
uint8x8_t signbits_row1 = vld1_u8(coef_sign_bits + 1 * DCTSIZE);
|
||||
uint8x8_t signbits_row2 = vld1_u8(coef_sign_bits + 2 * DCTSIZE);
|
||||
uint8x8_t signbits_row3 = vld1_u8(coef_sign_bits + 3 * DCTSIZE);
|
||||
uint8x8_t signbits_row4 = vld1_u8(coef_sign_bits + 4 * DCTSIZE);
|
||||
uint8x8_t signbits_row5 = vld1_u8(coef_sign_bits + 5 * DCTSIZE);
|
||||
uint8x8_t signbits_row6 = vld1_u8(coef_sign_bits + 6 * DCTSIZE);
|
||||
uint8x8_t signbits_row7 = vld1_u8(coef_sign_bits + 7 * DCTSIZE);
|
||||
|
||||
signbits_row0 = vand_u8(signbits_row0, bitmap_mask);
|
||||
signbits_row1 = vand_u8(signbits_row1, bitmap_mask);
|
||||
signbits_row2 = vand_u8(signbits_row2, bitmap_mask);
|
||||
signbits_row3 = vand_u8(signbits_row3, bitmap_mask);
|
||||
signbits_row4 = vand_u8(signbits_row4, bitmap_mask);
|
||||
signbits_row5 = vand_u8(signbits_row5, bitmap_mask);
|
||||
signbits_row6 = vand_u8(signbits_row6, bitmap_mask);
|
||||
signbits_row7 = vand_u8(signbits_row7, bitmap_mask);
|
||||
|
||||
bitmap_rows_01 = vpadd_u8(signbits_row0, signbits_row1);
|
||||
bitmap_rows_23 = vpadd_u8(signbits_row2, signbits_row3);
|
||||
bitmap_rows_45 = vpadd_u8(signbits_row4, signbits_row5);
|
||||
bitmap_rows_67 = vpadd_u8(signbits_row6, signbits_row7);
|
||||
bitmap_rows_0123 = vpadd_u8(bitmap_rows_01, bitmap_rows_23);
|
||||
bitmap_rows_4567 = vpadd_u8(bitmap_rows_45, bitmap_rows_67);
|
||||
bitmap_all = vpadd_u8(bitmap_rows_0123, bitmap_rows_4567);
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
/* Move bitmap to a 64-bit scalar register. */
|
||||
bitmap = vget_lane_u64(vreinterpret_u64_u8(bitmap_all), 0);
|
||||
/* Store signbits bitmap. */
|
||||
bits[1] = ~bitmap;
|
||||
#else
|
||||
/* Move bitmap to two 32-bit scalar registers. */
|
||||
bitmap0 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 0);
|
||||
bitmap1 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 1);
|
||||
/* Store signbits bitmap. */
|
||||
bits[2] = ~bitmap0;
|
||||
bits[3] = ~bitmap1;
|
||||
#endif
|
||||
|
||||
/* Construct bitmap to find EOB position (the index of the last coefficient
|
||||
* equal to 1.)
|
||||
*/
|
||||
uint8x8_t row0_eq1 = vld1_u8(coef_eq1_bits + 0 * DCTSIZE);
|
||||
uint8x8_t row1_eq1 = vld1_u8(coef_eq1_bits + 1 * DCTSIZE);
|
||||
uint8x8_t row2_eq1 = vld1_u8(coef_eq1_bits + 2 * DCTSIZE);
|
||||
uint8x8_t row3_eq1 = vld1_u8(coef_eq1_bits + 3 * DCTSIZE);
|
||||
uint8x8_t row4_eq1 = vld1_u8(coef_eq1_bits + 4 * DCTSIZE);
|
||||
uint8x8_t row5_eq1 = vld1_u8(coef_eq1_bits + 5 * DCTSIZE);
|
||||
uint8x8_t row6_eq1 = vld1_u8(coef_eq1_bits + 6 * DCTSIZE);
|
||||
uint8x8_t row7_eq1 = vld1_u8(coef_eq1_bits + 7 * DCTSIZE);
|
||||
|
||||
row0_eq1 = vand_u8(row0_eq1, bitmap_mask);
|
||||
row1_eq1 = vand_u8(row1_eq1, bitmap_mask);
|
||||
row2_eq1 = vand_u8(row2_eq1, bitmap_mask);
|
||||
row3_eq1 = vand_u8(row3_eq1, bitmap_mask);
|
||||
row4_eq1 = vand_u8(row4_eq1, bitmap_mask);
|
||||
row5_eq1 = vand_u8(row5_eq1, bitmap_mask);
|
||||
row6_eq1 = vand_u8(row6_eq1, bitmap_mask);
|
||||
row7_eq1 = vand_u8(row7_eq1, bitmap_mask);
|
||||
|
||||
bitmap_rows_01 = vpadd_u8(row0_eq1, row1_eq1);
|
||||
bitmap_rows_23 = vpadd_u8(row2_eq1, row3_eq1);
|
||||
bitmap_rows_45 = vpadd_u8(row4_eq1, row5_eq1);
|
||||
bitmap_rows_67 = vpadd_u8(row6_eq1, row7_eq1);
|
||||
bitmap_rows_0123 = vpadd_u8(bitmap_rows_01, bitmap_rows_23);
|
||||
bitmap_rows_4567 = vpadd_u8(bitmap_rows_45, bitmap_rows_67);
|
||||
bitmap_all = vpadd_u8(bitmap_rows_0123, bitmap_rows_4567);
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
/* Move bitmap to a 64-bit scalar register. */
|
||||
bitmap = vget_lane_u64(vreinterpret_u64_u8(bitmap_all), 0);
|
||||
|
||||
/* Return EOB position. */
|
||||
if (bitmap == 0) {
|
||||
/* EOB position is defined to be 0 if all coefficients != 1. */
|
||||
return 0;
|
||||
} else {
|
||||
return 63 - BUILTIN_CLZLL(bitmap);
|
||||
}
|
||||
#else
|
||||
/* Move bitmap to two 32-bit scalar registers. */
|
||||
bitmap0 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 0);
|
||||
bitmap1 = vget_lane_u32(vreinterpret_u32_u8(bitmap_all), 1);
|
||||
|
||||
/* Return EOB position. */
|
||||
if (bitmap0 == 0 && bitmap1 == 0) {
|
||||
return 0;
|
||||
} else if (bitmap1 != 0) {
|
||||
return 63 - BUILTIN_CLZ(bitmap1);
|
||||
} else {
|
||||
return 31 - BUILTIN_CLZ(bitmap0);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
+192
@@ -0,0 +1,192 @@
|
||||
/*
|
||||
* jcsample-neon.c - downsampling (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
ALIGN(16) static const uint8_t jsimd_h2_downsample_consts[] = {
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 0 */
|
||||
0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 1 */
|
||||
0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0E,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 2 */
|
||||
0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0D, 0x0D,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 3 */
|
||||
0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0C, 0x0C, 0x0C,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 4 */
|
||||
0x08, 0x09, 0x0A, 0x0B, 0x0B, 0x0B, 0x0B, 0x0B,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 5 */
|
||||
0x08, 0x09, 0x0A, 0x0A, 0x0A, 0x0A, 0x0A, 0x0A,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 6 */
|
||||
0x08, 0x09, 0x09, 0x09, 0x09, 0x09, 0x09, 0x09,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 7 */
|
||||
0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08, 0x08,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, /* Pad 8 */
|
||||
0x07, 0x07, 0x07, 0x07, 0x07, 0x07, 0x07, 0x07,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x06, /* Pad 9 */
|
||||
0x06, 0x06, 0x06, 0x06, 0x06, 0x06, 0x06, 0x06,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x05, 0x05, /* Pad 10 */
|
||||
0x05, 0x05, 0x05, 0x05, 0x05, 0x05, 0x05, 0x05,
|
||||
0x00, 0x01, 0x02, 0x03, 0x04, 0x04, 0x04, 0x04, /* Pad 11 */
|
||||
0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04, 0x04,
|
||||
0x00, 0x01, 0x02, 0x03, 0x03, 0x03, 0x03, 0x03, /* Pad 12 */
|
||||
0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03,
|
||||
0x00, 0x01, 0x02, 0x02, 0x02, 0x02, 0x02, 0x02, /* Pad 13 */
|
||||
0x02, 0x02, 0x02, 0x02, 0x02, 0x02, 0x02, 0x02,
|
||||
0x00, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, /* Pad 14 */
|
||||
0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01,
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* Pad 15 */
|
||||
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00
|
||||
};
|
||||
|
||||
|
||||
/* Downsample pixel values of a single component.
|
||||
* This version handles the common case of 2:1 horizontal and 1:1 vertical,
|
||||
* without smoothing.
|
||||
*/
|
||||
|
||||
void jsimd_h2v1_downsample_neon(JDIMENSION image_width, int max_v_samp_factor,
|
||||
JDIMENSION v_samp_factor,
|
||||
JDIMENSION width_in_blocks,
|
||||
JSAMPARRAY input_data, JSAMPARRAY output_data)
|
||||
{
|
||||
JSAMPROW inptr, outptr;
|
||||
/* Load expansion mask to pad remaining elements of last DCT block. */
|
||||
const int mask_offset = 16 * ((width_in_blocks * 2 * DCTSIZE) - image_width);
|
||||
const uint8x16_t expand_mask =
|
||||
vld1q_u8(&jsimd_h2_downsample_consts[mask_offset]);
|
||||
/* Load bias pattern (alternating every pixel.) */
|
||||
/* { 0, 1, 0, 1, 0, 1, 0, 1 } */
|
||||
const uint16x8_t bias = vreinterpretq_u16_u32(vdupq_n_u32(0x00010000));
|
||||
unsigned i, outrow;
|
||||
|
||||
for (outrow = 0; outrow < v_samp_factor; outrow++) {
|
||||
outptr = output_data[outrow];
|
||||
inptr = input_data[outrow];
|
||||
|
||||
/* Downsample all but the last DCT block of pixels. */
|
||||
for (i = 0; i < width_in_blocks - 1; i++) {
|
||||
uint8x16_t pixels = vld1q_u8(inptr + i * 2 * DCTSIZE);
|
||||
/* Add adjacent pixel values, widen to 16-bit, and add bias. */
|
||||
uint16x8_t samples_u16 = vpadalq_u8(bias, pixels);
|
||||
/* Divide total by 2 and narrow to 8-bit. */
|
||||
uint8x8_t samples_u8 = vshrn_n_u16(samples_u16, 1);
|
||||
/* Store samples to memory. */
|
||||
vst1_u8(outptr + i * DCTSIZE, samples_u8);
|
||||
}
|
||||
|
||||
/* Load pixels in last DCT block into a table. */
|
||||
uint8x16_t pixels = vld1q_u8(inptr + (width_in_blocks - 1) * 2 * DCTSIZE);
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
/* Pad the empty elements with the value of the last pixel. */
|
||||
pixels = vqtbl1q_u8(pixels, expand_mask);
|
||||
#else
|
||||
uint8x8x2_t table = { { vget_low_u8(pixels), vget_high_u8(pixels) } };
|
||||
pixels = vcombine_u8(vtbl2_u8(table, vget_low_u8(expand_mask)),
|
||||
vtbl2_u8(table, vget_high_u8(expand_mask)));
|
||||
#endif
|
||||
/* Add adjacent pixel values, widen to 16-bit, and add bias. */
|
||||
uint16x8_t samples_u16 = vpadalq_u8(bias, pixels);
|
||||
/* Divide total by 2, narrow to 8-bit, and store. */
|
||||
uint8x8_t samples_u8 = vshrn_n_u16(samples_u16, 1);
|
||||
vst1_u8(outptr + (width_in_blocks - 1) * DCTSIZE, samples_u8);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* Downsample pixel values of a single component.
|
||||
* This version handles the standard case of 2:1 horizontal and 2:1 vertical,
|
||||
* without smoothing.
|
||||
*/
|
||||
|
||||
void jsimd_h2v2_downsample_neon(JDIMENSION image_width, int max_v_samp_factor,
|
||||
JDIMENSION v_samp_factor,
|
||||
JDIMENSION width_in_blocks,
|
||||
JSAMPARRAY input_data, JSAMPARRAY output_data)
|
||||
{
|
||||
JSAMPROW inptr0, inptr1, outptr;
|
||||
/* Load expansion mask to pad remaining elements of last DCT block. */
|
||||
const int mask_offset = 16 * ((width_in_blocks * 2 * DCTSIZE) - image_width);
|
||||
const uint8x16_t expand_mask =
|
||||
vld1q_u8(&jsimd_h2_downsample_consts[mask_offset]);
|
||||
/* Load bias pattern (alternating every pixel.) */
|
||||
/* { 1, 2, 1, 2, 1, 2, 1, 2 } */
|
||||
const uint16x8_t bias = vreinterpretq_u16_u32(vdupq_n_u32(0x00020001));
|
||||
unsigned i, outrow;
|
||||
|
||||
for (outrow = 0; outrow < v_samp_factor; outrow++) {
|
||||
outptr = output_data[outrow];
|
||||
inptr0 = input_data[outrow];
|
||||
inptr1 = input_data[outrow + 1];
|
||||
|
||||
/* Downsample all but the last DCT block of pixels. */
|
||||
for (i = 0; i < width_in_blocks - 1; i++) {
|
||||
uint8x16_t pixels_r0 = vld1q_u8(inptr0 + i * 2 * DCTSIZE);
|
||||
uint8x16_t pixels_r1 = vld1q_u8(inptr1 + i * 2 * DCTSIZE);
|
||||
/* Add adjacent pixel values in row 0, widen to 16-bit, and add bias. */
|
||||
uint16x8_t samples_u16 = vpadalq_u8(bias, pixels_r0);
|
||||
/* Add adjacent pixel values in row 1, widen to 16-bit, and accumulate.
|
||||
*/
|
||||
samples_u16 = vpadalq_u8(samples_u16, pixels_r1);
|
||||
/* Divide total by 4 and narrow to 8-bit. */
|
||||
uint8x8_t samples_u8 = vshrn_n_u16(samples_u16, 2);
|
||||
/* Store samples to memory and increment pointers. */
|
||||
vst1_u8(outptr + i * DCTSIZE, samples_u8);
|
||||
}
|
||||
|
||||
/* Load pixels in last DCT block into a table. */
|
||||
uint8x16_t pixels_r0 =
|
||||
vld1q_u8(inptr0 + (width_in_blocks - 1) * 2 * DCTSIZE);
|
||||
uint8x16_t pixels_r1 =
|
||||
vld1q_u8(inptr1 + (width_in_blocks - 1) * 2 * DCTSIZE);
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
/* Pad the empty elements with the value of the last pixel. */
|
||||
pixels_r0 = vqtbl1q_u8(pixels_r0, expand_mask);
|
||||
pixels_r1 = vqtbl1q_u8(pixels_r1, expand_mask);
|
||||
#else
|
||||
uint8x8x2_t table_r0 =
|
||||
{ { vget_low_u8(pixels_r0), vget_high_u8(pixels_r0) } };
|
||||
uint8x8x2_t table_r1 =
|
||||
{ { vget_low_u8(pixels_r1), vget_high_u8(pixels_r1) } };
|
||||
pixels_r0 = vcombine_u8(vtbl2_u8(table_r0, vget_low_u8(expand_mask)),
|
||||
vtbl2_u8(table_r0, vget_high_u8(expand_mask)));
|
||||
pixels_r1 = vcombine_u8(vtbl2_u8(table_r1, vget_low_u8(expand_mask)),
|
||||
vtbl2_u8(table_r1, vget_high_u8(expand_mask)));
|
||||
#endif
|
||||
/* Add adjacent pixel values in row 0, widen to 16-bit, and add bias. */
|
||||
uint16x8_t samples_u16 = vpadalq_u8(bias, pixels_r0);
|
||||
/* Add adjacent pixel values in row 1, widen to 16-bit, and accumulate. */
|
||||
samples_u16 = vpadalq_u8(samples_u16, pixels_r1);
|
||||
/* Divide total by 4, narrow to 8-bit, and store. */
|
||||
uint8x8_t samples_u8 = vshrn_n_u16(samples_u16, 2);
|
||||
vst1_u8(outptr + (width_in_blocks - 1) * DCTSIZE, samples_u8);
|
||||
}
|
||||
}
|
||||
+374
@@ -0,0 +1,374 @@
|
||||
/*
|
||||
* jdcolext-neon.c - colorspace conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
/* This file is included by jdcolor-neon.c. */
|
||||
|
||||
|
||||
/* YCbCr -> RGB conversion is defined by the following equations:
|
||||
* R = Y + 1.40200 * (Cr - 128)
|
||||
* G = Y - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128)
|
||||
* B = Y + 1.77200 * (Cb - 128)
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.3441467 = 11277 * 2^-15
|
||||
* 0.7141418 = 23401 * 2^-15
|
||||
* 1.4020386 = 22971 * 2^-14
|
||||
* 1.7720337 = 29033 * 2^-14
|
||||
* These constants are defined in jdcolor-neon.c.
|
||||
*
|
||||
* To ensure correct results, rounding is used when descaling.
|
||||
*/
|
||||
|
||||
/* Notes on safe memory access for YCbCr -> RGB conversion routines:
|
||||
*
|
||||
* Input memory buffers can be safely overread up to the next multiple of
|
||||
* ALIGN_SIZE bytes, since they are always allocated by alloc_sarray() in
|
||||
* jmemmgr.c.
|
||||
*
|
||||
* The output buffer cannot safely be written beyond output_width, since
|
||||
* output_buf points to a possibly unpadded row in the decompressed image
|
||||
* buffer allocated by the calling program.
|
||||
*/
|
||||
|
||||
void jsimd_ycc_rgb_convert_neon(JDIMENSION output_width, JSAMPIMAGE input_buf,
|
||||
JDIMENSION input_row, JSAMPARRAY output_buf,
|
||||
int num_rows)
|
||||
{
|
||||
JSAMPROW outptr;
|
||||
/* Pointers to Y, Cb, and Cr data */
|
||||
JSAMPROW inptr0, inptr1, inptr2;
|
||||
|
||||
const int16x4_t consts = vld1_s16(jsimd_ycc_rgb_convert_neon_consts);
|
||||
const int16x8_t neg_128 = vdupq_n_s16(-128);
|
||||
|
||||
while (--num_rows >= 0) {
|
||||
inptr0 = input_buf[0][input_row];
|
||||
inptr1 = input_buf[1][input_row];
|
||||
inptr2 = input_buf[2][input_row];
|
||||
input_row++;
|
||||
outptr = *output_buf++;
|
||||
int cols_remaining = output_width;
|
||||
for (; cols_remaining >= 16; cols_remaining -= 16) {
|
||||
uint8x16_t y = vld1q_u8(inptr0);
|
||||
uint8x16_t cb = vld1q_u8(inptr1);
|
||||
uint8x16_t cr = vld1q_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128_l =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128),
|
||||
vget_low_u8(cr)));
|
||||
int16x8_t cr_128_h =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128),
|
||||
vget_high_u8(cr)));
|
||||
int16x8_t cb_128_l =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128),
|
||||
vget_low_u8(cb)));
|
||||
int16x8_t cb_128_h =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128),
|
||||
vget_high_u8(cb)));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_ll = vmull_lane_s16(vget_low_s16(cb_128_l), consts, 0);
|
||||
int32x4_t g_sub_y_lh = vmull_lane_s16(vget_high_s16(cb_128_l),
|
||||
consts, 0);
|
||||
int32x4_t g_sub_y_hl = vmull_lane_s16(vget_low_s16(cb_128_h), consts, 0);
|
||||
int32x4_t g_sub_y_hh = vmull_lane_s16(vget_high_s16(cb_128_h),
|
||||
consts, 0);
|
||||
g_sub_y_ll = vmlsl_lane_s16(g_sub_y_ll, vget_low_s16(cr_128_l),
|
||||
consts, 1);
|
||||
g_sub_y_lh = vmlsl_lane_s16(g_sub_y_lh, vget_high_s16(cr_128_l),
|
||||
consts, 1);
|
||||
g_sub_y_hl = vmlsl_lane_s16(g_sub_y_hl, vget_low_s16(cr_128_h),
|
||||
consts, 1);
|
||||
g_sub_y_hh = vmlsl_lane_s16(g_sub_y_hh, vget_high_s16(cr_128_h),
|
||||
consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y_l = vcombine_s16(vrshrn_n_s32(g_sub_y_ll, 15),
|
||||
vrshrn_n_s32(g_sub_y_lh, 15));
|
||||
int16x8_t g_sub_y_h = vcombine_s16(vrshrn_n_s32(g_sub_y_hl, 15),
|
||||
vrshrn_n_s32(g_sub_y_hh, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y_l = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128_l, 1),
|
||||
consts, 2);
|
||||
int16x8_t r_sub_y_h = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128_h, 1),
|
||||
consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y_l = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128_l, 1),
|
||||
consts, 3);
|
||||
int16x8_t b_sub_y_h = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128_h, 1),
|
||||
consts, 3);
|
||||
/* Add Y. */
|
||||
int16x8_t r_l =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y_l),
|
||||
vget_low_u8(y)));
|
||||
int16x8_t r_h =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y_h),
|
||||
vget_high_u8(y)));
|
||||
int16x8_t b_l =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y_l),
|
||||
vget_low_u8(y)));
|
||||
int16x8_t b_h =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y_h),
|
||||
vget_high_u8(y)));
|
||||
int16x8_t g_l =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y_l),
|
||||
vget_low_u8(y)));
|
||||
int16x8_t g_h =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y_h),
|
||||
vget_high_u8(y)));
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x16x4_t rgba;
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255]. */
|
||||
rgba.val[RGB_RED] = vcombine_u8(vqmovun_s16(r_l), vqmovun_s16(r_h));
|
||||
rgba.val[RGB_GREEN] = vcombine_u8(vqmovun_s16(g_l), vqmovun_s16(g_h));
|
||||
rgba.val[RGB_BLUE] = vcombine_u8(vqmovun_s16(b_l), vqmovun_s16(b_h));
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba.val[RGB_ALPHA] = vdupq_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
vst4q_u8(outptr, rgba);
|
||||
#elif RGB_PIXELSIZE == 3
|
||||
uint8x16x3_t rgb;
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255]. */
|
||||
rgb.val[RGB_RED] = vcombine_u8(vqmovun_s16(r_l), vqmovun_s16(r_h));
|
||||
rgb.val[RGB_GREEN] = vcombine_u8(vqmovun_s16(g_l), vqmovun_s16(g_h));
|
||||
rgb.val[RGB_BLUE] = vcombine_u8(vqmovun_s16(b_l), vqmovun_s16(b_h));
|
||||
/* Store RGB pixel data to memory. */
|
||||
vst3q_u8(outptr, rgb);
|
||||
#else
|
||||
/* Pack R, G, and B values in ratio 5:6:5. */
|
||||
uint16x8_t rgb565_l = vqshluq_n_s16(r_l, 8);
|
||||
rgb565_l = vsriq_n_u16(rgb565_l, vqshluq_n_s16(g_l, 8), 5);
|
||||
rgb565_l = vsriq_n_u16(rgb565_l, vqshluq_n_s16(b_l, 8), 11);
|
||||
uint16x8_t rgb565_h = vqshluq_n_s16(r_h, 8);
|
||||
rgb565_h = vsriq_n_u16(rgb565_h, vqshluq_n_s16(g_h, 8), 5);
|
||||
rgb565_h = vsriq_n_u16(rgb565_h, vqshluq_n_s16(b_h, 8), 11);
|
||||
/* Store RGB pixel data to memory. */
|
||||
vst1q_u16((uint16_t *)outptr, rgb565_l);
|
||||
vst1q_u16(((uint16_t *)outptr) + 8, rgb565_h);
|
||||
#endif
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr0 += 16;
|
||||
inptr1 += 16;
|
||||
inptr2 += 16;
|
||||
outptr += (RGB_PIXELSIZE * 16);
|
||||
}
|
||||
|
||||
if (cols_remaining >= 8) {
|
||||
uint8x8_t y = vld1_u8(inptr0);
|
||||
uint8x8_t cb = vld1_u8(inptr1);
|
||||
uint8x8_t cr = vld1_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cr));
|
||||
int16x8_t cb_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cb));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_l = vmull_lane_s16(vget_low_s16(cb_128), consts, 0);
|
||||
int32x4_t g_sub_y_h = vmull_lane_s16(vget_high_s16(cb_128), consts, 0);
|
||||
g_sub_y_l = vmlsl_lane_s16(g_sub_y_l, vget_low_s16(cr_128), consts, 1);
|
||||
g_sub_y_h = vmlsl_lane_s16(g_sub_y_h, vget_high_s16(cr_128), consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y = vcombine_s16(vrshrn_n_s32(g_sub_y_l, 15),
|
||||
vrshrn_n_s32(g_sub_y_h, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128, 1),
|
||||
consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128, 1),
|
||||
consts, 3);
|
||||
/* Add Y. */
|
||||
int16x8_t r =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y), y));
|
||||
int16x8_t b =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y), y));
|
||||
int16x8_t g =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y), y));
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x8x4_t rgba;
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255]. */
|
||||
rgba.val[RGB_RED] = vqmovun_s16(r);
|
||||
rgba.val[RGB_GREEN] = vqmovun_s16(g);
|
||||
rgba.val[RGB_BLUE] = vqmovun_s16(b);
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
vst4_u8(outptr, rgba);
|
||||
#elif RGB_PIXELSIZE == 3
|
||||
uint8x8x3_t rgb;
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255]. */
|
||||
rgb.val[RGB_RED] = vqmovun_s16(r);
|
||||
rgb.val[RGB_GREEN] = vqmovun_s16(g);
|
||||
rgb.val[RGB_BLUE] = vqmovun_s16(b);
|
||||
/* Store RGB pixel data to memory. */
|
||||
vst3_u8(outptr, rgb);
|
||||
#else
|
||||
/* Pack R, G, and B values in ratio 5:6:5. */
|
||||
uint16x8_t rgb565 = vqshluq_n_s16(r, 8);
|
||||
rgb565 = vsriq_n_u16(rgb565, vqshluq_n_s16(g, 8), 5);
|
||||
rgb565 = vsriq_n_u16(rgb565, vqshluq_n_s16(b, 8), 11);
|
||||
/* Store RGB pixel data to memory. */
|
||||
vst1q_u16((uint16_t *)outptr, rgb565);
|
||||
#endif
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr0 += 8;
|
||||
inptr1 += 8;
|
||||
inptr2 += 8;
|
||||
outptr += (RGB_PIXELSIZE * 8);
|
||||
cols_remaining -= 8;
|
||||
}
|
||||
|
||||
/* Handle the tail elements. */
|
||||
if (cols_remaining > 0) {
|
||||
uint8x8_t y = vld1_u8(inptr0);
|
||||
uint8x8_t cb = vld1_u8(inptr1);
|
||||
uint8x8_t cr = vld1_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cr));
|
||||
int16x8_t cb_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cb));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_l = vmull_lane_s16(vget_low_s16(cb_128), consts, 0);
|
||||
int32x4_t g_sub_y_h = vmull_lane_s16(vget_high_s16(cb_128), consts, 0);
|
||||
g_sub_y_l = vmlsl_lane_s16(g_sub_y_l, vget_low_s16(cr_128), consts, 1);
|
||||
g_sub_y_h = vmlsl_lane_s16(g_sub_y_h, vget_high_s16(cr_128), consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y = vcombine_s16(vrshrn_n_s32(g_sub_y_l, 15),
|
||||
vrshrn_n_s32(g_sub_y_h, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128, 1),
|
||||
consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128, 1),
|
||||
consts, 3);
|
||||
/* Add Y. */
|
||||
int16x8_t r =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y), y));
|
||||
int16x8_t b =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y), y));
|
||||
int16x8_t g =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y), y));
|
||||
|
||||
#if RGB_PIXELSIZE == 4
|
||||
uint8x8x4_t rgba;
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255]. */
|
||||
rgba.val[RGB_RED] = vqmovun_s16(r);
|
||||
rgba.val[RGB_GREEN] = vqmovun_s16(g);
|
||||
rgba.val[RGB_BLUE] = vqmovun_s16(b);
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 7:
|
||||
vst4_lane_u8(outptr + 6 * RGB_PIXELSIZE, rgba, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst4_lane_u8(outptr + 5 * RGB_PIXELSIZE, rgba, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst4_lane_u8(outptr + 4 * RGB_PIXELSIZE, rgba, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst4_lane_u8(outptr + 3 * RGB_PIXELSIZE, rgba, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst4_lane_u8(outptr + 2 * RGB_PIXELSIZE, rgba, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst4_lane_u8(outptr + RGB_PIXELSIZE, rgba, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst4_lane_u8(outptr, rgba, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#elif RGB_PIXELSIZE == 3
|
||||
uint8x8x3_t rgb;
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255]. */
|
||||
rgb.val[RGB_RED] = vqmovun_s16(r);
|
||||
rgb.val[RGB_GREEN] = vqmovun_s16(g);
|
||||
rgb.val[RGB_BLUE] = vqmovun_s16(b);
|
||||
/* Store RGB pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 7:
|
||||
vst3_lane_u8(outptr + 6 * RGB_PIXELSIZE, rgb, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst3_lane_u8(outptr + 5 * RGB_PIXELSIZE, rgb, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst3_lane_u8(outptr + 4 * RGB_PIXELSIZE, rgb, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst3_lane_u8(outptr + 3 * RGB_PIXELSIZE, rgb, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst3_lane_u8(outptr + 2 * RGB_PIXELSIZE, rgb, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst3_lane_u8(outptr + RGB_PIXELSIZE, rgb, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst3_lane_u8(outptr, rgb, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#else
|
||||
/* Pack R, G, and B values in ratio 5:6:5. */
|
||||
uint16x8_t rgb565 = vqshluq_n_s16(r, 8);
|
||||
rgb565 = vsriq_n_u16(rgb565, vqshluq_n_s16(g, 8), 5);
|
||||
rgb565 = vsriq_n_u16(rgb565, vqshluq_n_s16(b, 8), 11);
|
||||
/* Store RGB565 pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 7:
|
||||
vst1q_lane_u16((uint16_t *)(outptr + 6 * RGB_PIXELSIZE), rgb565, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst1q_lane_u16((uint16_t *)(outptr + 5 * RGB_PIXELSIZE), rgb565, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst1q_lane_u16((uint16_t *)(outptr + 4 * RGB_PIXELSIZE), rgb565, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst1q_lane_u16((uint16_t *)(outptr + 3 * RGB_PIXELSIZE), rgb565, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst1q_lane_u16((uint16_t *)(outptr + 2 * RGB_PIXELSIZE), rgb565, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst1q_lane_u16((uint16_t *)(outptr + RGB_PIXELSIZE), rgb565, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst1q_lane_u16((uint16_t *)outptr, rgb565, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
+142
@@ -0,0 +1,142 @@
|
||||
/*
|
||||
* jdcolor-neon.c - colorspace conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "jconfigint.h"
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* YCbCr -> RGB conversion constants */
|
||||
|
||||
#define F_0_344 11277 /* 0.3441467 = 11277 * 2^-15 */
|
||||
#define F_0_714 23401 /* 0.7141418 = 23401 * 2^-15 */
|
||||
#define F_1_402 22971 /* 1.4020386 = 22971 * 2^-14 */
|
||||
#define F_1_772 29033 /* 1.7720337 = 29033 * 2^-14 */
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_ycc_rgb_convert_neon_consts[] = {
|
||||
-F_0_344, F_0_714, F_1_402, F_1_772
|
||||
};
|
||||
|
||||
|
||||
/* Include inline routines for colorspace extensions. */
|
||||
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
|
||||
#define RGB_RED EXT_RGB_RED
|
||||
#define RGB_GREEN EXT_RGB_GREEN
|
||||
#define RGB_BLUE EXT_RGB_BLUE
|
||||
#define RGB_PIXELSIZE EXT_RGB_PIXELSIZE
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_extrgb_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
|
||||
#define RGB_RED EXT_RGBX_RED
|
||||
#define RGB_GREEN EXT_RGBX_GREEN
|
||||
#define RGB_BLUE EXT_RGBX_BLUE
|
||||
#define RGB_ALPHA 3
|
||||
#define RGB_PIXELSIZE EXT_RGBX_PIXELSIZE
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_extrgbx_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
|
||||
#define RGB_RED EXT_BGR_RED
|
||||
#define RGB_GREEN EXT_BGR_GREEN
|
||||
#define RGB_BLUE EXT_BGR_BLUE
|
||||
#define RGB_PIXELSIZE EXT_BGR_PIXELSIZE
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_extbgr_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
|
||||
#define RGB_RED EXT_BGRX_RED
|
||||
#define RGB_GREEN EXT_BGRX_GREEN
|
||||
#define RGB_BLUE EXT_BGRX_BLUE
|
||||
#define RGB_ALPHA 3
|
||||
#define RGB_PIXELSIZE EXT_BGRX_PIXELSIZE
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_extbgrx_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
|
||||
#define RGB_RED EXT_XBGR_RED
|
||||
#define RGB_GREEN EXT_XBGR_GREEN
|
||||
#define RGB_BLUE EXT_XBGR_BLUE
|
||||
#define RGB_ALPHA 0
|
||||
#define RGB_PIXELSIZE EXT_XBGR_PIXELSIZE
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_extxbgr_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
|
||||
#define RGB_RED EXT_XRGB_RED
|
||||
#define RGB_GREEN EXT_XRGB_GREEN
|
||||
#define RGB_BLUE EXT_XRGB_BLUE
|
||||
#define RGB_ALPHA 0
|
||||
#define RGB_PIXELSIZE EXT_XRGB_PIXELSIZE
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_extxrgb_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
|
||||
/* YCbCr -> RGB565 Conversion */
|
||||
|
||||
#define RGB_PIXELSIZE 2
|
||||
#define jsimd_ycc_rgb_convert_neon jsimd_ycc_rgb565_convert_neon
|
||||
#include "jdcolext-neon.c"
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_ycc_rgb_convert_neon
|
||||
+145
@@ -0,0 +1,145 @@
|
||||
/*
|
||||
* jdmerge-neon.c - merged upsampling/color conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "jconfigint.h"
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* YCbCr -> RGB conversion constants */
|
||||
|
||||
#define F_0_344 11277 /* 0.3441467 = 11277 * 2^-15 */
|
||||
#define F_0_714 23401 /* 0.7141418 = 23401 * 2^-15 */
|
||||
#define F_1_402 22971 /* 1.4020386 = 22971 * 2^-14 */
|
||||
#define F_1_772 29033 /* 1.7720337 = 29033 * 2^-14 */
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_ycc_rgb_convert_neon_consts[] = {
|
||||
-F_0_344, F_0_714, F_1_402, F_1_772
|
||||
};
|
||||
|
||||
|
||||
/* Include inline routines for colorspace extensions. */
|
||||
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
|
||||
#define RGB_RED EXT_RGB_RED
|
||||
#define RGB_GREEN EXT_RGB_GREEN
|
||||
#define RGB_BLUE EXT_RGB_BLUE
|
||||
#define RGB_PIXELSIZE EXT_RGB_PIXELSIZE
|
||||
#define jsimd_h2v1_merged_upsample_neon jsimd_h2v1_extrgb_merged_upsample_neon
|
||||
#define jsimd_h2v2_merged_upsample_neon jsimd_h2v2_extrgb_merged_upsample_neon
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_h2v1_merged_upsample_neon
|
||||
#undef jsimd_h2v2_merged_upsample_neon
|
||||
|
||||
#define RGB_RED EXT_RGBX_RED
|
||||
#define RGB_GREEN EXT_RGBX_GREEN
|
||||
#define RGB_BLUE EXT_RGBX_BLUE
|
||||
#define RGB_ALPHA 3
|
||||
#define RGB_PIXELSIZE EXT_RGBX_PIXELSIZE
|
||||
#define jsimd_h2v1_merged_upsample_neon jsimd_h2v1_extrgbx_merged_upsample_neon
|
||||
#define jsimd_h2v2_merged_upsample_neon jsimd_h2v2_extrgbx_merged_upsample_neon
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_h2v1_merged_upsample_neon
|
||||
#undef jsimd_h2v2_merged_upsample_neon
|
||||
|
||||
#define RGB_RED EXT_BGR_RED
|
||||
#define RGB_GREEN EXT_BGR_GREEN
|
||||
#define RGB_BLUE EXT_BGR_BLUE
|
||||
#define RGB_PIXELSIZE EXT_BGR_PIXELSIZE
|
||||
#define jsimd_h2v1_merged_upsample_neon jsimd_h2v1_extbgr_merged_upsample_neon
|
||||
#define jsimd_h2v2_merged_upsample_neon jsimd_h2v2_extbgr_merged_upsample_neon
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_h2v1_merged_upsample_neon
|
||||
#undef jsimd_h2v2_merged_upsample_neon
|
||||
|
||||
#define RGB_RED EXT_BGRX_RED
|
||||
#define RGB_GREEN EXT_BGRX_GREEN
|
||||
#define RGB_BLUE EXT_BGRX_BLUE
|
||||
#define RGB_ALPHA 3
|
||||
#define RGB_PIXELSIZE EXT_BGRX_PIXELSIZE
|
||||
#define jsimd_h2v1_merged_upsample_neon jsimd_h2v1_extbgrx_merged_upsample_neon
|
||||
#define jsimd_h2v2_merged_upsample_neon jsimd_h2v2_extbgrx_merged_upsample_neon
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_h2v1_merged_upsample_neon
|
||||
#undef jsimd_h2v2_merged_upsample_neon
|
||||
|
||||
#define RGB_RED EXT_XBGR_RED
|
||||
#define RGB_GREEN EXT_XBGR_GREEN
|
||||
#define RGB_BLUE EXT_XBGR_BLUE
|
||||
#define RGB_ALPHA 0
|
||||
#define RGB_PIXELSIZE EXT_XBGR_PIXELSIZE
|
||||
#define jsimd_h2v1_merged_upsample_neon jsimd_h2v1_extxbgr_merged_upsample_neon
|
||||
#define jsimd_h2v2_merged_upsample_neon jsimd_h2v2_extxbgr_merged_upsample_neon
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_h2v1_merged_upsample_neon
|
||||
#undef jsimd_h2v2_merged_upsample_neon
|
||||
|
||||
#define RGB_RED EXT_XRGB_RED
|
||||
#define RGB_GREEN EXT_XRGB_GREEN
|
||||
#define RGB_BLUE EXT_XRGB_BLUE
|
||||
#define RGB_ALPHA 0
|
||||
#define RGB_PIXELSIZE EXT_XRGB_PIXELSIZE
|
||||
#define jsimd_h2v1_merged_upsample_neon jsimd_h2v1_extxrgb_merged_upsample_neon
|
||||
#define jsimd_h2v2_merged_upsample_neon jsimd_h2v2_extxrgb_merged_upsample_neon
|
||||
#include "jdmrgext-neon.c"
|
||||
#undef RGB_RED
|
||||
#undef RGB_GREEN
|
||||
#undef RGB_BLUE
|
||||
#undef RGB_ALPHA
|
||||
#undef RGB_PIXELSIZE
|
||||
#undef jsimd_h2v1_merged_upsample_neon
|
||||
+723
@@ -0,0 +1,723 @@
|
||||
/*
|
||||
* jdmrgext-neon.c - merged upsampling/color conversion (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
/* This file is included by jdmerge-neon.c. */
|
||||
|
||||
|
||||
/* These routines combine simple (non-fancy, i.e. non-smooth) h2v1 or h2v2
|
||||
* chroma upsampling and YCbCr -> RGB color conversion into a single function.
|
||||
*
|
||||
* As with the standalone functions, YCbCr -> RGB conversion is defined by the
|
||||
* following equations:
|
||||
* R = Y + 1.40200 * (Cr - 128)
|
||||
* G = Y - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128)
|
||||
* B = Y + 1.77200 * (Cb - 128)
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.3441467 = 11277 * 2^-15
|
||||
* 0.7141418 = 23401 * 2^-15
|
||||
* 1.4020386 = 22971 * 2^-14
|
||||
* 1.7720337 = 29033 * 2^-14
|
||||
* These constants are defined in jdmerge-neon.c.
|
||||
*
|
||||
* To ensure correct results, rounding is used when descaling.
|
||||
*/
|
||||
|
||||
/* Notes on safe memory access for merged upsampling/YCbCr -> RGB conversion
|
||||
* routines:
|
||||
*
|
||||
* Input memory buffers can be safely overread up to the next multiple of
|
||||
* ALIGN_SIZE bytes, since they are always allocated by alloc_sarray() in
|
||||
* jmemmgr.c.
|
||||
*
|
||||
* The output buffer cannot safely be written beyond output_width, since
|
||||
* output_buf points to a possibly unpadded row in the decompressed image
|
||||
* buffer allocated by the calling program.
|
||||
*/
|
||||
|
||||
/* Upsample and color convert for the case of 2:1 horizontal and 1:1 vertical.
|
||||
*/
|
||||
|
||||
void jsimd_h2v1_merged_upsample_neon(JDIMENSION output_width,
|
||||
JSAMPIMAGE input_buf,
|
||||
JDIMENSION in_row_group_ctr,
|
||||
JSAMPARRAY output_buf)
|
||||
{
|
||||
JSAMPROW outptr;
|
||||
/* Pointers to Y, Cb, and Cr data */
|
||||
JSAMPROW inptr0, inptr1, inptr2;
|
||||
|
||||
const int16x4_t consts = vld1_s16(jsimd_ycc_rgb_convert_neon_consts);
|
||||
const int16x8_t neg_128 = vdupq_n_s16(-128);
|
||||
|
||||
inptr0 = input_buf[0][in_row_group_ctr];
|
||||
inptr1 = input_buf[1][in_row_group_ctr];
|
||||
inptr2 = input_buf[2][in_row_group_ctr];
|
||||
outptr = output_buf[0];
|
||||
|
||||
int cols_remaining = output_width;
|
||||
for (; cols_remaining >= 16; cols_remaining -= 16) {
|
||||
/* De-interleave Y component values into two separate vectors, one
|
||||
* containing the component values with even-numbered indices and one
|
||||
* containing the component values with odd-numbered indices.
|
||||
*/
|
||||
uint8x8x2_t y = vld2_u8(inptr0);
|
||||
uint8x8_t cb = vld1_u8(inptr1);
|
||||
uint8x8_t cr = vld1_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cr));
|
||||
int16x8_t cb_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cb));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_l = vmull_lane_s16(vget_low_s16(cb_128), consts, 0);
|
||||
int32x4_t g_sub_y_h = vmull_lane_s16(vget_high_s16(cb_128), consts, 0);
|
||||
g_sub_y_l = vmlsl_lane_s16(g_sub_y_l, vget_low_s16(cr_128), consts, 1);
|
||||
g_sub_y_h = vmlsl_lane_s16(g_sub_y_h, vget_high_s16(cr_128), consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y = vcombine_s16(vrshrn_n_s32(g_sub_y_l, 15),
|
||||
vrshrn_n_s32(g_sub_y_h, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128, 1), consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128, 1), consts, 3);
|
||||
/* Add the chroma-derived values (G-Y, R-Y, and B-Y) to both the "even" and
|
||||
* "odd" Y component values. This effectively upsamples the chroma
|
||||
* components horizontally.
|
||||
*/
|
||||
int16x8_t g_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y.val[0]));
|
||||
int16x8_t r_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y.val[0]));
|
||||
int16x8_t b_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y.val[0]));
|
||||
int16x8_t g_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y.val[1]));
|
||||
int16x8_t r_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y.val[1]));
|
||||
int16x8_t b_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y.val[1]));
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255].
|
||||
* Re-interleave the "even" and "odd" component values.
|
||||
*/
|
||||
uint8x8x2_t r = vzip_u8(vqmovun_s16(r_even), vqmovun_s16(r_odd));
|
||||
uint8x8x2_t g = vzip_u8(vqmovun_s16(g_even), vqmovun_s16(g_odd));
|
||||
uint8x8x2_t b = vzip_u8(vqmovun_s16(b_even), vqmovun_s16(b_odd));
|
||||
|
||||
#ifdef RGB_ALPHA
|
||||
uint8x16x4_t rgba;
|
||||
rgba.val[RGB_RED] = vcombine_u8(r.val[0], r.val[1]);
|
||||
rgba.val[RGB_GREEN] = vcombine_u8(g.val[0], g.val[1]);
|
||||
rgba.val[RGB_BLUE] = vcombine_u8(b.val[0], b.val[1]);
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba.val[RGB_ALPHA] = vdupq_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
vst4q_u8(outptr, rgba);
|
||||
#else
|
||||
uint8x16x3_t rgb;
|
||||
rgb.val[RGB_RED] = vcombine_u8(r.val[0], r.val[1]);
|
||||
rgb.val[RGB_GREEN] = vcombine_u8(g.val[0], g.val[1]);
|
||||
rgb.val[RGB_BLUE] = vcombine_u8(b.val[0], b.val[1]);
|
||||
/* Store RGB pixel data to memory. */
|
||||
vst3q_u8(outptr, rgb);
|
||||
#endif
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr0 += 16;
|
||||
inptr1 += 8;
|
||||
inptr2 += 8;
|
||||
outptr += (RGB_PIXELSIZE * 16);
|
||||
}
|
||||
|
||||
if (cols_remaining > 0) {
|
||||
/* De-interleave Y component values into two separate vectors, one
|
||||
* containing the component values with even-numbered indices and one
|
||||
* containing the component values with odd-numbered indices.
|
||||
*/
|
||||
uint8x8x2_t y = vld2_u8(inptr0);
|
||||
uint8x8_t cb = vld1_u8(inptr1);
|
||||
uint8x8_t cr = vld1_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cr));
|
||||
int16x8_t cb_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cb));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_l = vmull_lane_s16(vget_low_s16(cb_128), consts, 0);
|
||||
int32x4_t g_sub_y_h = vmull_lane_s16(vget_high_s16(cb_128), consts, 0);
|
||||
g_sub_y_l = vmlsl_lane_s16(g_sub_y_l, vget_low_s16(cr_128), consts, 1);
|
||||
g_sub_y_h = vmlsl_lane_s16(g_sub_y_h, vget_high_s16(cr_128), consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y = vcombine_s16(vrshrn_n_s32(g_sub_y_l, 15),
|
||||
vrshrn_n_s32(g_sub_y_h, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128, 1), consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128, 1), consts, 3);
|
||||
/* Add the chroma-derived values (G-Y, R-Y, and B-Y) to both the "even" and
|
||||
* "odd" Y component values. This effectively upsamples the chroma
|
||||
* components horizontally.
|
||||
*/
|
||||
int16x8_t g_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y.val[0]));
|
||||
int16x8_t r_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y.val[0]));
|
||||
int16x8_t b_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y.val[0]));
|
||||
int16x8_t g_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y.val[1]));
|
||||
int16x8_t r_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y.val[1]));
|
||||
int16x8_t b_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y.val[1]));
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255].
|
||||
* Re-interleave the "even" and "odd" component values.
|
||||
*/
|
||||
uint8x8x2_t r = vzip_u8(vqmovun_s16(r_even), vqmovun_s16(r_odd));
|
||||
uint8x8x2_t g = vzip_u8(vqmovun_s16(g_even), vqmovun_s16(g_odd));
|
||||
uint8x8x2_t b = vzip_u8(vqmovun_s16(b_even), vqmovun_s16(b_odd));
|
||||
|
||||
#ifdef RGB_ALPHA
|
||||
uint8x8x4_t rgba_h;
|
||||
rgba_h.val[RGB_RED] = r.val[1];
|
||||
rgba_h.val[RGB_GREEN] = g.val[1];
|
||||
rgba_h.val[RGB_BLUE] = b.val[1];
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba_h.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
uint8x8x4_t rgba_l;
|
||||
rgba_l.val[RGB_RED] = r.val[0];
|
||||
rgba_l.val[RGB_GREEN] = g.val[0];
|
||||
rgba_l.val[RGB_BLUE] = b.val[0];
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba_l.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 15:
|
||||
vst4_lane_u8(outptr + 14 * RGB_PIXELSIZE, rgba_h, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 14:
|
||||
vst4_lane_u8(outptr + 13 * RGB_PIXELSIZE, rgba_h, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 13:
|
||||
vst4_lane_u8(outptr + 12 * RGB_PIXELSIZE, rgba_h, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 12:
|
||||
vst4_lane_u8(outptr + 11 * RGB_PIXELSIZE, rgba_h, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 11:
|
||||
vst4_lane_u8(outptr + 10 * RGB_PIXELSIZE, rgba_h, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 10:
|
||||
vst4_lane_u8(outptr + 9 * RGB_PIXELSIZE, rgba_h, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 9:
|
||||
vst4_lane_u8(outptr + 8 * RGB_PIXELSIZE, rgba_h, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 8:
|
||||
vst4_u8(outptr, rgba_l);
|
||||
break;
|
||||
case 7:
|
||||
vst4_lane_u8(outptr + 6 * RGB_PIXELSIZE, rgba_l, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst4_lane_u8(outptr + 5 * RGB_PIXELSIZE, rgba_l, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst4_lane_u8(outptr + 4 * RGB_PIXELSIZE, rgba_l, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst4_lane_u8(outptr + 3 * RGB_PIXELSIZE, rgba_l, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst4_lane_u8(outptr + 2 * RGB_PIXELSIZE, rgba_l, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst4_lane_u8(outptr + RGB_PIXELSIZE, rgba_l, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst4_lane_u8(outptr, rgba_l, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#else
|
||||
uint8x8x3_t rgb_h;
|
||||
rgb_h.val[RGB_RED] = r.val[1];
|
||||
rgb_h.val[RGB_GREEN] = g.val[1];
|
||||
rgb_h.val[RGB_BLUE] = b.val[1];
|
||||
uint8x8x3_t rgb_l;
|
||||
rgb_l.val[RGB_RED] = r.val[0];
|
||||
rgb_l.val[RGB_GREEN] = g.val[0];
|
||||
rgb_l.val[RGB_BLUE] = b.val[0];
|
||||
/* Store RGB pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 15:
|
||||
vst3_lane_u8(outptr + 14 * RGB_PIXELSIZE, rgb_h, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 14:
|
||||
vst3_lane_u8(outptr + 13 * RGB_PIXELSIZE, rgb_h, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 13:
|
||||
vst3_lane_u8(outptr + 12 * RGB_PIXELSIZE, rgb_h, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 12:
|
||||
vst3_lane_u8(outptr + 11 * RGB_PIXELSIZE, rgb_h, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 11:
|
||||
vst3_lane_u8(outptr + 10 * RGB_PIXELSIZE, rgb_h, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 10:
|
||||
vst3_lane_u8(outptr + 9 * RGB_PIXELSIZE, rgb_h, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 9:
|
||||
vst3_lane_u8(outptr + 8 * RGB_PIXELSIZE, rgb_h, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 8:
|
||||
vst3_u8(outptr, rgb_l);
|
||||
break;
|
||||
case 7:
|
||||
vst3_lane_u8(outptr + 6 * RGB_PIXELSIZE, rgb_l, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst3_lane_u8(outptr + 5 * RGB_PIXELSIZE, rgb_l, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst3_lane_u8(outptr + 4 * RGB_PIXELSIZE, rgb_l, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst3_lane_u8(outptr + 3 * RGB_PIXELSIZE, rgb_l, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst3_lane_u8(outptr + 2 * RGB_PIXELSIZE, rgb_l, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst3_lane_u8(outptr + RGB_PIXELSIZE, rgb_l, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst3_lane_u8(outptr, rgb_l, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* Upsample and color convert for the case of 2:1 horizontal and 2:1 vertical.
|
||||
*
|
||||
* See comments above for details regarding color conversion and safe memory
|
||||
* access.
|
||||
*/
|
||||
|
||||
void jsimd_h2v2_merged_upsample_neon(JDIMENSION output_width,
|
||||
JSAMPIMAGE input_buf,
|
||||
JDIMENSION in_row_group_ctr,
|
||||
JSAMPARRAY output_buf)
|
||||
{
|
||||
JSAMPROW outptr0, outptr1;
|
||||
/* Pointers to Y (both rows), Cb, and Cr data */
|
||||
JSAMPROW inptr0_0, inptr0_1, inptr1, inptr2;
|
||||
|
||||
const int16x4_t consts = vld1_s16(jsimd_ycc_rgb_convert_neon_consts);
|
||||
const int16x8_t neg_128 = vdupq_n_s16(-128);
|
||||
|
||||
inptr0_0 = input_buf[0][in_row_group_ctr * 2];
|
||||
inptr0_1 = input_buf[0][in_row_group_ctr * 2 + 1];
|
||||
inptr1 = input_buf[1][in_row_group_ctr];
|
||||
inptr2 = input_buf[2][in_row_group_ctr];
|
||||
outptr0 = output_buf[0];
|
||||
outptr1 = output_buf[1];
|
||||
|
||||
int cols_remaining = output_width;
|
||||
for (; cols_remaining >= 16; cols_remaining -= 16) {
|
||||
/* For each row, de-interleave Y component values into two separate
|
||||
* vectors, one containing the component values with even-numbered indices
|
||||
* and one containing the component values with odd-numbered indices.
|
||||
*/
|
||||
uint8x8x2_t y0 = vld2_u8(inptr0_0);
|
||||
uint8x8x2_t y1 = vld2_u8(inptr0_1);
|
||||
uint8x8_t cb = vld1_u8(inptr1);
|
||||
uint8x8_t cr = vld1_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cr));
|
||||
int16x8_t cb_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cb));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_l = vmull_lane_s16(vget_low_s16(cb_128), consts, 0);
|
||||
int32x4_t g_sub_y_h = vmull_lane_s16(vget_high_s16(cb_128), consts, 0);
|
||||
g_sub_y_l = vmlsl_lane_s16(g_sub_y_l, vget_low_s16(cr_128), consts, 1);
|
||||
g_sub_y_h = vmlsl_lane_s16(g_sub_y_h, vget_high_s16(cr_128), consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y = vcombine_s16(vrshrn_n_s32(g_sub_y_l, 15),
|
||||
vrshrn_n_s32(g_sub_y_h, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128, 1), consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128, 1), consts, 3);
|
||||
/* For each row, add the chroma-derived values (G-Y, R-Y, and B-Y) to both
|
||||
* the "even" and "odd" Y component values. This effectively upsamples the
|
||||
* chroma components both horizontally and vertically.
|
||||
*/
|
||||
int16x8_t g0_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y0.val[0]));
|
||||
int16x8_t r0_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y0.val[0]));
|
||||
int16x8_t b0_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y0.val[0]));
|
||||
int16x8_t g0_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y0.val[1]));
|
||||
int16x8_t r0_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y0.val[1]));
|
||||
int16x8_t b0_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y0.val[1]));
|
||||
int16x8_t g1_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y1.val[0]));
|
||||
int16x8_t r1_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y1.val[0]));
|
||||
int16x8_t b1_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y1.val[0]));
|
||||
int16x8_t g1_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y1.val[1]));
|
||||
int16x8_t r1_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y1.val[1]));
|
||||
int16x8_t b1_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y1.val[1]));
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255].
|
||||
* Re-interleave the "even" and "odd" component values.
|
||||
*/
|
||||
uint8x8x2_t r0 = vzip_u8(vqmovun_s16(r0_even), vqmovun_s16(r0_odd));
|
||||
uint8x8x2_t r1 = vzip_u8(vqmovun_s16(r1_even), vqmovun_s16(r1_odd));
|
||||
uint8x8x2_t g0 = vzip_u8(vqmovun_s16(g0_even), vqmovun_s16(g0_odd));
|
||||
uint8x8x2_t g1 = vzip_u8(vqmovun_s16(g1_even), vqmovun_s16(g1_odd));
|
||||
uint8x8x2_t b0 = vzip_u8(vqmovun_s16(b0_even), vqmovun_s16(b0_odd));
|
||||
uint8x8x2_t b1 = vzip_u8(vqmovun_s16(b1_even), vqmovun_s16(b1_odd));
|
||||
|
||||
#ifdef RGB_ALPHA
|
||||
uint8x16x4_t rgba0, rgba1;
|
||||
rgba0.val[RGB_RED] = vcombine_u8(r0.val[0], r0.val[1]);
|
||||
rgba1.val[RGB_RED] = vcombine_u8(r1.val[0], r1.val[1]);
|
||||
rgba0.val[RGB_GREEN] = vcombine_u8(g0.val[0], g0.val[1]);
|
||||
rgba1.val[RGB_GREEN] = vcombine_u8(g1.val[0], g1.val[1]);
|
||||
rgba0.val[RGB_BLUE] = vcombine_u8(b0.val[0], b0.val[1]);
|
||||
rgba1.val[RGB_BLUE] = vcombine_u8(b1.val[0], b1.val[1]);
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba0.val[RGB_ALPHA] = vdupq_n_u8(0xFF);
|
||||
rgba1.val[RGB_ALPHA] = vdupq_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
vst4q_u8(outptr0, rgba0);
|
||||
vst4q_u8(outptr1, rgba1);
|
||||
#else
|
||||
uint8x16x3_t rgb0, rgb1;
|
||||
rgb0.val[RGB_RED] = vcombine_u8(r0.val[0], r0.val[1]);
|
||||
rgb1.val[RGB_RED] = vcombine_u8(r1.val[0], r1.val[1]);
|
||||
rgb0.val[RGB_GREEN] = vcombine_u8(g0.val[0], g0.val[1]);
|
||||
rgb1.val[RGB_GREEN] = vcombine_u8(g1.val[0], g1.val[1]);
|
||||
rgb0.val[RGB_BLUE] = vcombine_u8(b0.val[0], b0.val[1]);
|
||||
rgb1.val[RGB_BLUE] = vcombine_u8(b1.val[0], b1.val[1]);
|
||||
/* Store RGB pixel data to memory. */
|
||||
vst3q_u8(outptr0, rgb0);
|
||||
vst3q_u8(outptr1, rgb1);
|
||||
#endif
|
||||
|
||||
/* Increment pointers. */
|
||||
inptr0_0 += 16;
|
||||
inptr0_1 += 16;
|
||||
inptr1 += 8;
|
||||
inptr2 += 8;
|
||||
outptr0 += (RGB_PIXELSIZE * 16);
|
||||
outptr1 += (RGB_PIXELSIZE * 16);
|
||||
}
|
||||
|
||||
if (cols_remaining > 0) {
|
||||
/* For each row, de-interleave Y component values into two separate
|
||||
* vectors, one containing the component values with even-numbered indices
|
||||
* and one containing the component values with odd-numbered indices.
|
||||
*/
|
||||
uint8x8x2_t y0 = vld2_u8(inptr0_0);
|
||||
uint8x8x2_t y1 = vld2_u8(inptr0_1);
|
||||
uint8x8_t cb = vld1_u8(inptr1);
|
||||
uint8x8_t cr = vld1_u8(inptr2);
|
||||
/* Subtract 128 from Cb and Cr. */
|
||||
int16x8_t cr_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cr));
|
||||
int16x8_t cb_128 =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(neg_128), cb));
|
||||
/* Compute G-Y: - 0.34414 * (Cb - 128) - 0.71414 * (Cr - 128) */
|
||||
int32x4_t g_sub_y_l = vmull_lane_s16(vget_low_s16(cb_128), consts, 0);
|
||||
int32x4_t g_sub_y_h = vmull_lane_s16(vget_high_s16(cb_128), consts, 0);
|
||||
g_sub_y_l = vmlsl_lane_s16(g_sub_y_l, vget_low_s16(cr_128), consts, 1);
|
||||
g_sub_y_h = vmlsl_lane_s16(g_sub_y_h, vget_high_s16(cr_128), consts, 1);
|
||||
/* Descale G components: shift right 15, round, and narrow to 16-bit. */
|
||||
int16x8_t g_sub_y = vcombine_s16(vrshrn_n_s32(g_sub_y_l, 15),
|
||||
vrshrn_n_s32(g_sub_y_h, 15));
|
||||
/* Compute R-Y: 1.40200 * (Cr - 128) */
|
||||
int16x8_t r_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cr_128, 1), consts, 2);
|
||||
/* Compute B-Y: 1.77200 * (Cb - 128) */
|
||||
int16x8_t b_sub_y = vqrdmulhq_lane_s16(vshlq_n_s16(cb_128, 1), consts, 3);
|
||||
/* For each row, add the chroma-derived values (G-Y, R-Y, and B-Y) to both
|
||||
* the "even" and "odd" Y component values. This effectively upsamples the
|
||||
* chroma components both horizontally and vertically.
|
||||
*/
|
||||
int16x8_t g0_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y0.val[0]));
|
||||
int16x8_t r0_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y0.val[0]));
|
||||
int16x8_t b0_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y0.val[0]));
|
||||
int16x8_t g0_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y0.val[1]));
|
||||
int16x8_t r0_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y0.val[1]));
|
||||
int16x8_t b0_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y0.val[1]));
|
||||
int16x8_t g1_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y1.val[0]));
|
||||
int16x8_t r1_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y1.val[0]));
|
||||
int16x8_t b1_even =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y1.val[0]));
|
||||
int16x8_t g1_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(g_sub_y),
|
||||
y1.val[1]));
|
||||
int16x8_t r1_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(r_sub_y),
|
||||
y1.val[1]));
|
||||
int16x8_t b1_odd =
|
||||
vreinterpretq_s16_u16(vaddw_u8(vreinterpretq_u16_s16(b_sub_y),
|
||||
y1.val[1]));
|
||||
/* Convert each component to unsigned and narrow, clamping to [0-255].
|
||||
* Re-interleave the "even" and "odd" component values.
|
||||
*/
|
||||
uint8x8x2_t r0 = vzip_u8(vqmovun_s16(r0_even), vqmovun_s16(r0_odd));
|
||||
uint8x8x2_t r1 = vzip_u8(vqmovun_s16(r1_even), vqmovun_s16(r1_odd));
|
||||
uint8x8x2_t g0 = vzip_u8(vqmovun_s16(g0_even), vqmovun_s16(g0_odd));
|
||||
uint8x8x2_t g1 = vzip_u8(vqmovun_s16(g1_even), vqmovun_s16(g1_odd));
|
||||
uint8x8x2_t b0 = vzip_u8(vqmovun_s16(b0_even), vqmovun_s16(b0_odd));
|
||||
uint8x8x2_t b1 = vzip_u8(vqmovun_s16(b1_even), vqmovun_s16(b1_odd));
|
||||
|
||||
#ifdef RGB_ALPHA
|
||||
uint8x8x4_t rgba0_h, rgba1_h;
|
||||
rgba0_h.val[RGB_RED] = r0.val[1];
|
||||
rgba1_h.val[RGB_RED] = r1.val[1];
|
||||
rgba0_h.val[RGB_GREEN] = g0.val[1];
|
||||
rgba1_h.val[RGB_GREEN] = g1.val[1];
|
||||
rgba0_h.val[RGB_BLUE] = b0.val[1];
|
||||
rgba1_h.val[RGB_BLUE] = b1.val[1];
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba0_h.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
rgba1_h.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
|
||||
uint8x8x4_t rgba0_l, rgba1_l;
|
||||
rgba0_l.val[RGB_RED] = r0.val[0];
|
||||
rgba1_l.val[RGB_RED] = r1.val[0];
|
||||
rgba0_l.val[RGB_GREEN] = g0.val[0];
|
||||
rgba1_l.val[RGB_GREEN] = g1.val[0];
|
||||
rgba0_l.val[RGB_BLUE] = b0.val[0];
|
||||
rgba1_l.val[RGB_BLUE] = b1.val[0];
|
||||
/* Set alpha channel to opaque (0xFF). */
|
||||
rgba0_l.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
rgba1_l.val[RGB_ALPHA] = vdup_n_u8(0xFF);
|
||||
/* Store RGBA pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 15:
|
||||
vst4_lane_u8(outptr0 + 14 * RGB_PIXELSIZE, rgba0_h, 6);
|
||||
vst4_lane_u8(outptr1 + 14 * RGB_PIXELSIZE, rgba1_h, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 14:
|
||||
vst4_lane_u8(outptr0 + 13 * RGB_PIXELSIZE, rgba0_h, 5);
|
||||
vst4_lane_u8(outptr1 + 13 * RGB_PIXELSIZE, rgba1_h, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 13:
|
||||
vst4_lane_u8(outptr0 + 12 * RGB_PIXELSIZE, rgba0_h, 4);
|
||||
vst4_lane_u8(outptr1 + 12 * RGB_PIXELSIZE, rgba1_h, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 12:
|
||||
vst4_lane_u8(outptr0 + 11 * RGB_PIXELSIZE, rgba0_h, 3);
|
||||
vst4_lane_u8(outptr1 + 11 * RGB_PIXELSIZE, rgba1_h, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 11:
|
||||
vst4_lane_u8(outptr0 + 10 * RGB_PIXELSIZE, rgba0_h, 2);
|
||||
vst4_lane_u8(outptr1 + 10 * RGB_PIXELSIZE, rgba1_h, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 10:
|
||||
vst4_lane_u8(outptr0 + 9 * RGB_PIXELSIZE, rgba0_h, 1);
|
||||
vst4_lane_u8(outptr1 + 9 * RGB_PIXELSIZE, rgba1_h, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 9:
|
||||
vst4_lane_u8(outptr0 + 8 * RGB_PIXELSIZE, rgba0_h, 0);
|
||||
vst4_lane_u8(outptr1 + 8 * RGB_PIXELSIZE, rgba1_h, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 8:
|
||||
vst4_u8(outptr0, rgba0_l);
|
||||
vst4_u8(outptr1, rgba1_l);
|
||||
break;
|
||||
case 7:
|
||||
vst4_lane_u8(outptr0 + 6 * RGB_PIXELSIZE, rgba0_l, 6);
|
||||
vst4_lane_u8(outptr1 + 6 * RGB_PIXELSIZE, rgba1_l, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst4_lane_u8(outptr0 + 5 * RGB_PIXELSIZE, rgba0_l, 5);
|
||||
vst4_lane_u8(outptr1 + 5 * RGB_PIXELSIZE, rgba1_l, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst4_lane_u8(outptr0 + 4 * RGB_PIXELSIZE, rgba0_l, 4);
|
||||
vst4_lane_u8(outptr1 + 4 * RGB_PIXELSIZE, rgba1_l, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst4_lane_u8(outptr0 + 3 * RGB_PIXELSIZE, rgba0_l, 3);
|
||||
vst4_lane_u8(outptr1 + 3 * RGB_PIXELSIZE, rgba1_l, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst4_lane_u8(outptr0 + 2 * RGB_PIXELSIZE, rgba0_l, 2);
|
||||
vst4_lane_u8(outptr1 + 2 * RGB_PIXELSIZE, rgba1_l, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst4_lane_u8(outptr0 + 1 * RGB_PIXELSIZE, rgba0_l, 1);
|
||||
vst4_lane_u8(outptr1 + 1 * RGB_PIXELSIZE, rgba1_l, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst4_lane_u8(outptr0, rgba0_l, 0);
|
||||
vst4_lane_u8(outptr1, rgba1_l, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#else
|
||||
uint8x8x3_t rgb0_h, rgb1_h;
|
||||
rgb0_h.val[RGB_RED] = r0.val[1];
|
||||
rgb1_h.val[RGB_RED] = r1.val[1];
|
||||
rgb0_h.val[RGB_GREEN] = g0.val[1];
|
||||
rgb1_h.val[RGB_GREEN] = g1.val[1];
|
||||
rgb0_h.val[RGB_BLUE] = b0.val[1];
|
||||
rgb1_h.val[RGB_BLUE] = b1.val[1];
|
||||
|
||||
uint8x8x3_t rgb0_l, rgb1_l;
|
||||
rgb0_l.val[RGB_RED] = r0.val[0];
|
||||
rgb1_l.val[RGB_RED] = r1.val[0];
|
||||
rgb0_l.val[RGB_GREEN] = g0.val[0];
|
||||
rgb1_l.val[RGB_GREEN] = g1.val[0];
|
||||
rgb0_l.val[RGB_BLUE] = b0.val[0];
|
||||
rgb1_l.val[RGB_BLUE] = b1.val[0];
|
||||
/* Store RGB pixel data to memory. */
|
||||
switch (cols_remaining) {
|
||||
case 15:
|
||||
vst3_lane_u8(outptr0 + 14 * RGB_PIXELSIZE, rgb0_h, 6);
|
||||
vst3_lane_u8(outptr1 + 14 * RGB_PIXELSIZE, rgb1_h, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 14:
|
||||
vst3_lane_u8(outptr0 + 13 * RGB_PIXELSIZE, rgb0_h, 5);
|
||||
vst3_lane_u8(outptr1 + 13 * RGB_PIXELSIZE, rgb1_h, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 13:
|
||||
vst3_lane_u8(outptr0 + 12 * RGB_PIXELSIZE, rgb0_h, 4);
|
||||
vst3_lane_u8(outptr1 + 12 * RGB_PIXELSIZE, rgb1_h, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 12:
|
||||
vst3_lane_u8(outptr0 + 11 * RGB_PIXELSIZE, rgb0_h, 3);
|
||||
vst3_lane_u8(outptr1 + 11 * RGB_PIXELSIZE, rgb1_h, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 11:
|
||||
vst3_lane_u8(outptr0 + 10 * RGB_PIXELSIZE, rgb0_h, 2);
|
||||
vst3_lane_u8(outptr1 + 10 * RGB_PIXELSIZE, rgb1_h, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 10:
|
||||
vst3_lane_u8(outptr0 + 9 * RGB_PIXELSIZE, rgb0_h, 1);
|
||||
vst3_lane_u8(outptr1 + 9 * RGB_PIXELSIZE, rgb1_h, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 9:
|
||||
vst3_lane_u8(outptr0 + 8 * RGB_PIXELSIZE, rgb0_h, 0);
|
||||
vst3_lane_u8(outptr1 + 8 * RGB_PIXELSIZE, rgb1_h, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 8:
|
||||
vst3_u8(outptr0, rgb0_l);
|
||||
vst3_u8(outptr1, rgb1_l);
|
||||
break;
|
||||
case 7:
|
||||
vst3_lane_u8(outptr0 + 6 * RGB_PIXELSIZE, rgb0_l, 6);
|
||||
vst3_lane_u8(outptr1 + 6 * RGB_PIXELSIZE, rgb1_l, 6);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 6:
|
||||
vst3_lane_u8(outptr0 + 5 * RGB_PIXELSIZE, rgb0_l, 5);
|
||||
vst3_lane_u8(outptr1 + 5 * RGB_PIXELSIZE, rgb1_l, 5);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 5:
|
||||
vst3_lane_u8(outptr0 + 4 * RGB_PIXELSIZE, rgb0_l, 4);
|
||||
vst3_lane_u8(outptr1 + 4 * RGB_PIXELSIZE, rgb1_l, 4);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 4:
|
||||
vst3_lane_u8(outptr0 + 3 * RGB_PIXELSIZE, rgb0_l, 3);
|
||||
vst3_lane_u8(outptr1 + 3 * RGB_PIXELSIZE, rgb1_l, 3);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 3:
|
||||
vst3_lane_u8(outptr0 + 2 * RGB_PIXELSIZE, rgb0_l, 2);
|
||||
vst3_lane_u8(outptr1 + 2 * RGB_PIXELSIZE, rgb1_l, 2);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 2:
|
||||
vst3_lane_u8(outptr0 + 1 * RGB_PIXELSIZE, rgb0_l, 1);
|
||||
vst3_lane_u8(outptr1 + 1 * RGB_PIXELSIZE, rgb1_l, 1);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
case 1:
|
||||
vst3_lane_u8(outptr0, rgb0_l, 0);
|
||||
vst3_lane_u8(outptr1, rgb1_l, 0);
|
||||
FALLTHROUGH /*FALLTHROUGH*/
|
||||
default:
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
+569
@@ -0,0 +1,569 @@
|
||||
/*
|
||||
* jdsample-neon.c - upsampling (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* The diagram below shows a row of samples produced by h2v1 downsampling.
|
||||
*
|
||||
* s0 s1 s2
|
||||
* +---------+---------+---------+
|
||||
* | | | |
|
||||
* | p0 p1 | p2 p3 | p4 p5 |
|
||||
* | | | |
|
||||
* +---------+---------+---------+
|
||||
*
|
||||
* Samples s0-s2 were created by averaging the original pixel component values
|
||||
* centered at positions p0-p5 above. To approximate those original pixel
|
||||
* component values, we proportionally blend the adjacent samples in each row.
|
||||
*
|
||||
* An upsampled pixel component value is computed by blending the sample
|
||||
* containing the pixel center with the nearest neighboring sample, in the
|
||||
* ratio 3:1. For example:
|
||||
* p1(upsampled) = 3/4 * s0 + 1/4 * s1
|
||||
* p2(upsampled) = 3/4 * s1 + 1/4 * s0
|
||||
* When computing the first and last pixel component values in the row, there
|
||||
* is no adjacent sample to blend, so:
|
||||
* p0(upsampled) = s0
|
||||
* p5(upsampled) = s2
|
||||
*/
|
||||
|
||||
void jsimd_h2v1_fancy_upsample_neon(int max_v_samp_factor,
|
||||
JDIMENSION downsampled_width,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
JSAMPARRAY output_data = *output_data_ptr;
|
||||
JSAMPROW inptr, outptr;
|
||||
int inrow;
|
||||
unsigned colctr;
|
||||
/* Set up constants. */
|
||||
const uint16x8_t one_u16 = vdupq_n_u16(1);
|
||||
const uint8x8_t three_u8 = vdup_n_u8(3);
|
||||
|
||||
for (inrow = 0; inrow < max_v_samp_factor; inrow++) {
|
||||
inptr = input_data[inrow];
|
||||
outptr = output_data[inrow];
|
||||
/* First pixel component value in this row of the original image */
|
||||
*outptr = (JSAMPLE)GETJSAMPLE(*inptr);
|
||||
|
||||
/* 3/4 * containing sample + 1/4 * nearest neighboring sample
|
||||
* For p1: containing sample = s0, nearest neighboring sample = s1
|
||||
* For p2: containing sample = s1, nearest neighboring sample = s0
|
||||
*/
|
||||
uint8x16_t s0 = vld1q_u8(inptr);
|
||||
uint8x16_t s1 = vld1q_u8(inptr + 1);
|
||||
/* Multiplication makes vectors twice as wide. '_l' and '_h' suffixes
|
||||
* denote low half and high half respectively.
|
||||
*/
|
||||
uint16x8_t s1_add_3s0_l =
|
||||
vmlal_u8(vmovl_u8(vget_low_u8(s1)), vget_low_u8(s0), three_u8);
|
||||
uint16x8_t s1_add_3s0_h =
|
||||
vmlal_u8(vmovl_u8(vget_high_u8(s1)), vget_high_u8(s0), three_u8);
|
||||
uint16x8_t s0_add_3s1_l =
|
||||
vmlal_u8(vmovl_u8(vget_low_u8(s0)), vget_low_u8(s1), three_u8);
|
||||
uint16x8_t s0_add_3s1_h =
|
||||
vmlal_u8(vmovl_u8(vget_high_u8(s0)), vget_high_u8(s1), three_u8);
|
||||
/* Add ordered dithering bias to odd pixel values. */
|
||||
s0_add_3s1_l = vaddq_u16(s0_add_3s1_l, one_u16);
|
||||
s0_add_3s1_h = vaddq_u16(s0_add_3s1_h, one_u16);
|
||||
|
||||
/* The offset is initially 1, because the first pixel component has already
|
||||
* been stored. However, in subsequent iterations of the SIMD loop, this
|
||||
* offset is (2 * colctr - 1) to stay within the bounds of the sample
|
||||
* buffers without having to resort to a slow scalar tail case for the last
|
||||
* (downsampled_width % 16) samples. See "Creation of 2-D sample arrays"
|
||||
* in jmemmgr.c for more details.
|
||||
*/
|
||||
unsigned outptr_offset = 1;
|
||||
uint8x16x2_t output_pixels;
|
||||
|
||||
/* We use software pipelining to maximise performance. The code indented
|
||||
* an extra two spaces begins the next iteration of the loop.
|
||||
*/
|
||||
for (colctr = 16; colctr < downsampled_width; colctr += 16) {
|
||||
|
||||
s0 = vld1q_u8(inptr + colctr - 1);
|
||||
s1 = vld1q_u8(inptr + colctr);
|
||||
|
||||
/* Right-shift by 2 (divide by 4), narrow to 8-bit, and combine. */
|
||||
output_pixels.val[0] = vcombine_u8(vrshrn_n_u16(s1_add_3s0_l, 2),
|
||||
vrshrn_n_u16(s1_add_3s0_h, 2));
|
||||
output_pixels.val[1] = vcombine_u8(vshrn_n_u16(s0_add_3s1_l, 2),
|
||||
vshrn_n_u16(s0_add_3s1_h, 2));
|
||||
|
||||
/* Multiplication makes vectors twice as wide. '_l' and '_h' suffixes
|
||||
* denote low half and high half respectively.
|
||||
*/
|
||||
s1_add_3s0_l =
|
||||
vmlal_u8(vmovl_u8(vget_low_u8(s1)), vget_low_u8(s0), three_u8);
|
||||
s1_add_3s0_h =
|
||||
vmlal_u8(vmovl_u8(vget_high_u8(s1)), vget_high_u8(s0), three_u8);
|
||||
s0_add_3s1_l =
|
||||
vmlal_u8(vmovl_u8(vget_low_u8(s0)), vget_low_u8(s1), three_u8);
|
||||
s0_add_3s1_h =
|
||||
vmlal_u8(vmovl_u8(vget_high_u8(s0)), vget_high_u8(s1), three_u8);
|
||||
/* Add ordered dithering bias to odd pixel values. */
|
||||
s0_add_3s1_l = vaddq_u16(s0_add_3s1_l, one_u16);
|
||||
s0_add_3s1_h = vaddq_u16(s0_add_3s1_h, one_u16);
|
||||
|
||||
/* Store pixel component values to memory. */
|
||||
vst2q_u8(outptr + outptr_offset, output_pixels);
|
||||
outptr_offset = 2 * colctr - 1;
|
||||
}
|
||||
|
||||
/* Complete the last iteration of the loop. */
|
||||
|
||||
/* Right-shift by 2 (divide by 4), narrow to 8-bit, and combine. */
|
||||
output_pixels.val[0] = vcombine_u8(vrshrn_n_u16(s1_add_3s0_l, 2),
|
||||
vrshrn_n_u16(s1_add_3s0_h, 2));
|
||||
output_pixels.val[1] = vcombine_u8(vshrn_n_u16(s0_add_3s1_l, 2),
|
||||
vshrn_n_u16(s0_add_3s1_h, 2));
|
||||
/* Store pixel component values to memory. */
|
||||
vst2q_u8(outptr + outptr_offset, output_pixels);
|
||||
|
||||
/* Last pixel component value in this row of the original image */
|
||||
outptr[2 * downsampled_width - 1] =
|
||||
GETJSAMPLE(inptr[downsampled_width - 1]);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* The diagram below shows an array of samples produced by h2v2 downsampling.
|
||||
*
|
||||
* s0 s1 s2
|
||||
* +---------+---------+---------+
|
||||
* | p0 p1 | p2 p3 | p4 p5 |
|
||||
* sA | | | |
|
||||
* | p6 p7 | p8 p9 | p10 p11|
|
||||
* +---------+---------+---------+
|
||||
* | p12 p13| p14 p15| p16 p17|
|
||||
* sB | | | |
|
||||
* | p18 p19| p20 p21| p22 p23|
|
||||
* +---------+---------+---------+
|
||||
* | p24 p25| p26 p27| p28 p29|
|
||||
* sC | | | |
|
||||
* | p30 p31| p32 p33| p34 p35|
|
||||
* +---------+---------+---------+
|
||||
*
|
||||
* Samples s0A-s2C were created by averaging the original pixel component
|
||||
* values centered at positions p0-p35 above. To approximate one of those
|
||||
* original pixel component values, we proportionally blend the sample
|
||||
* containing the pixel center with the nearest neighboring samples in each
|
||||
* row, column, and diagonal.
|
||||
*
|
||||
* An upsampled pixel component value is computed by first blending the sample
|
||||
* containing the pixel center with the nearest neighboring samples in the
|
||||
* same column, in the ratio 3:1, and then blending each column sum with the
|
||||
* nearest neighboring column sum, in the ratio 3:1. For example:
|
||||
* p14(upsampled) = 3/4 * (3/4 * s1B + 1/4 * s1A) +
|
||||
* 1/4 * (3/4 * s0B + 1/4 * s0A)
|
||||
* = 9/16 * s1B + 3/16 * s1A + 3/16 * s0B + 1/16 * s0A
|
||||
* When computing the first and last pixel component values in the row, there
|
||||
* is no horizontally adjacent sample to blend, so:
|
||||
* p12(upsampled) = 3/4 * s0B + 1/4 * s0A
|
||||
* p23(upsampled) = 3/4 * s2B + 1/4 * s2C
|
||||
* When computing the first and last pixel component values in the column,
|
||||
* there is no vertically adjacent sample to blend, so:
|
||||
* p2(upsampled) = 3/4 * s1A + 1/4 * s0A
|
||||
* p33(upsampled) = 3/4 * s1C + 1/4 * s2C
|
||||
* When computing the corner pixel component values, there is no adjacent
|
||||
* sample to blend, so:
|
||||
* p0(upsampled) = s0A
|
||||
* p35(upsampled) = s2C
|
||||
*/
|
||||
|
||||
void jsimd_h2v2_fancy_upsample_neon(int max_v_samp_factor,
|
||||
JDIMENSION downsampled_width,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
JSAMPARRAY output_data = *output_data_ptr;
|
||||
JSAMPROW inptr0, inptr1, inptr2, outptr0, outptr1;
|
||||
int inrow, outrow;
|
||||
unsigned colctr;
|
||||
/* Set up constants. */
|
||||
const uint16x8_t seven_u16 = vdupq_n_u16(7);
|
||||
const uint8x8_t three_u8 = vdup_n_u8(3);
|
||||
const uint16x8_t three_u16 = vdupq_n_u16(3);
|
||||
|
||||
inrow = outrow = 0;
|
||||
while (outrow < max_v_samp_factor) {
|
||||
inptr0 = input_data[inrow - 1];
|
||||
inptr1 = input_data[inrow];
|
||||
inptr2 = input_data[inrow + 1];
|
||||
/* Suffixes 0 and 1 denote the upper and lower rows of output pixels,
|
||||
* respectively.
|
||||
*/
|
||||
outptr0 = output_data[outrow++];
|
||||
outptr1 = output_data[outrow++];
|
||||
|
||||
/* First pixel component value in this row of the original image */
|
||||
int s0colsum0 = GETJSAMPLE(*inptr1) * 3 + GETJSAMPLE(*inptr0);
|
||||
*outptr0 = (JSAMPLE)((s0colsum0 * 4 + 8) >> 4);
|
||||
int s0colsum1 = GETJSAMPLE(*inptr1) * 3 + GETJSAMPLE(*inptr2);
|
||||
*outptr1 = (JSAMPLE)((s0colsum1 * 4 + 8) >> 4);
|
||||
|
||||
/* Step 1: Blend samples vertically in columns s0 and s1.
|
||||
* Leave the divide by 4 until the end, when it can be done for both
|
||||
* dimensions at once, right-shifting by 4.
|
||||
*/
|
||||
|
||||
/* Load and compute s0colsum0 and s0colsum1. */
|
||||
uint8x16_t s0A = vld1q_u8(inptr0);
|
||||
uint8x16_t s0B = vld1q_u8(inptr1);
|
||||
uint8x16_t s0C = vld1q_u8(inptr2);
|
||||
/* Multiplication makes vectors twice as wide. '_l' and '_h' suffixes
|
||||
* denote low half and high half respectively.
|
||||
*/
|
||||
uint16x8_t s0colsum0_l = vmlal_u8(vmovl_u8(vget_low_u8(s0A)),
|
||||
vget_low_u8(s0B), three_u8);
|
||||
uint16x8_t s0colsum0_h = vmlal_u8(vmovl_u8(vget_high_u8(s0A)),
|
||||
vget_high_u8(s0B), three_u8);
|
||||
uint16x8_t s0colsum1_l = vmlal_u8(vmovl_u8(vget_low_u8(s0C)),
|
||||
vget_low_u8(s0B), three_u8);
|
||||
uint16x8_t s0colsum1_h = vmlal_u8(vmovl_u8(vget_high_u8(s0C)),
|
||||
vget_high_u8(s0B), three_u8);
|
||||
/* Load and compute s1colsum0 and s1colsum1. */
|
||||
uint8x16_t s1A = vld1q_u8(inptr0 + 1);
|
||||
uint8x16_t s1B = vld1q_u8(inptr1 + 1);
|
||||
uint8x16_t s1C = vld1q_u8(inptr2 + 1);
|
||||
uint16x8_t s1colsum0_l = vmlal_u8(vmovl_u8(vget_low_u8(s1A)),
|
||||
vget_low_u8(s1B), three_u8);
|
||||
uint16x8_t s1colsum0_h = vmlal_u8(vmovl_u8(vget_high_u8(s1A)),
|
||||
vget_high_u8(s1B), three_u8);
|
||||
uint16x8_t s1colsum1_l = vmlal_u8(vmovl_u8(vget_low_u8(s1C)),
|
||||
vget_low_u8(s1B), three_u8);
|
||||
uint16x8_t s1colsum1_h = vmlal_u8(vmovl_u8(vget_high_u8(s1C)),
|
||||
vget_high_u8(s1B), three_u8);
|
||||
|
||||
/* Step 2: Blend the already-blended columns. */
|
||||
|
||||
uint16x8_t output0_p1_l = vmlaq_u16(s1colsum0_l, s0colsum0_l, three_u16);
|
||||
uint16x8_t output0_p1_h = vmlaq_u16(s1colsum0_h, s0colsum0_h, three_u16);
|
||||
uint16x8_t output0_p2_l = vmlaq_u16(s0colsum0_l, s1colsum0_l, three_u16);
|
||||
uint16x8_t output0_p2_h = vmlaq_u16(s0colsum0_h, s1colsum0_h, three_u16);
|
||||
uint16x8_t output1_p1_l = vmlaq_u16(s1colsum1_l, s0colsum1_l, three_u16);
|
||||
uint16x8_t output1_p1_h = vmlaq_u16(s1colsum1_h, s0colsum1_h, three_u16);
|
||||
uint16x8_t output1_p2_l = vmlaq_u16(s0colsum1_l, s1colsum1_l, three_u16);
|
||||
uint16x8_t output1_p2_h = vmlaq_u16(s0colsum1_h, s1colsum1_h, three_u16);
|
||||
/* Add ordered dithering bias to odd pixel values. */
|
||||
output0_p1_l = vaddq_u16(output0_p1_l, seven_u16);
|
||||
output0_p1_h = vaddq_u16(output0_p1_h, seven_u16);
|
||||
output1_p1_l = vaddq_u16(output1_p1_l, seven_u16);
|
||||
output1_p1_h = vaddq_u16(output1_p1_h, seven_u16);
|
||||
/* Right-shift by 4 (divide by 16), narrow to 8-bit, and combine. */
|
||||
uint8x16x2_t output_pixels0 = { {
|
||||
vcombine_u8(vshrn_n_u16(output0_p1_l, 4), vshrn_n_u16(output0_p1_h, 4)),
|
||||
vcombine_u8(vrshrn_n_u16(output0_p2_l, 4), vrshrn_n_u16(output0_p2_h, 4))
|
||||
} };
|
||||
uint8x16x2_t output_pixels1 = { {
|
||||
vcombine_u8(vshrn_n_u16(output1_p1_l, 4), vshrn_n_u16(output1_p1_h, 4)),
|
||||
vcombine_u8(vrshrn_n_u16(output1_p2_l, 4), vrshrn_n_u16(output1_p2_h, 4))
|
||||
} };
|
||||
|
||||
/* Store pixel component values to memory.
|
||||
* The minimum size of the output buffer for each row is 64 bytes => no
|
||||
* need to worry about buffer overflow here. See "Creation of 2-D sample
|
||||
* arrays" in jmemmgr.c for more details.
|
||||
*/
|
||||
vst2q_u8(outptr0 + 1, output_pixels0);
|
||||
vst2q_u8(outptr1 + 1, output_pixels1);
|
||||
|
||||
/* The first pixel of the image shifted our loads and stores by one byte.
|
||||
* We have to re-align on a 32-byte boundary at some point before the end
|
||||
* of the row (we do it now on the 32/33 pixel boundary) to stay within the
|
||||
* bounds of the sample buffers without having to resort to a slow scalar
|
||||
* tail case for the last (downsampled_width % 16) samples. See "Creation
|
||||
* of 2-D sample arrays" in jmemmgr.c for more details.
|
||||
*/
|
||||
for (colctr = 16; colctr < downsampled_width; colctr += 16) {
|
||||
/* Step 1: Blend samples vertically in columns s0 and s1. */
|
||||
|
||||
/* Load and compute s0colsum0 and s0colsum1. */
|
||||
s0A = vld1q_u8(inptr0 + colctr - 1);
|
||||
s0B = vld1q_u8(inptr1 + colctr - 1);
|
||||
s0C = vld1q_u8(inptr2 + colctr - 1);
|
||||
s0colsum0_l = vmlal_u8(vmovl_u8(vget_low_u8(s0A)), vget_low_u8(s0B),
|
||||
three_u8);
|
||||
s0colsum0_h = vmlal_u8(vmovl_u8(vget_high_u8(s0A)), vget_high_u8(s0B),
|
||||
three_u8);
|
||||
s0colsum1_l = vmlal_u8(vmovl_u8(vget_low_u8(s0C)), vget_low_u8(s0B),
|
||||
three_u8);
|
||||
s0colsum1_h = vmlal_u8(vmovl_u8(vget_high_u8(s0C)), vget_high_u8(s0B),
|
||||
three_u8);
|
||||
/* Load and compute s1colsum0 and s1colsum1. */
|
||||
s1A = vld1q_u8(inptr0 + colctr);
|
||||
s1B = vld1q_u8(inptr1 + colctr);
|
||||
s1C = vld1q_u8(inptr2 + colctr);
|
||||
s1colsum0_l = vmlal_u8(vmovl_u8(vget_low_u8(s1A)), vget_low_u8(s1B),
|
||||
three_u8);
|
||||
s1colsum0_h = vmlal_u8(vmovl_u8(vget_high_u8(s1A)), vget_high_u8(s1B),
|
||||
three_u8);
|
||||
s1colsum1_l = vmlal_u8(vmovl_u8(vget_low_u8(s1C)), vget_low_u8(s1B),
|
||||
three_u8);
|
||||
s1colsum1_h = vmlal_u8(vmovl_u8(vget_high_u8(s1C)), vget_high_u8(s1B),
|
||||
three_u8);
|
||||
|
||||
/* Step 2: Blend the already-blended columns. */
|
||||
|
||||
output0_p1_l = vmlaq_u16(s1colsum0_l, s0colsum0_l, three_u16);
|
||||
output0_p1_h = vmlaq_u16(s1colsum0_h, s0colsum0_h, three_u16);
|
||||
output0_p2_l = vmlaq_u16(s0colsum0_l, s1colsum0_l, three_u16);
|
||||
output0_p2_h = vmlaq_u16(s0colsum0_h, s1colsum0_h, three_u16);
|
||||
output1_p1_l = vmlaq_u16(s1colsum1_l, s0colsum1_l, three_u16);
|
||||
output1_p1_h = vmlaq_u16(s1colsum1_h, s0colsum1_h, three_u16);
|
||||
output1_p2_l = vmlaq_u16(s0colsum1_l, s1colsum1_l, three_u16);
|
||||
output1_p2_h = vmlaq_u16(s0colsum1_h, s1colsum1_h, three_u16);
|
||||
/* Add ordered dithering bias to odd pixel values. */
|
||||
output0_p1_l = vaddq_u16(output0_p1_l, seven_u16);
|
||||
output0_p1_h = vaddq_u16(output0_p1_h, seven_u16);
|
||||
output1_p1_l = vaddq_u16(output1_p1_l, seven_u16);
|
||||
output1_p1_h = vaddq_u16(output1_p1_h, seven_u16);
|
||||
/* Right-shift by 4 (divide by 16), narrow to 8-bit, and combine. */
|
||||
output_pixels0.val[0] = vcombine_u8(vshrn_n_u16(output0_p1_l, 4),
|
||||
vshrn_n_u16(output0_p1_h, 4));
|
||||
output_pixels0.val[1] = vcombine_u8(vrshrn_n_u16(output0_p2_l, 4),
|
||||
vrshrn_n_u16(output0_p2_h, 4));
|
||||
output_pixels1.val[0] = vcombine_u8(vshrn_n_u16(output1_p1_l, 4),
|
||||
vshrn_n_u16(output1_p1_h, 4));
|
||||
output_pixels1.val[1] = vcombine_u8(vrshrn_n_u16(output1_p2_l, 4),
|
||||
vrshrn_n_u16(output1_p2_h, 4));
|
||||
/* Store pixel component values to memory. */
|
||||
vst2q_u8(outptr0 + 2 * colctr - 1, output_pixels0);
|
||||
vst2q_u8(outptr1 + 2 * colctr - 1, output_pixels1);
|
||||
}
|
||||
|
||||
/* Last pixel component value in this row of the original image */
|
||||
int s1colsum0 = GETJSAMPLE(inptr1[downsampled_width - 1]) * 3 +
|
||||
GETJSAMPLE(inptr0[downsampled_width - 1]);
|
||||
outptr0[2 * downsampled_width - 1] = (JSAMPLE)((s1colsum0 * 4 + 7) >> 4);
|
||||
int s1colsum1 = GETJSAMPLE(inptr1[downsampled_width - 1]) * 3 +
|
||||
GETJSAMPLE(inptr2[downsampled_width - 1]);
|
||||
outptr1[2 * downsampled_width - 1] = (JSAMPLE)((s1colsum1 * 4 + 7) >> 4);
|
||||
inrow++;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* The diagram below shows a column of samples produced by h1v2 downsampling
|
||||
* (or by losslessly rotating or transposing an h2v1-downsampled image.)
|
||||
*
|
||||
* +---------+
|
||||
* | p0 |
|
||||
* sA | |
|
||||
* | p1 |
|
||||
* +---------+
|
||||
* | p2 |
|
||||
* sB | |
|
||||
* | p3 |
|
||||
* +---------+
|
||||
* | p4 |
|
||||
* sC | |
|
||||
* | p5 |
|
||||
* +---------+
|
||||
*
|
||||
* Samples sA-sC were created by averaging the original pixel component values
|
||||
* centered at positions p0-p5 above. To approximate those original pixel
|
||||
* component values, we proportionally blend the adjacent samples in each
|
||||
* column.
|
||||
*
|
||||
* An upsampled pixel component value is computed by blending the sample
|
||||
* containing the pixel center with the nearest neighboring sample, in the
|
||||
* ratio 3:1. For example:
|
||||
* p1(upsampled) = 3/4 * sA + 1/4 * sB
|
||||
* p2(upsampled) = 3/4 * sB + 1/4 * sA
|
||||
* When computing the first and last pixel component values in the column,
|
||||
* there is no adjacent sample to blend, so:
|
||||
* p0(upsampled) = sA
|
||||
* p5(upsampled) = sC
|
||||
*/
|
||||
|
||||
void jsimd_h1v2_fancy_upsample_neon(int max_v_samp_factor,
|
||||
JDIMENSION downsampled_width,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
JSAMPARRAY output_data = *output_data_ptr;
|
||||
JSAMPROW inptr0, inptr1, inptr2, outptr0, outptr1;
|
||||
int inrow, outrow;
|
||||
unsigned colctr;
|
||||
/* Set up constants. */
|
||||
const uint16x8_t one_u16 = vdupq_n_u16(1);
|
||||
const uint8x8_t three_u8 = vdup_n_u8(3);
|
||||
|
||||
inrow = outrow = 0;
|
||||
while (outrow < max_v_samp_factor) {
|
||||
inptr0 = input_data[inrow - 1];
|
||||
inptr1 = input_data[inrow];
|
||||
inptr2 = input_data[inrow + 1];
|
||||
/* Suffixes 0 and 1 denote the upper and lower rows of output pixels,
|
||||
* respectively.
|
||||
*/
|
||||
outptr0 = output_data[outrow++];
|
||||
outptr1 = output_data[outrow++];
|
||||
inrow++;
|
||||
|
||||
/* The size of the input and output buffers is always a multiple of 32
|
||||
* bytes => no need to worry about buffer overflow when reading/writing
|
||||
* memory. See "Creation of 2-D sample arrays" in jmemmgr.c for more
|
||||
* details.
|
||||
*/
|
||||
for (colctr = 0; colctr < downsampled_width; colctr += 16) {
|
||||
/* Load samples. */
|
||||
uint8x16_t sA = vld1q_u8(inptr0 + colctr);
|
||||
uint8x16_t sB = vld1q_u8(inptr1 + colctr);
|
||||
uint8x16_t sC = vld1q_u8(inptr2 + colctr);
|
||||
/* Blend samples vertically. */
|
||||
uint16x8_t colsum0_l = vmlal_u8(vmovl_u8(vget_low_u8(sA)),
|
||||
vget_low_u8(sB), three_u8);
|
||||
uint16x8_t colsum0_h = vmlal_u8(vmovl_u8(vget_high_u8(sA)),
|
||||
vget_high_u8(sB), three_u8);
|
||||
uint16x8_t colsum1_l = vmlal_u8(vmovl_u8(vget_low_u8(sC)),
|
||||
vget_low_u8(sB), three_u8);
|
||||
uint16x8_t colsum1_h = vmlal_u8(vmovl_u8(vget_high_u8(sC)),
|
||||
vget_high_u8(sB), three_u8);
|
||||
/* Add ordered dithering bias to pixel values in even output rows. */
|
||||
colsum0_l = vaddq_u16(colsum0_l, one_u16);
|
||||
colsum0_h = vaddq_u16(colsum0_h, one_u16);
|
||||
/* Right-shift by 2 (divide by 4), narrow to 8-bit, and combine. */
|
||||
uint8x16_t output_pixels0 = vcombine_u8(vshrn_n_u16(colsum0_l, 2),
|
||||
vshrn_n_u16(colsum0_h, 2));
|
||||
uint8x16_t output_pixels1 = vcombine_u8(vrshrn_n_u16(colsum1_l, 2),
|
||||
vrshrn_n_u16(colsum1_h, 2));
|
||||
/* Store pixel component values to memory. */
|
||||
vst1q_u8(outptr0 + colctr, output_pixels0);
|
||||
vst1q_u8(outptr1 + colctr, output_pixels1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* The diagram below shows a row of samples produced by h2v1 downsampling.
|
||||
*
|
||||
* s0 s1
|
||||
* +---------+---------+
|
||||
* | | |
|
||||
* | p0 p1 | p2 p3 |
|
||||
* | | |
|
||||
* +---------+---------+
|
||||
*
|
||||
* Samples s0 and s1 were created by averaging the original pixel component
|
||||
* values centered at positions p0-p3 above. To approximate those original
|
||||
* pixel component values, we duplicate the samples horizontally:
|
||||
* p0(upsampled) = p1(upsampled) = s0
|
||||
* p2(upsampled) = p3(upsampled) = s1
|
||||
*/
|
||||
|
||||
void jsimd_h2v1_upsample_neon(int max_v_samp_factor, JDIMENSION output_width,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
JSAMPARRAY output_data = *output_data_ptr;
|
||||
JSAMPROW inptr, outptr;
|
||||
int inrow;
|
||||
unsigned colctr;
|
||||
|
||||
for (inrow = 0; inrow < max_v_samp_factor; inrow++) {
|
||||
inptr = input_data[inrow];
|
||||
outptr = output_data[inrow];
|
||||
for (colctr = 0; 2 * colctr < output_width; colctr += 16) {
|
||||
uint8x16_t samples = vld1q_u8(inptr + colctr);
|
||||
/* Duplicate the samples. The store operation below interleaves them so
|
||||
* that adjacent pixel component values take on the same sample value,
|
||||
* per above.
|
||||
*/
|
||||
uint8x16x2_t output_pixels = { { samples, samples } };
|
||||
/* Store pixel component values to memory.
|
||||
* Due to the way sample buffers are allocated, we don't need to worry
|
||||
* about tail cases when output_width is not a multiple of 32. See
|
||||
* "Creation of 2-D sample arrays" in jmemmgr.c for details.
|
||||
*/
|
||||
vst2q_u8(outptr + 2 * colctr, output_pixels);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* The diagram below shows an array of samples produced by h2v2 downsampling.
|
||||
*
|
||||
* s0 s1
|
||||
* +---------+---------+
|
||||
* | p0 p1 | p2 p3 |
|
||||
* sA | | |
|
||||
* | p4 p5 | p6 p7 |
|
||||
* +---------+---------+
|
||||
* | p8 p9 | p10 p11|
|
||||
* sB | | |
|
||||
* | p12 p13| p14 p15|
|
||||
* +---------+---------+
|
||||
*
|
||||
* Samples s0A-s1B were created by averaging the original pixel component
|
||||
* values centered at positions p0-p15 above. To approximate those original
|
||||
* pixel component values, we duplicate the samples both horizontally and
|
||||
* vertically:
|
||||
* p0(upsampled) = p1(upsampled) = p4(upsampled) = p5(upsampled) = s0A
|
||||
* p2(upsampled) = p3(upsampled) = p6(upsampled) = p7(upsampled) = s1A
|
||||
* p8(upsampled) = p9(upsampled) = p12(upsampled) = p13(upsampled) = s0B
|
||||
* p10(upsampled) = p11(upsampled) = p14(upsampled) = p15(upsampled) = s1B
|
||||
*/
|
||||
|
||||
void jsimd_h2v2_upsample_neon(int max_v_samp_factor, JDIMENSION output_width,
|
||||
JSAMPARRAY input_data,
|
||||
JSAMPARRAY *output_data_ptr)
|
||||
{
|
||||
JSAMPARRAY output_data = *output_data_ptr;
|
||||
JSAMPROW inptr, outptr0, outptr1;
|
||||
int inrow, outrow;
|
||||
unsigned colctr;
|
||||
|
||||
for (inrow = 0, outrow = 0; outrow < max_v_samp_factor; inrow++) {
|
||||
inptr = input_data[inrow];
|
||||
outptr0 = output_data[outrow++];
|
||||
outptr1 = output_data[outrow++];
|
||||
|
||||
for (colctr = 0; 2 * colctr < output_width; colctr += 16) {
|
||||
uint8x16_t samples = vld1q_u8(inptr + colctr);
|
||||
/* Duplicate the samples. The store operation below interleaves them so
|
||||
* that adjacent pixel component values take on the same sample value,
|
||||
* per above.
|
||||
*/
|
||||
uint8x16x2_t output_pixels = { { samples, samples } };
|
||||
/* Store pixel component values for both output rows to memory.
|
||||
* Due to the way sample buffers are allocated, we don't need to worry
|
||||
* about tail cases when output_width is not a multiple of 32. See
|
||||
* "Creation of 2-D sample arrays" in jmemmgr.c for details.
|
||||
*/
|
||||
vst2q_u8(outptr0 + 2 * colctr, output_pixels);
|
||||
vst2q_u8(outptr1 + 2 * colctr, output_pixels);
|
||||
}
|
||||
}
|
||||
}
|
||||
+214
@@ -0,0 +1,214 @@
|
||||
/*
|
||||
* jfdctfst-neon.c - fast integer FDCT (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* jsimd_fdct_ifast_neon() performs a fast, not so accurate forward DCT
|
||||
* (Discrete Cosine Transform) on one block of samples. It uses the same
|
||||
* calculations and produces exactly the same output as IJG's original
|
||||
* jpeg_fdct_ifast() function, which can be found in jfdctfst.c.
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.382683433 = 12544 * 2^-15
|
||||
* 0.541196100 = 17795 * 2^-15
|
||||
* 0.707106781 = 23168 * 2^-15
|
||||
* 0.306562965 = 9984 * 2^-15
|
||||
*
|
||||
* See jfdctfst.c for further details of the DCT algorithm. Where possible,
|
||||
* the variable names and comments here in jsimd_fdct_ifast_neon() match up
|
||||
* with those in jpeg_fdct_ifast().
|
||||
*/
|
||||
|
||||
#define F_0_382 12544
|
||||
#define F_0_541 17792
|
||||
#define F_0_707 23168
|
||||
#define F_0_306 9984
|
||||
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_fdct_ifast_neon_consts[] = {
|
||||
F_0_382, F_0_541, F_0_707, F_0_306
|
||||
};
|
||||
|
||||
void jsimd_fdct_ifast_neon(DCTELEM *data)
|
||||
{
|
||||
/* Load an 8x8 block of samples into Neon registers. De-interleaving loads
|
||||
* are used, followed by vuzp to transpose the block such that we have a
|
||||
* column of samples per vector - allowing all rows to be processed at once.
|
||||
*/
|
||||
int16x8x4_t data1 = vld4q_s16(data);
|
||||
int16x8x4_t data2 = vld4q_s16(data + 4 * DCTSIZE);
|
||||
|
||||
int16x8x2_t cols_04 = vuzpq_s16(data1.val[0], data2.val[0]);
|
||||
int16x8x2_t cols_15 = vuzpq_s16(data1.val[1], data2.val[1]);
|
||||
int16x8x2_t cols_26 = vuzpq_s16(data1.val[2], data2.val[2]);
|
||||
int16x8x2_t cols_37 = vuzpq_s16(data1.val[3], data2.val[3]);
|
||||
|
||||
int16x8_t col0 = cols_04.val[0];
|
||||
int16x8_t col1 = cols_15.val[0];
|
||||
int16x8_t col2 = cols_26.val[0];
|
||||
int16x8_t col3 = cols_37.val[0];
|
||||
int16x8_t col4 = cols_04.val[1];
|
||||
int16x8_t col5 = cols_15.val[1];
|
||||
int16x8_t col6 = cols_26.val[1];
|
||||
int16x8_t col7 = cols_37.val[1];
|
||||
|
||||
/* Pass 1: process rows. */
|
||||
|
||||
/* Load DCT conversion constants. */
|
||||
const int16x4_t consts = vld1_s16(jsimd_fdct_ifast_neon_consts);
|
||||
|
||||
int16x8_t tmp0 = vaddq_s16(col0, col7);
|
||||
int16x8_t tmp7 = vsubq_s16(col0, col7);
|
||||
int16x8_t tmp1 = vaddq_s16(col1, col6);
|
||||
int16x8_t tmp6 = vsubq_s16(col1, col6);
|
||||
int16x8_t tmp2 = vaddq_s16(col2, col5);
|
||||
int16x8_t tmp5 = vsubq_s16(col2, col5);
|
||||
int16x8_t tmp3 = vaddq_s16(col3, col4);
|
||||
int16x8_t tmp4 = vsubq_s16(col3, col4);
|
||||
|
||||
/* Even part */
|
||||
int16x8_t tmp10 = vaddq_s16(tmp0, tmp3); /* phase 2 */
|
||||
int16x8_t tmp13 = vsubq_s16(tmp0, tmp3);
|
||||
int16x8_t tmp11 = vaddq_s16(tmp1, tmp2);
|
||||
int16x8_t tmp12 = vsubq_s16(tmp1, tmp2);
|
||||
|
||||
col0 = vaddq_s16(tmp10, tmp11); /* phase 3 */
|
||||
col4 = vsubq_s16(tmp10, tmp11);
|
||||
|
||||
int16x8_t z1 = vqdmulhq_lane_s16(vaddq_s16(tmp12, tmp13), consts, 2);
|
||||
col2 = vaddq_s16(tmp13, z1); /* phase 5 */
|
||||
col6 = vsubq_s16(tmp13, z1);
|
||||
|
||||
/* Odd part */
|
||||
tmp10 = vaddq_s16(tmp4, tmp5); /* phase 2 */
|
||||
tmp11 = vaddq_s16(tmp5, tmp6);
|
||||
tmp12 = vaddq_s16(tmp6, tmp7);
|
||||
|
||||
int16x8_t z5 = vqdmulhq_lane_s16(vsubq_s16(tmp10, tmp12), consts, 0);
|
||||
int16x8_t z2 = vqdmulhq_lane_s16(tmp10, consts, 1);
|
||||
z2 = vaddq_s16(z2, z5);
|
||||
int16x8_t z4 = vqdmulhq_lane_s16(tmp12, consts, 3);
|
||||
z5 = vaddq_s16(tmp12, z5);
|
||||
z4 = vaddq_s16(z4, z5);
|
||||
int16x8_t z3 = vqdmulhq_lane_s16(tmp11, consts, 2);
|
||||
|
||||
int16x8_t z11 = vaddq_s16(tmp7, z3); /* phase 5 */
|
||||
int16x8_t z13 = vsubq_s16(tmp7, z3);
|
||||
|
||||
col5 = vaddq_s16(z13, z2); /* phase 6 */
|
||||
col3 = vsubq_s16(z13, z2);
|
||||
col1 = vaddq_s16(z11, z4);
|
||||
col7 = vsubq_s16(z11, z4);
|
||||
|
||||
/* Transpose to work on columns in pass 2. */
|
||||
int16x8x2_t cols_01 = vtrnq_s16(col0, col1);
|
||||
int16x8x2_t cols_23 = vtrnq_s16(col2, col3);
|
||||
int16x8x2_t cols_45 = vtrnq_s16(col4, col5);
|
||||
int16x8x2_t cols_67 = vtrnq_s16(col6, col7);
|
||||
|
||||
int32x4x2_t cols_0145_l = vtrnq_s32(vreinterpretq_s32_s16(cols_01.val[0]),
|
||||
vreinterpretq_s32_s16(cols_45.val[0]));
|
||||
int32x4x2_t cols_0145_h = vtrnq_s32(vreinterpretq_s32_s16(cols_01.val[1]),
|
||||
vreinterpretq_s32_s16(cols_45.val[1]));
|
||||
int32x4x2_t cols_2367_l = vtrnq_s32(vreinterpretq_s32_s16(cols_23.val[0]),
|
||||
vreinterpretq_s32_s16(cols_67.val[0]));
|
||||
int32x4x2_t cols_2367_h = vtrnq_s32(vreinterpretq_s32_s16(cols_23.val[1]),
|
||||
vreinterpretq_s32_s16(cols_67.val[1]));
|
||||
|
||||
int32x4x2_t rows_04 = vzipq_s32(cols_0145_l.val[0], cols_2367_l.val[0]);
|
||||
int32x4x2_t rows_15 = vzipq_s32(cols_0145_h.val[0], cols_2367_h.val[0]);
|
||||
int32x4x2_t rows_26 = vzipq_s32(cols_0145_l.val[1], cols_2367_l.val[1]);
|
||||
int32x4x2_t rows_37 = vzipq_s32(cols_0145_h.val[1], cols_2367_h.val[1]);
|
||||
|
||||
int16x8_t row0 = vreinterpretq_s16_s32(rows_04.val[0]);
|
||||
int16x8_t row1 = vreinterpretq_s16_s32(rows_15.val[0]);
|
||||
int16x8_t row2 = vreinterpretq_s16_s32(rows_26.val[0]);
|
||||
int16x8_t row3 = vreinterpretq_s16_s32(rows_37.val[0]);
|
||||
int16x8_t row4 = vreinterpretq_s16_s32(rows_04.val[1]);
|
||||
int16x8_t row5 = vreinterpretq_s16_s32(rows_15.val[1]);
|
||||
int16x8_t row6 = vreinterpretq_s16_s32(rows_26.val[1]);
|
||||
int16x8_t row7 = vreinterpretq_s16_s32(rows_37.val[1]);
|
||||
|
||||
/* Pass 2: process columns. */
|
||||
|
||||
tmp0 = vaddq_s16(row0, row7);
|
||||
tmp7 = vsubq_s16(row0, row7);
|
||||
tmp1 = vaddq_s16(row1, row6);
|
||||
tmp6 = vsubq_s16(row1, row6);
|
||||
tmp2 = vaddq_s16(row2, row5);
|
||||
tmp5 = vsubq_s16(row2, row5);
|
||||
tmp3 = vaddq_s16(row3, row4);
|
||||
tmp4 = vsubq_s16(row3, row4);
|
||||
|
||||
/* Even part */
|
||||
tmp10 = vaddq_s16(tmp0, tmp3); /* phase 2 */
|
||||
tmp13 = vsubq_s16(tmp0, tmp3);
|
||||
tmp11 = vaddq_s16(tmp1, tmp2);
|
||||
tmp12 = vsubq_s16(tmp1, tmp2);
|
||||
|
||||
row0 = vaddq_s16(tmp10, tmp11); /* phase 3 */
|
||||
row4 = vsubq_s16(tmp10, tmp11);
|
||||
|
||||
z1 = vqdmulhq_lane_s16(vaddq_s16(tmp12, tmp13), consts, 2);
|
||||
row2 = vaddq_s16(tmp13, z1); /* phase 5 */
|
||||
row6 = vsubq_s16(tmp13, z1);
|
||||
|
||||
/* Odd part */
|
||||
tmp10 = vaddq_s16(tmp4, tmp5); /* phase 2 */
|
||||
tmp11 = vaddq_s16(tmp5, tmp6);
|
||||
tmp12 = vaddq_s16(tmp6, tmp7);
|
||||
|
||||
z5 = vqdmulhq_lane_s16(vsubq_s16(tmp10, tmp12), consts, 0);
|
||||
z2 = vqdmulhq_lane_s16(tmp10, consts, 1);
|
||||
z2 = vaddq_s16(z2, z5);
|
||||
z4 = vqdmulhq_lane_s16(tmp12, consts, 3);
|
||||
z5 = vaddq_s16(tmp12, z5);
|
||||
z4 = vaddq_s16(z4, z5);
|
||||
z3 = vqdmulhq_lane_s16(tmp11, consts, 2);
|
||||
|
||||
z11 = vaddq_s16(tmp7, z3); /* phase 5 */
|
||||
z13 = vsubq_s16(tmp7, z3);
|
||||
|
||||
row5 = vaddq_s16(z13, z2); /* phase 6 */
|
||||
row3 = vsubq_s16(z13, z2);
|
||||
row1 = vaddq_s16(z11, z4);
|
||||
row7 = vsubq_s16(z11, z4);
|
||||
|
||||
vst1q_s16(data + 0 * DCTSIZE, row0);
|
||||
vst1q_s16(data + 1 * DCTSIZE, row1);
|
||||
vst1q_s16(data + 2 * DCTSIZE, row2);
|
||||
vst1q_s16(data + 3 * DCTSIZE, row3);
|
||||
vst1q_s16(data + 4 * DCTSIZE, row4);
|
||||
vst1q_s16(data + 5 * DCTSIZE, row5);
|
||||
vst1q_s16(data + 6 * DCTSIZE, row6);
|
||||
vst1q_s16(data + 7 * DCTSIZE, row7);
|
||||
}
|
||||
+376
@@ -0,0 +1,376 @@
|
||||
/*
|
||||
* jfdctint-neon.c - accurate integer FDCT (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* jsimd_fdct_islow_neon() performs a slower but more accurate forward DCT
|
||||
* (Discrete Cosine Transform) on one block of samples. It uses the same
|
||||
* calculations and produces exactly the same output as IJG's original
|
||||
* jpeg_fdct_islow() function, which can be found in jfdctint.c.
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.298631336 = 2446 * 2^-13
|
||||
* 0.390180644 = 3196 * 2^-13
|
||||
* 0.541196100 = 4433 * 2^-13
|
||||
* 0.765366865 = 6270 * 2^-13
|
||||
* 0.899976223 = 7373 * 2^-13
|
||||
* 1.175875602 = 9633 * 2^-13
|
||||
* 1.501321110 = 12299 * 2^-13
|
||||
* 1.847759065 = 15137 * 2^-13
|
||||
* 1.961570560 = 16069 * 2^-13
|
||||
* 2.053119869 = 16819 * 2^-13
|
||||
* 2.562915447 = 20995 * 2^-13
|
||||
* 3.072711026 = 25172 * 2^-13
|
||||
*
|
||||
* See jfdctint.c for further details of the DCT algorithm. Where possible,
|
||||
* the variable names and comments here in jsimd_fdct_islow_neon() match up
|
||||
* with those in jpeg_fdct_islow().
|
||||
*/
|
||||
|
||||
#define CONST_BITS 13
|
||||
#define PASS1_BITS 2
|
||||
|
||||
#define DESCALE_P1 (CONST_BITS - PASS1_BITS)
|
||||
#define DESCALE_P2 (CONST_BITS + PASS1_BITS)
|
||||
|
||||
#define F_0_298 2446
|
||||
#define F_0_390 3196
|
||||
#define F_0_541 4433
|
||||
#define F_0_765 6270
|
||||
#define F_0_899 7373
|
||||
#define F_1_175 9633
|
||||
#define F_1_501 12299
|
||||
#define F_1_847 15137
|
||||
#define F_1_961 16069
|
||||
#define F_2_053 16819
|
||||
#define F_2_562 20995
|
||||
#define F_3_072 25172
|
||||
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_fdct_islow_neon_consts[] = {
|
||||
F_0_298, -F_0_390, F_0_541, F_0_765,
|
||||
-F_0_899, F_1_175, F_1_501, -F_1_847,
|
||||
-F_1_961, F_2_053, -F_2_562, F_3_072
|
||||
};
|
||||
|
||||
void jsimd_fdct_islow_neon(DCTELEM *data)
|
||||
{
|
||||
/* Load DCT constants. */
|
||||
#ifdef HAVE_VLD1_S16_X3
|
||||
const int16x4x3_t consts = vld1_s16_x3(jsimd_fdct_islow_neon_consts);
|
||||
#else
|
||||
/* GCC does not currently support the intrinsic vld1_<type>_x3(). */
|
||||
const int16x4_t consts1 = vld1_s16(jsimd_fdct_islow_neon_consts);
|
||||
const int16x4_t consts2 = vld1_s16(jsimd_fdct_islow_neon_consts + 4);
|
||||
const int16x4_t consts3 = vld1_s16(jsimd_fdct_islow_neon_consts + 8);
|
||||
const int16x4x3_t consts = { { consts1, consts2, consts3 } };
|
||||
#endif
|
||||
|
||||
/* Load an 8x8 block of samples into Neon registers. De-interleaving loads
|
||||
* are used, followed by vuzp to transpose the block such that we have a
|
||||
* column of samples per vector - allowing all rows to be processed at once.
|
||||
*/
|
||||
int16x8x4_t s_rows_0123 = vld4q_s16(data);
|
||||
int16x8x4_t s_rows_4567 = vld4q_s16(data + 4 * DCTSIZE);
|
||||
|
||||
int16x8x2_t cols_04 = vuzpq_s16(s_rows_0123.val[0], s_rows_4567.val[0]);
|
||||
int16x8x2_t cols_15 = vuzpq_s16(s_rows_0123.val[1], s_rows_4567.val[1]);
|
||||
int16x8x2_t cols_26 = vuzpq_s16(s_rows_0123.val[2], s_rows_4567.val[2]);
|
||||
int16x8x2_t cols_37 = vuzpq_s16(s_rows_0123.val[3], s_rows_4567.val[3]);
|
||||
|
||||
int16x8_t col0 = cols_04.val[0];
|
||||
int16x8_t col1 = cols_15.val[0];
|
||||
int16x8_t col2 = cols_26.val[0];
|
||||
int16x8_t col3 = cols_37.val[0];
|
||||
int16x8_t col4 = cols_04.val[1];
|
||||
int16x8_t col5 = cols_15.val[1];
|
||||
int16x8_t col6 = cols_26.val[1];
|
||||
int16x8_t col7 = cols_37.val[1];
|
||||
|
||||
/* Pass 1: process rows. */
|
||||
|
||||
int16x8_t tmp0 = vaddq_s16(col0, col7);
|
||||
int16x8_t tmp7 = vsubq_s16(col0, col7);
|
||||
int16x8_t tmp1 = vaddq_s16(col1, col6);
|
||||
int16x8_t tmp6 = vsubq_s16(col1, col6);
|
||||
int16x8_t tmp2 = vaddq_s16(col2, col5);
|
||||
int16x8_t tmp5 = vsubq_s16(col2, col5);
|
||||
int16x8_t tmp3 = vaddq_s16(col3, col4);
|
||||
int16x8_t tmp4 = vsubq_s16(col3, col4);
|
||||
|
||||
/* Even part */
|
||||
int16x8_t tmp10 = vaddq_s16(tmp0, tmp3);
|
||||
int16x8_t tmp13 = vsubq_s16(tmp0, tmp3);
|
||||
int16x8_t tmp11 = vaddq_s16(tmp1, tmp2);
|
||||
int16x8_t tmp12 = vsubq_s16(tmp1, tmp2);
|
||||
|
||||
col0 = vshlq_n_s16(vaddq_s16(tmp10, tmp11), PASS1_BITS);
|
||||
col4 = vshlq_n_s16(vsubq_s16(tmp10, tmp11), PASS1_BITS);
|
||||
|
||||
int16x8_t tmp12_add_tmp13 = vaddq_s16(tmp12, tmp13);
|
||||
int32x4_t z1_l =
|
||||
vmull_lane_s16(vget_low_s16(tmp12_add_tmp13), consts.val[0], 2);
|
||||
int32x4_t z1_h =
|
||||
vmull_lane_s16(vget_high_s16(tmp12_add_tmp13), consts.val[0], 2);
|
||||
|
||||
int32x4_t col2_scaled_l =
|
||||
vmlal_lane_s16(z1_l, vget_low_s16(tmp13), consts.val[0], 3);
|
||||
int32x4_t col2_scaled_h =
|
||||
vmlal_lane_s16(z1_h, vget_high_s16(tmp13), consts.val[0], 3);
|
||||
col2 = vcombine_s16(vrshrn_n_s32(col2_scaled_l, DESCALE_P1),
|
||||
vrshrn_n_s32(col2_scaled_h, DESCALE_P1));
|
||||
|
||||
int32x4_t col6_scaled_l =
|
||||
vmlal_lane_s16(z1_l, vget_low_s16(tmp12), consts.val[1], 3);
|
||||
int32x4_t col6_scaled_h =
|
||||
vmlal_lane_s16(z1_h, vget_high_s16(tmp12), consts.val[1], 3);
|
||||
col6 = vcombine_s16(vrshrn_n_s32(col6_scaled_l, DESCALE_P1),
|
||||
vrshrn_n_s32(col6_scaled_h, DESCALE_P1));
|
||||
|
||||
/* Odd part */
|
||||
int16x8_t z1 = vaddq_s16(tmp4, tmp7);
|
||||
int16x8_t z2 = vaddq_s16(tmp5, tmp6);
|
||||
int16x8_t z3 = vaddq_s16(tmp4, tmp6);
|
||||
int16x8_t z4 = vaddq_s16(tmp5, tmp7);
|
||||
/* sqrt(2) * c3 */
|
||||
int32x4_t z5_l = vmull_lane_s16(vget_low_s16(z3), consts.val[1], 1);
|
||||
int32x4_t z5_h = vmull_lane_s16(vget_high_s16(z3), consts.val[1], 1);
|
||||
z5_l = vmlal_lane_s16(z5_l, vget_low_s16(z4), consts.val[1], 1);
|
||||
z5_h = vmlal_lane_s16(z5_h, vget_high_s16(z4), consts.val[1], 1);
|
||||
|
||||
/* sqrt(2) * (-c1+c3+c5-c7) */
|
||||
int32x4_t tmp4_l = vmull_lane_s16(vget_low_s16(tmp4), consts.val[0], 0);
|
||||
int32x4_t tmp4_h = vmull_lane_s16(vget_high_s16(tmp4), consts.val[0], 0);
|
||||
/* sqrt(2) * ( c1+c3-c5+c7) */
|
||||
int32x4_t tmp5_l = vmull_lane_s16(vget_low_s16(tmp5), consts.val[2], 1);
|
||||
int32x4_t tmp5_h = vmull_lane_s16(vget_high_s16(tmp5), consts.val[2], 1);
|
||||
/* sqrt(2) * ( c1+c3+c5-c7) */
|
||||
int32x4_t tmp6_l = vmull_lane_s16(vget_low_s16(tmp6), consts.val[2], 3);
|
||||
int32x4_t tmp6_h = vmull_lane_s16(vget_high_s16(tmp6), consts.val[2], 3);
|
||||
/* sqrt(2) * ( c1+c3-c5-c7) */
|
||||
int32x4_t tmp7_l = vmull_lane_s16(vget_low_s16(tmp7), consts.val[1], 2);
|
||||
int32x4_t tmp7_h = vmull_lane_s16(vget_high_s16(tmp7), consts.val[1], 2);
|
||||
|
||||
/* sqrt(2) * (c7-c3) */
|
||||
z1_l = vmull_lane_s16(vget_low_s16(z1), consts.val[1], 0);
|
||||
z1_h = vmull_lane_s16(vget_high_s16(z1), consts.val[1], 0);
|
||||
/* sqrt(2) * (-c1-c3) */
|
||||
int32x4_t z2_l = vmull_lane_s16(vget_low_s16(z2), consts.val[2], 2);
|
||||
int32x4_t z2_h = vmull_lane_s16(vget_high_s16(z2), consts.val[2], 2);
|
||||
/* sqrt(2) * (-c3-c5) */
|
||||
int32x4_t z3_l = vmull_lane_s16(vget_low_s16(z3), consts.val[2], 0);
|
||||
int32x4_t z3_h = vmull_lane_s16(vget_high_s16(z3), consts.val[2], 0);
|
||||
/* sqrt(2) * (c5-c3) */
|
||||
int32x4_t z4_l = vmull_lane_s16(vget_low_s16(z4), consts.val[0], 1);
|
||||
int32x4_t z4_h = vmull_lane_s16(vget_high_s16(z4), consts.val[0], 1);
|
||||
|
||||
z3_l = vaddq_s32(z3_l, z5_l);
|
||||
z3_h = vaddq_s32(z3_h, z5_h);
|
||||
z4_l = vaddq_s32(z4_l, z5_l);
|
||||
z4_h = vaddq_s32(z4_h, z5_h);
|
||||
|
||||
tmp4_l = vaddq_s32(tmp4_l, z1_l);
|
||||
tmp4_h = vaddq_s32(tmp4_h, z1_h);
|
||||
tmp4_l = vaddq_s32(tmp4_l, z3_l);
|
||||
tmp4_h = vaddq_s32(tmp4_h, z3_h);
|
||||
col7 = vcombine_s16(vrshrn_n_s32(tmp4_l, DESCALE_P1),
|
||||
vrshrn_n_s32(tmp4_h, DESCALE_P1));
|
||||
|
||||
tmp5_l = vaddq_s32(tmp5_l, z2_l);
|
||||
tmp5_h = vaddq_s32(tmp5_h, z2_h);
|
||||
tmp5_l = vaddq_s32(tmp5_l, z4_l);
|
||||
tmp5_h = vaddq_s32(tmp5_h, z4_h);
|
||||
col5 = vcombine_s16(vrshrn_n_s32(tmp5_l, DESCALE_P1),
|
||||
vrshrn_n_s32(tmp5_h, DESCALE_P1));
|
||||
|
||||
tmp6_l = vaddq_s32(tmp6_l, z2_l);
|
||||
tmp6_h = vaddq_s32(tmp6_h, z2_h);
|
||||
tmp6_l = vaddq_s32(tmp6_l, z3_l);
|
||||
tmp6_h = vaddq_s32(tmp6_h, z3_h);
|
||||
col3 = vcombine_s16(vrshrn_n_s32(tmp6_l, DESCALE_P1),
|
||||
vrshrn_n_s32(tmp6_h, DESCALE_P1));
|
||||
|
||||
tmp7_l = vaddq_s32(tmp7_l, z1_l);
|
||||
tmp7_h = vaddq_s32(tmp7_h, z1_h);
|
||||
tmp7_l = vaddq_s32(tmp7_l, z4_l);
|
||||
tmp7_h = vaddq_s32(tmp7_h, z4_h);
|
||||
col1 = vcombine_s16(vrshrn_n_s32(tmp7_l, DESCALE_P1),
|
||||
vrshrn_n_s32(tmp7_h, DESCALE_P1));
|
||||
|
||||
/* Transpose to work on columns in pass 2. */
|
||||
int16x8x2_t cols_01 = vtrnq_s16(col0, col1);
|
||||
int16x8x2_t cols_23 = vtrnq_s16(col2, col3);
|
||||
int16x8x2_t cols_45 = vtrnq_s16(col4, col5);
|
||||
int16x8x2_t cols_67 = vtrnq_s16(col6, col7);
|
||||
|
||||
int32x4x2_t cols_0145_l = vtrnq_s32(vreinterpretq_s32_s16(cols_01.val[0]),
|
||||
vreinterpretq_s32_s16(cols_45.val[0]));
|
||||
int32x4x2_t cols_0145_h = vtrnq_s32(vreinterpretq_s32_s16(cols_01.val[1]),
|
||||
vreinterpretq_s32_s16(cols_45.val[1]));
|
||||
int32x4x2_t cols_2367_l = vtrnq_s32(vreinterpretq_s32_s16(cols_23.val[0]),
|
||||
vreinterpretq_s32_s16(cols_67.val[0]));
|
||||
int32x4x2_t cols_2367_h = vtrnq_s32(vreinterpretq_s32_s16(cols_23.val[1]),
|
||||
vreinterpretq_s32_s16(cols_67.val[1]));
|
||||
|
||||
int32x4x2_t rows_04 = vzipq_s32(cols_0145_l.val[0], cols_2367_l.val[0]);
|
||||
int32x4x2_t rows_15 = vzipq_s32(cols_0145_h.val[0], cols_2367_h.val[0]);
|
||||
int32x4x2_t rows_26 = vzipq_s32(cols_0145_l.val[1], cols_2367_l.val[1]);
|
||||
int32x4x2_t rows_37 = vzipq_s32(cols_0145_h.val[1], cols_2367_h.val[1]);
|
||||
|
||||
int16x8_t row0 = vreinterpretq_s16_s32(rows_04.val[0]);
|
||||
int16x8_t row1 = vreinterpretq_s16_s32(rows_15.val[0]);
|
||||
int16x8_t row2 = vreinterpretq_s16_s32(rows_26.val[0]);
|
||||
int16x8_t row3 = vreinterpretq_s16_s32(rows_37.val[0]);
|
||||
int16x8_t row4 = vreinterpretq_s16_s32(rows_04.val[1]);
|
||||
int16x8_t row5 = vreinterpretq_s16_s32(rows_15.val[1]);
|
||||
int16x8_t row6 = vreinterpretq_s16_s32(rows_26.val[1]);
|
||||
int16x8_t row7 = vreinterpretq_s16_s32(rows_37.val[1]);
|
||||
|
||||
/* Pass 2: process columns. */
|
||||
|
||||
tmp0 = vaddq_s16(row0, row7);
|
||||
tmp7 = vsubq_s16(row0, row7);
|
||||
tmp1 = vaddq_s16(row1, row6);
|
||||
tmp6 = vsubq_s16(row1, row6);
|
||||
tmp2 = vaddq_s16(row2, row5);
|
||||
tmp5 = vsubq_s16(row2, row5);
|
||||
tmp3 = vaddq_s16(row3, row4);
|
||||
tmp4 = vsubq_s16(row3, row4);
|
||||
|
||||
/* Even part */
|
||||
tmp10 = vaddq_s16(tmp0, tmp3);
|
||||
tmp13 = vsubq_s16(tmp0, tmp3);
|
||||
tmp11 = vaddq_s16(tmp1, tmp2);
|
||||
tmp12 = vsubq_s16(tmp1, tmp2);
|
||||
|
||||
row0 = vrshrq_n_s16(vaddq_s16(tmp10, tmp11), PASS1_BITS);
|
||||
row4 = vrshrq_n_s16(vsubq_s16(tmp10, tmp11), PASS1_BITS);
|
||||
|
||||
tmp12_add_tmp13 = vaddq_s16(tmp12, tmp13);
|
||||
z1_l = vmull_lane_s16(vget_low_s16(tmp12_add_tmp13), consts.val[0], 2);
|
||||
z1_h = vmull_lane_s16(vget_high_s16(tmp12_add_tmp13), consts.val[0], 2);
|
||||
|
||||
int32x4_t row2_scaled_l =
|
||||
vmlal_lane_s16(z1_l, vget_low_s16(tmp13), consts.val[0], 3);
|
||||
int32x4_t row2_scaled_h =
|
||||
vmlal_lane_s16(z1_h, vget_high_s16(tmp13), consts.val[0], 3);
|
||||
row2 = vcombine_s16(vrshrn_n_s32(row2_scaled_l, DESCALE_P2),
|
||||
vrshrn_n_s32(row2_scaled_h, DESCALE_P2));
|
||||
|
||||
int32x4_t row6_scaled_l =
|
||||
vmlal_lane_s16(z1_l, vget_low_s16(tmp12), consts.val[1], 3);
|
||||
int32x4_t row6_scaled_h =
|
||||
vmlal_lane_s16(z1_h, vget_high_s16(tmp12), consts.val[1], 3);
|
||||
row6 = vcombine_s16(vrshrn_n_s32(row6_scaled_l, DESCALE_P2),
|
||||
vrshrn_n_s32(row6_scaled_h, DESCALE_P2));
|
||||
|
||||
/* Odd part */
|
||||
z1 = vaddq_s16(tmp4, tmp7);
|
||||
z2 = vaddq_s16(tmp5, tmp6);
|
||||
z3 = vaddq_s16(tmp4, tmp6);
|
||||
z4 = vaddq_s16(tmp5, tmp7);
|
||||
/* sqrt(2) * c3 */
|
||||
z5_l = vmull_lane_s16(vget_low_s16(z3), consts.val[1], 1);
|
||||
z5_h = vmull_lane_s16(vget_high_s16(z3), consts.val[1], 1);
|
||||
z5_l = vmlal_lane_s16(z5_l, vget_low_s16(z4), consts.val[1], 1);
|
||||
z5_h = vmlal_lane_s16(z5_h, vget_high_s16(z4), consts.val[1], 1);
|
||||
|
||||
/* sqrt(2) * (-c1+c3+c5-c7) */
|
||||
tmp4_l = vmull_lane_s16(vget_low_s16(tmp4), consts.val[0], 0);
|
||||
tmp4_h = vmull_lane_s16(vget_high_s16(tmp4), consts.val[0], 0);
|
||||
/* sqrt(2) * ( c1+c3-c5+c7) */
|
||||
tmp5_l = vmull_lane_s16(vget_low_s16(tmp5), consts.val[2], 1);
|
||||
tmp5_h = vmull_lane_s16(vget_high_s16(tmp5), consts.val[2], 1);
|
||||
/* sqrt(2) * ( c1+c3+c5-c7) */
|
||||
tmp6_l = vmull_lane_s16(vget_low_s16(tmp6), consts.val[2], 3);
|
||||
tmp6_h = vmull_lane_s16(vget_high_s16(tmp6), consts.val[2], 3);
|
||||
/* sqrt(2) * ( c1+c3-c5-c7) */
|
||||
tmp7_l = vmull_lane_s16(vget_low_s16(tmp7), consts.val[1], 2);
|
||||
tmp7_h = vmull_lane_s16(vget_high_s16(tmp7), consts.val[1], 2);
|
||||
|
||||
/* sqrt(2) * (c7-c3) */
|
||||
z1_l = vmull_lane_s16(vget_low_s16(z1), consts.val[1], 0);
|
||||
z1_h = vmull_lane_s16(vget_high_s16(z1), consts.val[1], 0);
|
||||
/* sqrt(2) * (-c1-c3) */
|
||||
z2_l = vmull_lane_s16(vget_low_s16(z2), consts.val[2], 2);
|
||||
z2_h = vmull_lane_s16(vget_high_s16(z2), consts.val[2], 2);
|
||||
/* sqrt(2) * (-c3-c5) */
|
||||
z3_l = vmull_lane_s16(vget_low_s16(z3), consts.val[2], 0);
|
||||
z3_h = vmull_lane_s16(vget_high_s16(z3), consts.val[2], 0);
|
||||
/* sqrt(2) * (c5-c3) */
|
||||
z4_l = vmull_lane_s16(vget_low_s16(z4), consts.val[0], 1);
|
||||
z4_h = vmull_lane_s16(vget_high_s16(z4), consts.val[0], 1);
|
||||
|
||||
z3_l = vaddq_s32(z3_l, z5_l);
|
||||
z3_h = vaddq_s32(z3_h, z5_h);
|
||||
z4_l = vaddq_s32(z4_l, z5_l);
|
||||
z4_h = vaddq_s32(z4_h, z5_h);
|
||||
|
||||
tmp4_l = vaddq_s32(tmp4_l, z1_l);
|
||||
tmp4_h = vaddq_s32(tmp4_h, z1_h);
|
||||
tmp4_l = vaddq_s32(tmp4_l, z3_l);
|
||||
tmp4_h = vaddq_s32(tmp4_h, z3_h);
|
||||
row7 = vcombine_s16(vrshrn_n_s32(tmp4_l, DESCALE_P2),
|
||||
vrshrn_n_s32(tmp4_h, DESCALE_P2));
|
||||
|
||||
tmp5_l = vaddq_s32(tmp5_l, z2_l);
|
||||
tmp5_h = vaddq_s32(tmp5_h, z2_h);
|
||||
tmp5_l = vaddq_s32(tmp5_l, z4_l);
|
||||
tmp5_h = vaddq_s32(tmp5_h, z4_h);
|
||||
row5 = vcombine_s16(vrshrn_n_s32(tmp5_l, DESCALE_P2),
|
||||
vrshrn_n_s32(tmp5_h, DESCALE_P2));
|
||||
|
||||
tmp6_l = vaddq_s32(tmp6_l, z2_l);
|
||||
tmp6_h = vaddq_s32(tmp6_h, z2_h);
|
||||
tmp6_l = vaddq_s32(tmp6_l, z3_l);
|
||||
tmp6_h = vaddq_s32(tmp6_h, z3_h);
|
||||
row3 = vcombine_s16(vrshrn_n_s32(tmp6_l, DESCALE_P2),
|
||||
vrshrn_n_s32(tmp6_h, DESCALE_P2));
|
||||
|
||||
tmp7_l = vaddq_s32(tmp7_l, z1_l);
|
||||
tmp7_h = vaddq_s32(tmp7_h, z1_h);
|
||||
tmp7_l = vaddq_s32(tmp7_l, z4_l);
|
||||
tmp7_h = vaddq_s32(tmp7_h, z4_h);
|
||||
row1 = vcombine_s16(vrshrn_n_s32(tmp7_l, DESCALE_P2),
|
||||
vrshrn_n_s32(tmp7_h, DESCALE_P2));
|
||||
|
||||
vst1q_s16(data + 0 * DCTSIZE, row0);
|
||||
vst1q_s16(data + 1 * DCTSIZE, row1);
|
||||
vst1q_s16(data + 2 * DCTSIZE, row2);
|
||||
vst1q_s16(data + 3 * DCTSIZE, row3);
|
||||
vst1q_s16(data + 4 * DCTSIZE, row4);
|
||||
vst1q_s16(data + 5 * DCTSIZE, row5);
|
||||
vst1q_s16(data + 6 * DCTSIZE, row6);
|
||||
vst1q_s16(data + 7 * DCTSIZE, row7);
|
||||
}
|
||||
+472
@@ -0,0 +1,472 @@
|
||||
/*
|
||||
* jidctfst-neon.c - fast integer IDCT (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* jsimd_idct_ifast_neon() performs dequantization and a fast, not so accurate
|
||||
* inverse DCT (Discrete Cosine Transform) on one block of coefficients. It
|
||||
* uses the same calculations and produces exactly the same output as IJG's
|
||||
* original jpeg_idct_ifast() function, which can be found in jidctfst.c.
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.082392200 = 2688 * 2^-15
|
||||
* 0.414213562 = 13568 * 2^-15
|
||||
* 0.847759065 = 27776 * 2^-15
|
||||
* 0.613125930 = 20096 * 2^-15
|
||||
*
|
||||
* See jidctfst.c for further details of the IDCT algorithm. Where possible,
|
||||
* the variable names and comments here in jsimd_idct_ifast_neon() match up
|
||||
* with those in jpeg_idct_ifast().
|
||||
*/
|
||||
|
||||
#define PASS1_BITS 2
|
||||
|
||||
#define F_0_082 2688
|
||||
#define F_0_414 13568
|
||||
#define F_0_847 27776
|
||||
#define F_0_613 20096
|
||||
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_idct_ifast_neon_consts[] = {
|
||||
F_0_082, F_0_414, F_0_847, F_0_613
|
||||
};
|
||||
|
||||
void jsimd_idct_ifast_neon(void *dct_table, JCOEFPTR coef_block,
|
||||
JSAMPARRAY output_buf, JDIMENSION output_col)
|
||||
{
|
||||
IFAST_MULT_TYPE *quantptr = dct_table;
|
||||
|
||||
/* Load DCT coefficients. */
|
||||
int16x8_t row0 = vld1q_s16(coef_block + 0 * DCTSIZE);
|
||||
int16x8_t row1 = vld1q_s16(coef_block + 1 * DCTSIZE);
|
||||
int16x8_t row2 = vld1q_s16(coef_block + 2 * DCTSIZE);
|
||||
int16x8_t row3 = vld1q_s16(coef_block + 3 * DCTSIZE);
|
||||
int16x8_t row4 = vld1q_s16(coef_block + 4 * DCTSIZE);
|
||||
int16x8_t row5 = vld1q_s16(coef_block + 5 * DCTSIZE);
|
||||
int16x8_t row6 = vld1q_s16(coef_block + 6 * DCTSIZE);
|
||||
int16x8_t row7 = vld1q_s16(coef_block + 7 * DCTSIZE);
|
||||
|
||||
/* Load quantization table values for DC coefficients. */
|
||||
int16x8_t quant_row0 = vld1q_s16(quantptr + 0 * DCTSIZE);
|
||||
/* Dequantize DC coefficients. */
|
||||
row0 = vmulq_s16(row0, quant_row0);
|
||||
|
||||
/* Construct bitmap to test if all AC coefficients are 0. */
|
||||
int16x8_t bitmap = vorrq_s16(row1, row2);
|
||||
bitmap = vorrq_s16(bitmap, row3);
|
||||
bitmap = vorrq_s16(bitmap, row4);
|
||||
bitmap = vorrq_s16(bitmap, row5);
|
||||
bitmap = vorrq_s16(bitmap, row6);
|
||||
bitmap = vorrq_s16(bitmap, row7);
|
||||
|
||||
int64_t left_ac_bitmap = vgetq_lane_s64(vreinterpretq_s64_s16(bitmap), 0);
|
||||
int64_t right_ac_bitmap = vgetq_lane_s64(vreinterpretq_s64_s16(bitmap), 1);
|
||||
|
||||
/* Load IDCT conversion constants. */
|
||||
const int16x4_t consts = vld1_s16(jsimd_idct_ifast_neon_consts);
|
||||
|
||||
if (left_ac_bitmap == 0 && right_ac_bitmap == 0) {
|
||||
/* All AC coefficients are zero.
|
||||
* Compute DC values and duplicate into vectors.
|
||||
*/
|
||||
int16x8_t dcval = row0;
|
||||
row1 = dcval;
|
||||
row2 = dcval;
|
||||
row3 = dcval;
|
||||
row4 = dcval;
|
||||
row5 = dcval;
|
||||
row6 = dcval;
|
||||
row7 = dcval;
|
||||
} else if (left_ac_bitmap == 0) {
|
||||
/* AC coefficients are zero for columns 0, 1, 2, and 3.
|
||||
* Use DC values for these columns.
|
||||
*/
|
||||
int16x4_t dcval = vget_low_s16(row0);
|
||||
|
||||
/* Commence regular fast IDCT computation for columns 4, 5, 6, and 7. */
|
||||
|
||||
/* Load quantization table. */
|
||||
int16x4_t quant_row1 = vld1_s16(quantptr + 1 * DCTSIZE + 4);
|
||||
int16x4_t quant_row2 = vld1_s16(quantptr + 2 * DCTSIZE + 4);
|
||||
int16x4_t quant_row3 = vld1_s16(quantptr + 3 * DCTSIZE + 4);
|
||||
int16x4_t quant_row4 = vld1_s16(quantptr + 4 * DCTSIZE + 4);
|
||||
int16x4_t quant_row5 = vld1_s16(quantptr + 5 * DCTSIZE + 4);
|
||||
int16x4_t quant_row6 = vld1_s16(quantptr + 6 * DCTSIZE + 4);
|
||||
int16x4_t quant_row7 = vld1_s16(quantptr + 7 * DCTSIZE + 4);
|
||||
|
||||
/* Even part: dequantize DCT coefficients. */
|
||||
int16x4_t tmp0 = vget_high_s16(row0);
|
||||
int16x4_t tmp1 = vmul_s16(vget_high_s16(row2), quant_row2);
|
||||
int16x4_t tmp2 = vmul_s16(vget_high_s16(row4), quant_row4);
|
||||
int16x4_t tmp3 = vmul_s16(vget_high_s16(row6), quant_row6);
|
||||
|
||||
int16x4_t tmp10 = vadd_s16(tmp0, tmp2); /* phase 3 */
|
||||
int16x4_t tmp11 = vsub_s16(tmp0, tmp2);
|
||||
|
||||
int16x4_t tmp13 = vadd_s16(tmp1, tmp3); /* phases 5-3 */
|
||||
int16x4_t tmp1_sub_tmp3 = vsub_s16(tmp1, tmp3);
|
||||
int16x4_t tmp12 = vqdmulh_lane_s16(tmp1_sub_tmp3, consts, 1);
|
||||
tmp12 = vadd_s16(tmp12, tmp1_sub_tmp3);
|
||||
tmp12 = vsub_s16(tmp12, tmp13);
|
||||
|
||||
tmp0 = vadd_s16(tmp10, tmp13); /* phase 2 */
|
||||
tmp3 = vsub_s16(tmp10, tmp13);
|
||||
tmp1 = vadd_s16(tmp11, tmp12);
|
||||
tmp2 = vsub_s16(tmp11, tmp12);
|
||||
|
||||
/* Odd part: dequantize DCT coefficients. */
|
||||
int16x4_t tmp4 = vmul_s16(vget_high_s16(row1), quant_row1);
|
||||
int16x4_t tmp5 = vmul_s16(vget_high_s16(row3), quant_row3);
|
||||
int16x4_t tmp6 = vmul_s16(vget_high_s16(row5), quant_row5);
|
||||
int16x4_t tmp7 = vmul_s16(vget_high_s16(row7), quant_row7);
|
||||
|
||||
int16x4_t z13 = vadd_s16(tmp6, tmp5); /* phase 6 */
|
||||
int16x4_t neg_z10 = vsub_s16(tmp5, tmp6);
|
||||
int16x4_t z11 = vadd_s16(tmp4, tmp7);
|
||||
int16x4_t z12 = vsub_s16(tmp4, tmp7);
|
||||
|
||||
tmp7 = vadd_s16(z11, z13); /* phase 5 */
|
||||
int16x4_t z11_sub_z13 = vsub_s16(z11, z13);
|
||||
tmp11 = vqdmulh_lane_s16(z11_sub_z13, consts, 1);
|
||||
tmp11 = vadd_s16(tmp11, z11_sub_z13);
|
||||
|
||||
int16x4_t z10_add_z12 = vsub_s16(z12, neg_z10);
|
||||
int16x4_t z5 = vqdmulh_lane_s16(z10_add_z12, consts, 2);
|
||||
z5 = vadd_s16(z5, z10_add_z12);
|
||||
tmp10 = vqdmulh_lane_s16(z12, consts, 0);
|
||||
tmp10 = vadd_s16(tmp10, z12);
|
||||
tmp10 = vsub_s16(tmp10, z5);
|
||||
tmp12 = vqdmulh_lane_s16(neg_z10, consts, 3);
|
||||
tmp12 = vadd_s16(tmp12, vadd_s16(neg_z10, neg_z10));
|
||||
tmp12 = vadd_s16(tmp12, z5);
|
||||
|
||||
tmp6 = vsub_s16(tmp12, tmp7); /* phase 2 */
|
||||
tmp5 = vsub_s16(tmp11, tmp6);
|
||||
tmp4 = vadd_s16(tmp10, tmp5);
|
||||
|
||||
row0 = vcombine_s16(dcval, vadd_s16(tmp0, tmp7));
|
||||
row7 = vcombine_s16(dcval, vsub_s16(tmp0, tmp7));
|
||||
row1 = vcombine_s16(dcval, vadd_s16(tmp1, tmp6));
|
||||
row6 = vcombine_s16(dcval, vsub_s16(tmp1, tmp6));
|
||||
row2 = vcombine_s16(dcval, vadd_s16(tmp2, tmp5));
|
||||
row5 = vcombine_s16(dcval, vsub_s16(tmp2, tmp5));
|
||||
row4 = vcombine_s16(dcval, vadd_s16(tmp3, tmp4));
|
||||
row3 = vcombine_s16(dcval, vsub_s16(tmp3, tmp4));
|
||||
} else if (right_ac_bitmap == 0) {
|
||||
/* AC coefficients are zero for columns 4, 5, 6, and 7.
|
||||
* Use DC values for these columns.
|
||||
*/
|
||||
int16x4_t dcval = vget_high_s16(row0);
|
||||
|
||||
/* Commence regular fast IDCT computation for columns 0, 1, 2, and 3. */
|
||||
|
||||
/* Load quantization table. */
|
||||
int16x4_t quant_row1 = vld1_s16(quantptr + 1 * DCTSIZE);
|
||||
int16x4_t quant_row2 = vld1_s16(quantptr + 2 * DCTSIZE);
|
||||
int16x4_t quant_row3 = vld1_s16(quantptr + 3 * DCTSIZE);
|
||||
int16x4_t quant_row4 = vld1_s16(quantptr + 4 * DCTSIZE);
|
||||
int16x4_t quant_row5 = vld1_s16(quantptr + 5 * DCTSIZE);
|
||||
int16x4_t quant_row6 = vld1_s16(quantptr + 6 * DCTSIZE);
|
||||
int16x4_t quant_row7 = vld1_s16(quantptr + 7 * DCTSIZE);
|
||||
|
||||
/* Even part: dequantize DCT coefficients. */
|
||||
int16x4_t tmp0 = vget_low_s16(row0);
|
||||
int16x4_t tmp1 = vmul_s16(vget_low_s16(row2), quant_row2);
|
||||
int16x4_t tmp2 = vmul_s16(vget_low_s16(row4), quant_row4);
|
||||
int16x4_t tmp3 = vmul_s16(vget_low_s16(row6), quant_row6);
|
||||
|
||||
int16x4_t tmp10 = vadd_s16(tmp0, tmp2); /* phase 3 */
|
||||
int16x4_t tmp11 = vsub_s16(tmp0, tmp2);
|
||||
|
||||
int16x4_t tmp13 = vadd_s16(tmp1, tmp3); /* phases 5-3 */
|
||||
int16x4_t tmp1_sub_tmp3 = vsub_s16(tmp1, tmp3);
|
||||
int16x4_t tmp12 = vqdmulh_lane_s16(tmp1_sub_tmp3, consts, 1);
|
||||
tmp12 = vadd_s16(tmp12, tmp1_sub_tmp3);
|
||||
tmp12 = vsub_s16(tmp12, tmp13);
|
||||
|
||||
tmp0 = vadd_s16(tmp10, tmp13); /* phase 2 */
|
||||
tmp3 = vsub_s16(tmp10, tmp13);
|
||||
tmp1 = vadd_s16(tmp11, tmp12);
|
||||
tmp2 = vsub_s16(tmp11, tmp12);
|
||||
|
||||
/* Odd part: dequantize DCT coefficients. */
|
||||
int16x4_t tmp4 = vmul_s16(vget_low_s16(row1), quant_row1);
|
||||
int16x4_t tmp5 = vmul_s16(vget_low_s16(row3), quant_row3);
|
||||
int16x4_t tmp6 = vmul_s16(vget_low_s16(row5), quant_row5);
|
||||
int16x4_t tmp7 = vmul_s16(vget_low_s16(row7), quant_row7);
|
||||
|
||||
int16x4_t z13 = vadd_s16(tmp6, tmp5); /* phase 6 */
|
||||
int16x4_t neg_z10 = vsub_s16(tmp5, tmp6);
|
||||
int16x4_t z11 = vadd_s16(tmp4, tmp7);
|
||||
int16x4_t z12 = vsub_s16(tmp4, tmp7);
|
||||
|
||||
tmp7 = vadd_s16(z11, z13); /* phase 5 */
|
||||
int16x4_t z11_sub_z13 = vsub_s16(z11, z13);
|
||||
tmp11 = vqdmulh_lane_s16(z11_sub_z13, consts, 1);
|
||||
tmp11 = vadd_s16(tmp11, z11_sub_z13);
|
||||
|
||||
int16x4_t z10_add_z12 = vsub_s16(z12, neg_z10);
|
||||
int16x4_t z5 = vqdmulh_lane_s16(z10_add_z12, consts, 2);
|
||||
z5 = vadd_s16(z5, z10_add_z12);
|
||||
tmp10 = vqdmulh_lane_s16(z12, consts, 0);
|
||||
tmp10 = vadd_s16(tmp10, z12);
|
||||
tmp10 = vsub_s16(tmp10, z5);
|
||||
tmp12 = vqdmulh_lane_s16(neg_z10, consts, 3);
|
||||
tmp12 = vadd_s16(tmp12, vadd_s16(neg_z10, neg_z10));
|
||||
tmp12 = vadd_s16(tmp12, z5);
|
||||
|
||||
tmp6 = vsub_s16(tmp12, tmp7); /* phase 2 */
|
||||
tmp5 = vsub_s16(tmp11, tmp6);
|
||||
tmp4 = vadd_s16(tmp10, tmp5);
|
||||
|
||||
row0 = vcombine_s16(vadd_s16(tmp0, tmp7), dcval);
|
||||
row7 = vcombine_s16(vsub_s16(tmp0, tmp7), dcval);
|
||||
row1 = vcombine_s16(vadd_s16(tmp1, tmp6), dcval);
|
||||
row6 = vcombine_s16(vsub_s16(tmp1, tmp6), dcval);
|
||||
row2 = vcombine_s16(vadd_s16(tmp2, tmp5), dcval);
|
||||
row5 = vcombine_s16(vsub_s16(tmp2, tmp5), dcval);
|
||||
row4 = vcombine_s16(vadd_s16(tmp3, tmp4), dcval);
|
||||
row3 = vcombine_s16(vsub_s16(tmp3, tmp4), dcval);
|
||||
} else {
|
||||
/* Some AC coefficients are non-zero; full IDCT calculation required. */
|
||||
|
||||
/* Load quantization table. */
|
||||
int16x8_t quant_row1 = vld1q_s16(quantptr + 1 * DCTSIZE);
|
||||
int16x8_t quant_row2 = vld1q_s16(quantptr + 2 * DCTSIZE);
|
||||
int16x8_t quant_row3 = vld1q_s16(quantptr + 3 * DCTSIZE);
|
||||
int16x8_t quant_row4 = vld1q_s16(quantptr + 4 * DCTSIZE);
|
||||
int16x8_t quant_row5 = vld1q_s16(quantptr + 5 * DCTSIZE);
|
||||
int16x8_t quant_row6 = vld1q_s16(quantptr + 6 * DCTSIZE);
|
||||
int16x8_t quant_row7 = vld1q_s16(quantptr + 7 * DCTSIZE);
|
||||
|
||||
/* Even part: dequantize DCT coefficients. */
|
||||
int16x8_t tmp0 = row0;
|
||||
int16x8_t tmp1 = vmulq_s16(row2, quant_row2);
|
||||
int16x8_t tmp2 = vmulq_s16(row4, quant_row4);
|
||||
int16x8_t tmp3 = vmulq_s16(row6, quant_row6);
|
||||
|
||||
int16x8_t tmp10 = vaddq_s16(tmp0, tmp2); /* phase 3 */
|
||||
int16x8_t tmp11 = vsubq_s16(tmp0, tmp2);
|
||||
|
||||
int16x8_t tmp13 = vaddq_s16(tmp1, tmp3); /* phases 5-3 */
|
||||
int16x8_t tmp1_sub_tmp3 = vsubq_s16(tmp1, tmp3);
|
||||
int16x8_t tmp12 = vqdmulhq_lane_s16(tmp1_sub_tmp3, consts, 1);
|
||||
tmp12 = vaddq_s16(tmp12, tmp1_sub_tmp3);
|
||||
tmp12 = vsubq_s16(tmp12, tmp13);
|
||||
|
||||
tmp0 = vaddq_s16(tmp10, tmp13); /* phase 2 */
|
||||
tmp3 = vsubq_s16(tmp10, tmp13);
|
||||
tmp1 = vaddq_s16(tmp11, tmp12);
|
||||
tmp2 = vsubq_s16(tmp11, tmp12);
|
||||
|
||||
/* Odd part: dequantize DCT coefficients. */
|
||||
int16x8_t tmp4 = vmulq_s16(row1, quant_row1);
|
||||
int16x8_t tmp5 = vmulq_s16(row3, quant_row3);
|
||||
int16x8_t tmp6 = vmulq_s16(row5, quant_row5);
|
||||
int16x8_t tmp7 = vmulq_s16(row7, quant_row7);
|
||||
|
||||
int16x8_t z13 = vaddq_s16(tmp6, tmp5); /* phase 6 */
|
||||
int16x8_t neg_z10 = vsubq_s16(tmp5, tmp6);
|
||||
int16x8_t z11 = vaddq_s16(tmp4, tmp7);
|
||||
int16x8_t z12 = vsubq_s16(tmp4, tmp7);
|
||||
|
||||
tmp7 = vaddq_s16(z11, z13); /* phase 5 */
|
||||
int16x8_t z11_sub_z13 = vsubq_s16(z11, z13);
|
||||
tmp11 = vqdmulhq_lane_s16(z11_sub_z13, consts, 1);
|
||||
tmp11 = vaddq_s16(tmp11, z11_sub_z13);
|
||||
|
||||
int16x8_t z10_add_z12 = vsubq_s16(z12, neg_z10);
|
||||
int16x8_t z5 = vqdmulhq_lane_s16(z10_add_z12, consts, 2);
|
||||
z5 = vaddq_s16(z5, z10_add_z12);
|
||||
tmp10 = vqdmulhq_lane_s16(z12, consts, 0);
|
||||
tmp10 = vaddq_s16(tmp10, z12);
|
||||
tmp10 = vsubq_s16(tmp10, z5);
|
||||
tmp12 = vqdmulhq_lane_s16(neg_z10, consts, 3);
|
||||
tmp12 = vaddq_s16(tmp12, vaddq_s16(neg_z10, neg_z10));
|
||||
tmp12 = vaddq_s16(tmp12, z5);
|
||||
|
||||
tmp6 = vsubq_s16(tmp12, tmp7); /* phase 2 */
|
||||
tmp5 = vsubq_s16(tmp11, tmp6);
|
||||
tmp4 = vaddq_s16(tmp10, tmp5);
|
||||
|
||||
row0 = vaddq_s16(tmp0, tmp7);
|
||||
row7 = vsubq_s16(tmp0, tmp7);
|
||||
row1 = vaddq_s16(tmp1, tmp6);
|
||||
row6 = vsubq_s16(tmp1, tmp6);
|
||||
row2 = vaddq_s16(tmp2, tmp5);
|
||||
row5 = vsubq_s16(tmp2, tmp5);
|
||||
row4 = vaddq_s16(tmp3, tmp4);
|
||||
row3 = vsubq_s16(tmp3, tmp4);
|
||||
}
|
||||
|
||||
/* Transpose rows to work on columns in pass 2. */
|
||||
int16x8x2_t rows_01 = vtrnq_s16(row0, row1);
|
||||
int16x8x2_t rows_23 = vtrnq_s16(row2, row3);
|
||||
int16x8x2_t rows_45 = vtrnq_s16(row4, row5);
|
||||
int16x8x2_t rows_67 = vtrnq_s16(row6, row7);
|
||||
|
||||
int32x4x2_t rows_0145_l = vtrnq_s32(vreinterpretq_s32_s16(rows_01.val[0]),
|
||||
vreinterpretq_s32_s16(rows_45.val[0]));
|
||||
int32x4x2_t rows_0145_h = vtrnq_s32(vreinterpretq_s32_s16(rows_01.val[1]),
|
||||
vreinterpretq_s32_s16(rows_45.val[1]));
|
||||
int32x4x2_t rows_2367_l = vtrnq_s32(vreinterpretq_s32_s16(rows_23.val[0]),
|
||||
vreinterpretq_s32_s16(rows_67.val[0]));
|
||||
int32x4x2_t rows_2367_h = vtrnq_s32(vreinterpretq_s32_s16(rows_23.val[1]),
|
||||
vreinterpretq_s32_s16(rows_67.val[1]));
|
||||
|
||||
int32x4x2_t cols_04 = vzipq_s32(rows_0145_l.val[0], rows_2367_l.val[0]);
|
||||
int32x4x2_t cols_15 = vzipq_s32(rows_0145_h.val[0], rows_2367_h.val[0]);
|
||||
int32x4x2_t cols_26 = vzipq_s32(rows_0145_l.val[1], rows_2367_l.val[1]);
|
||||
int32x4x2_t cols_37 = vzipq_s32(rows_0145_h.val[1], rows_2367_h.val[1]);
|
||||
|
||||
int16x8_t col0 = vreinterpretq_s16_s32(cols_04.val[0]);
|
||||
int16x8_t col1 = vreinterpretq_s16_s32(cols_15.val[0]);
|
||||
int16x8_t col2 = vreinterpretq_s16_s32(cols_26.val[0]);
|
||||
int16x8_t col3 = vreinterpretq_s16_s32(cols_37.val[0]);
|
||||
int16x8_t col4 = vreinterpretq_s16_s32(cols_04.val[1]);
|
||||
int16x8_t col5 = vreinterpretq_s16_s32(cols_15.val[1]);
|
||||
int16x8_t col6 = vreinterpretq_s16_s32(cols_26.val[1]);
|
||||
int16x8_t col7 = vreinterpretq_s16_s32(cols_37.val[1]);
|
||||
|
||||
/* 1-D IDCT, pass 2 */
|
||||
|
||||
/* Even part */
|
||||
int16x8_t tmp10 = vaddq_s16(col0, col4);
|
||||
int16x8_t tmp11 = vsubq_s16(col0, col4);
|
||||
|
||||
int16x8_t tmp13 = vaddq_s16(col2, col6);
|
||||
int16x8_t col2_sub_col6 = vsubq_s16(col2, col6);
|
||||
int16x8_t tmp12 = vqdmulhq_lane_s16(col2_sub_col6, consts, 1);
|
||||
tmp12 = vaddq_s16(tmp12, col2_sub_col6);
|
||||
tmp12 = vsubq_s16(tmp12, tmp13);
|
||||
|
||||
int16x8_t tmp0 = vaddq_s16(tmp10, tmp13);
|
||||
int16x8_t tmp3 = vsubq_s16(tmp10, tmp13);
|
||||
int16x8_t tmp1 = vaddq_s16(tmp11, tmp12);
|
||||
int16x8_t tmp2 = vsubq_s16(tmp11, tmp12);
|
||||
|
||||
/* Odd part */
|
||||
int16x8_t z13 = vaddq_s16(col5, col3);
|
||||
int16x8_t neg_z10 = vsubq_s16(col3, col5);
|
||||
int16x8_t z11 = vaddq_s16(col1, col7);
|
||||
int16x8_t z12 = vsubq_s16(col1, col7);
|
||||
|
||||
int16x8_t tmp7 = vaddq_s16(z11, z13); /* phase 5 */
|
||||
int16x8_t z11_sub_z13 = vsubq_s16(z11, z13);
|
||||
tmp11 = vqdmulhq_lane_s16(z11_sub_z13, consts, 1);
|
||||
tmp11 = vaddq_s16(tmp11, z11_sub_z13);
|
||||
|
||||
int16x8_t z10_add_z12 = vsubq_s16(z12, neg_z10);
|
||||
int16x8_t z5 = vqdmulhq_lane_s16(z10_add_z12, consts, 2);
|
||||
z5 = vaddq_s16(z5, z10_add_z12);
|
||||
tmp10 = vqdmulhq_lane_s16(z12, consts, 0);
|
||||
tmp10 = vaddq_s16(tmp10, z12);
|
||||
tmp10 = vsubq_s16(tmp10, z5);
|
||||
tmp12 = vqdmulhq_lane_s16(neg_z10, consts, 3);
|
||||
tmp12 = vaddq_s16(tmp12, vaddq_s16(neg_z10, neg_z10));
|
||||
tmp12 = vaddq_s16(tmp12, z5);
|
||||
|
||||
int16x8_t tmp6 = vsubq_s16(tmp12, tmp7); /* phase 2 */
|
||||
int16x8_t tmp5 = vsubq_s16(tmp11, tmp6);
|
||||
int16x8_t tmp4 = vaddq_s16(tmp10, tmp5);
|
||||
|
||||
col0 = vaddq_s16(tmp0, tmp7);
|
||||
col7 = vsubq_s16(tmp0, tmp7);
|
||||
col1 = vaddq_s16(tmp1, tmp6);
|
||||
col6 = vsubq_s16(tmp1, tmp6);
|
||||
col2 = vaddq_s16(tmp2, tmp5);
|
||||
col5 = vsubq_s16(tmp2, tmp5);
|
||||
col4 = vaddq_s16(tmp3, tmp4);
|
||||
col3 = vsubq_s16(tmp3, tmp4);
|
||||
|
||||
/* Scale down by a factor of 8, narrowing to 8-bit. */
|
||||
int8x16_t cols_01_s8 = vcombine_s8(vqshrn_n_s16(col0, PASS1_BITS + 3),
|
||||
vqshrn_n_s16(col1, PASS1_BITS + 3));
|
||||
int8x16_t cols_45_s8 = vcombine_s8(vqshrn_n_s16(col4, PASS1_BITS + 3),
|
||||
vqshrn_n_s16(col5, PASS1_BITS + 3));
|
||||
int8x16_t cols_23_s8 = vcombine_s8(vqshrn_n_s16(col2, PASS1_BITS + 3),
|
||||
vqshrn_n_s16(col3, PASS1_BITS + 3));
|
||||
int8x16_t cols_67_s8 = vcombine_s8(vqshrn_n_s16(col6, PASS1_BITS + 3),
|
||||
vqshrn_n_s16(col7, PASS1_BITS + 3));
|
||||
/* Clamp to range [0-255]. */
|
||||
uint8x16_t cols_01 =
|
||||
vreinterpretq_u8_s8
|
||||
(vaddq_s8(cols_01_s8, vreinterpretq_s8_u8(vdupq_n_u8(CENTERJSAMPLE))));
|
||||
uint8x16_t cols_45 =
|
||||
vreinterpretq_u8_s8
|
||||
(vaddq_s8(cols_45_s8, vreinterpretq_s8_u8(vdupq_n_u8(CENTERJSAMPLE))));
|
||||
uint8x16_t cols_23 =
|
||||
vreinterpretq_u8_s8
|
||||
(vaddq_s8(cols_23_s8, vreinterpretq_s8_u8(vdupq_n_u8(CENTERJSAMPLE))));
|
||||
uint8x16_t cols_67 =
|
||||
vreinterpretq_u8_s8
|
||||
(vaddq_s8(cols_67_s8, vreinterpretq_s8_u8(vdupq_n_u8(CENTERJSAMPLE))));
|
||||
|
||||
/* Transpose block to prepare for store. */
|
||||
uint32x4x2_t cols_0415 = vzipq_u32(vreinterpretq_u32_u8(cols_01),
|
||||
vreinterpretq_u32_u8(cols_45));
|
||||
uint32x4x2_t cols_2637 = vzipq_u32(vreinterpretq_u32_u8(cols_23),
|
||||
vreinterpretq_u32_u8(cols_67));
|
||||
|
||||
uint8x16x2_t cols_0145 = vtrnq_u8(vreinterpretq_u8_u32(cols_0415.val[0]),
|
||||
vreinterpretq_u8_u32(cols_0415.val[1]));
|
||||
uint8x16x2_t cols_2367 = vtrnq_u8(vreinterpretq_u8_u32(cols_2637.val[0]),
|
||||
vreinterpretq_u8_u32(cols_2637.val[1]));
|
||||
uint16x8x2_t rows_0426 = vtrnq_u16(vreinterpretq_u16_u8(cols_0145.val[0]),
|
||||
vreinterpretq_u16_u8(cols_2367.val[0]));
|
||||
uint16x8x2_t rows_1537 = vtrnq_u16(vreinterpretq_u16_u8(cols_0145.val[1]),
|
||||
vreinterpretq_u16_u8(cols_2367.val[1]));
|
||||
|
||||
uint8x16_t rows_04 = vreinterpretq_u8_u16(rows_0426.val[0]);
|
||||
uint8x16_t rows_15 = vreinterpretq_u8_u16(rows_1537.val[0]);
|
||||
uint8x16_t rows_26 = vreinterpretq_u8_u16(rows_0426.val[1]);
|
||||
uint8x16_t rows_37 = vreinterpretq_u8_u16(rows_1537.val[1]);
|
||||
|
||||
JSAMPROW outptr0 = output_buf[0] + output_col;
|
||||
JSAMPROW outptr1 = output_buf[1] + output_col;
|
||||
JSAMPROW outptr2 = output_buf[2] + output_col;
|
||||
JSAMPROW outptr3 = output_buf[3] + output_col;
|
||||
JSAMPROW outptr4 = output_buf[4] + output_col;
|
||||
JSAMPROW outptr5 = output_buf[5] + output_col;
|
||||
JSAMPROW outptr6 = output_buf[6] + output_col;
|
||||
JSAMPROW outptr7 = output_buf[7] + output_col;
|
||||
|
||||
/* Store DCT block to memory. */
|
||||
vst1q_lane_u64((uint64_t *)outptr0, vreinterpretq_u64_u8(rows_04), 0);
|
||||
vst1q_lane_u64((uint64_t *)outptr1, vreinterpretq_u64_u8(rows_15), 0);
|
||||
vst1q_lane_u64((uint64_t *)outptr2, vreinterpretq_u64_u8(rows_26), 0);
|
||||
vst1q_lane_u64((uint64_t *)outptr3, vreinterpretq_u64_u8(rows_37), 0);
|
||||
vst1q_lane_u64((uint64_t *)outptr4, vreinterpretq_u64_u8(rows_04), 1);
|
||||
vst1q_lane_u64((uint64_t *)outptr5, vreinterpretq_u64_u8(rows_15), 1);
|
||||
vst1q_lane_u64((uint64_t *)outptr6, vreinterpretq_u64_u8(rows_26), 1);
|
||||
vst1q_lane_u64((uint64_t *)outptr7, vreinterpretq_u64_u8(rows_37), 1);
|
||||
}
|
||||
+802
@@ -0,0 +1,802 @@
|
||||
/*
|
||||
* jidctint-neon.c - accurate integer IDCT (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "jconfigint.h"
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
#define CONST_BITS 13
|
||||
#define PASS1_BITS 2
|
||||
|
||||
#define DESCALE_P1 (CONST_BITS - PASS1_BITS)
|
||||
#define DESCALE_P2 (CONST_BITS + PASS1_BITS + 3)
|
||||
|
||||
/* The computation of the inverse DCT requires the use of constants known at
|
||||
* compile time. Scaled integer constants are used to avoid floating-point
|
||||
* arithmetic:
|
||||
* 0.298631336 = 2446 * 2^-13
|
||||
* 0.390180644 = 3196 * 2^-13
|
||||
* 0.541196100 = 4433 * 2^-13
|
||||
* 0.765366865 = 6270 * 2^-13
|
||||
* 0.899976223 = 7373 * 2^-13
|
||||
* 1.175875602 = 9633 * 2^-13
|
||||
* 1.501321110 = 12299 * 2^-13
|
||||
* 1.847759065 = 15137 * 2^-13
|
||||
* 1.961570560 = 16069 * 2^-13
|
||||
* 2.053119869 = 16819 * 2^-13
|
||||
* 2.562915447 = 20995 * 2^-13
|
||||
* 3.072711026 = 25172 * 2^-13
|
||||
*/
|
||||
|
||||
#define F_0_298 2446
|
||||
#define F_0_390 3196
|
||||
#define F_0_541 4433
|
||||
#define F_0_765 6270
|
||||
#define F_0_899 7373
|
||||
#define F_1_175 9633
|
||||
#define F_1_501 12299
|
||||
#define F_1_847 15137
|
||||
#define F_1_961 16069
|
||||
#define F_2_053 16819
|
||||
#define F_2_562 20995
|
||||
#define F_3_072 25172
|
||||
|
||||
#define F_1_175_MINUS_1_961 (F_1_175 - F_1_961)
|
||||
#define F_1_175_MINUS_0_390 (F_1_175 - F_0_390)
|
||||
#define F_0_541_MINUS_1_847 (F_0_541 - F_1_847)
|
||||
#define F_3_072_MINUS_2_562 (F_3_072 - F_2_562)
|
||||
#define F_0_298_MINUS_0_899 (F_0_298 - F_0_899)
|
||||
#define F_1_501_MINUS_0_899 (F_1_501 - F_0_899)
|
||||
#define F_2_053_MINUS_2_562 (F_2_053 - F_2_562)
|
||||
#define F_0_541_PLUS_0_765 (F_0_541 + F_0_765)
|
||||
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_idct_islow_neon_consts[] = {
|
||||
F_0_899, F_0_541,
|
||||
F_2_562, F_0_298_MINUS_0_899,
|
||||
F_1_501_MINUS_0_899, F_2_053_MINUS_2_562,
|
||||
F_0_541_PLUS_0_765, F_1_175,
|
||||
F_1_175_MINUS_0_390, F_0_541_MINUS_1_847,
|
||||
F_3_072_MINUS_2_562, F_1_175_MINUS_1_961,
|
||||
0, 0, 0, 0
|
||||
};
|
||||
|
||||
|
||||
/* Forward declaration of regular and sparse IDCT helper functions */
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass1_regular(int16x4_t row0,
|
||||
int16x4_t row1,
|
||||
int16x4_t row2,
|
||||
int16x4_t row3,
|
||||
int16x4_t row4,
|
||||
int16x4_t row5,
|
||||
int16x4_t row6,
|
||||
int16x4_t row7,
|
||||
int16x4_t quant_row0,
|
||||
int16x4_t quant_row1,
|
||||
int16x4_t quant_row2,
|
||||
int16x4_t quant_row3,
|
||||
int16x4_t quant_row4,
|
||||
int16x4_t quant_row5,
|
||||
int16x4_t quant_row6,
|
||||
int16x4_t quant_row7,
|
||||
int16_t *workspace_1,
|
||||
int16_t *workspace_2);
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass1_sparse(int16x4_t row0,
|
||||
int16x4_t row1,
|
||||
int16x4_t row2,
|
||||
int16x4_t row3,
|
||||
int16x4_t quant_row0,
|
||||
int16x4_t quant_row1,
|
||||
int16x4_t quant_row2,
|
||||
int16x4_t quant_row3,
|
||||
int16_t *workspace_1,
|
||||
int16_t *workspace_2);
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass2_regular(int16_t *workspace,
|
||||
JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col,
|
||||
unsigned buf_offset);
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass2_sparse(int16_t *workspace,
|
||||
JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col,
|
||||
unsigned buf_offset);
|
||||
|
||||
|
||||
/* Perform dequantization and inverse DCT on one block of coefficients. For
|
||||
* reference, the C implementation (jpeg_idct_slow()) can be found in
|
||||
* jidctint.c.
|
||||
*
|
||||
* Optimization techniques used for fast data access:
|
||||
*
|
||||
* In each pass, the inverse DCT is computed for the left and right 4x8 halves
|
||||
* of the DCT block. This avoids spilling due to register pressure, and the
|
||||
* increased granularity allows for an optimized calculation depending on the
|
||||
* values of the DCT coefficients. Between passes, intermediate data is stored
|
||||
* in 4x8 workspace buffers.
|
||||
*
|
||||
* Transposing the 8x8 DCT block after each pass can be achieved by transposing
|
||||
* each of the four 4x4 quadrants and swapping quadrants 1 and 2 (refer to the
|
||||
* diagram below.) Swapping quadrants is cheap, since the second pass can just
|
||||
* swap the workspace buffer pointers.
|
||||
*
|
||||
* +-------+-------+ +-------+-------+
|
||||
* | | | | | |
|
||||
* | 0 | 1 | | 0 | 2 |
|
||||
* | | | transpose | | |
|
||||
* +-------+-------+ ------> +-------+-------+
|
||||
* | | | | | |
|
||||
* | 2 | 3 | | 1 | 3 |
|
||||
* | | | | | |
|
||||
* +-------+-------+ +-------+-------+
|
||||
*
|
||||
* Optimization techniques used to accelerate the inverse DCT calculation:
|
||||
*
|
||||
* In a DCT coefficient block, the coefficients are increasingly likely to be 0
|
||||
* as you move diagonally from top left to bottom right. If whole rows of
|
||||
* coefficients are 0, then the inverse DCT calculation can be simplified. On
|
||||
* the first pass of the inverse DCT, we test for three special cases before
|
||||
* defaulting to a full "regular" inverse DCT:
|
||||
*
|
||||
* 1) Coefficients in rows 4-7 are all zero. In this case, we perform a
|
||||
* "sparse" simplified inverse DCT on rows 0-3.
|
||||
* 2) AC coefficients (rows 1-7) are all zero. In this case, the inverse DCT
|
||||
* result is equal to the dequantized DC coefficients.
|
||||
* 3) AC and DC coefficients are all zero. In this case, the inverse DCT
|
||||
* result is all zero. For the left 4x8 half, this is handled identically
|
||||
* to Case 2 above. For the right 4x8 half, we do no work and signal that
|
||||
* the "sparse" algorithm is required for the second pass.
|
||||
*
|
||||
* In the second pass, only a single special case is tested: whether the AC and
|
||||
* DC coefficients were all zero in the right 4x8 block during the first pass
|
||||
* (refer to Case 3 above.) If this is the case, then a "sparse" variant of
|
||||
* the second pass is performed for both the left and right halves of the DCT
|
||||
* block. (The transposition after the first pass means that the right 4x8
|
||||
* block during the first pass becomes rows 4-7 during the second pass.)
|
||||
*/
|
||||
|
||||
void jsimd_idct_islow_neon(void *dct_table, JCOEFPTR coef_block,
|
||||
JSAMPARRAY output_buf, JDIMENSION output_col)
|
||||
{
|
||||
ISLOW_MULT_TYPE *quantptr = dct_table;
|
||||
|
||||
int16_t workspace_l[8 * DCTSIZE / 2];
|
||||
int16_t workspace_r[8 * DCTSIZE / 2];
|
||||
|
||||
/* Compute IDCT first pass on left 4x8 coefficient block. */
|
||||
|
||||
/* Load DCT coefficients in left 4x8 block. */
|
||||
int16x4_t row0 = vld1_s16(coef_block + 0 * DCTSIZE);
|
||||
int16x4_t row1 = vld1_s16(coef_block + 1 * DCTSIZE);
|
||||
int16x4_t row2 = vld1_s16(coef_block + 2 * DCTSIZE);
|
||||
int16x4_t row3 = vld1_s16(coef_block + 3 * DCTSIZE);
|
||||
int16x4_t row4 = vld1_s16(coef_block + 4 * DCTSIZE);
|
||||
int16x4_t row5 = vld1_s16(coef_block + 5 * DCTSIZE);
|
||||
int16x4_t row6 = vld1_s16(coef_block + 6 * DCTSIZE);
|
||||
int16x4_t row7 = vld1_s16(coef_block + 7 * DCTSIZE);
|
||||
|
||||
/* Load quantization table for left 4x8 block. */
|
||||
int16x4_t quant_row0 = vld1_s16(quantptr + 0 * DCTSIZE);
|
||||
int16x4_t quant_row1 = vld1_s16(quantptr + 1 * DCTSIZE);
|
||||
int16x4_t quant_row2 = vld1_s16(quantptr + 2 * DCTSIZE);
|
||||
int16x4_t quant_row3 = vld1_s16(quantptr + 3 * DCTSIZE);
|
||||
int16x4_t quant_row4 = vld1_s16(quantptr + 4 * DCTSIZE);
|
||||
int16x4_t quant_row5 = vld1_s16(quantptr + 5 * DCTSIZE);
|
||||
int16x4_t quant_row6 = vld1_s16(quantptr + 6 * DCTSIZE);
|
||||
int16x4_t quant_row7 = vld1_s16(quantptr + 7 * DCTSIZE);
|
||||
|
||||
/* Construct bitmap to test if DCT coefficients in left 4x8 block are 0. */
|
||||
int16x4_t bitmap = vorr_s16(row7, row6);
|
||||
bitmap = vorr_s16(bitmap, row5);
|
||||
bitmap = vorr_s16(bitmap, row4);
|
||||
int64_t bitmap_rows_4567 = vget_lane_s64(vreinterpret_s64_s16(bitmap), 0);
|
||||
|
||||
if (bitmap_rows_4567 == 0) {
|
||||
bitmap = vorr_s16(bitmap, row3);
|
||||
bitmap = vorr_s16(bitmap, row2);
|
||||
bitmap = vorr_s16(bitmap, row1);
|
||||
int64_t left_ac_bitmap = vget_lane_s64(vreinterpret_s64_s16(bitmap), 0);
|
||||
|
||||
if (left_ac_bitmap == 0) {
|
||||
int16x4_t dcval = vshl_n_s16(vmul_s16(row0, quant_row0), PASS1_BITS);
|
||||
int16x4x4_t quadrant = { { dcval, dcval, dcval, dcval } };
|
||||
/* Store 4x4 blocks to workspace, transposing in the process. */
|
||||
vst4_s16(workspace_l, quadrant);
|
||||
vst4_s16(workspace_r, quadrant);
|
||||
} else {
|
||||
jsimd_idct_islow_pass1_sparse(row0, row1, row2, row3, quant_row0,
|
||||
quant_row1, quant_row2, quant_row3,
|
||||
workspace_l, workspace_r);
|
||||
}
|
||||
} else {
|
||||
jsimd_idct_islow_pass1_regular(row0, row1, row2, row3, row4, row5,
|
||||
row6, row7, quant_row0, quant_row1,
|
||||
quant_row2, quant_row3, quant_row4,
|
||||
quant_row5, quant_row6, quant_row7,
|
||||
workspace_l, workspace_r);
|
||||
}
|
||||
|
||||
/* Compute IDCT first pass on right 4x8 coefficient block. */
|
||||
|
||||
/* Load DCT coefficients in right 4x8 block. */
|
||||
row0 = vld1_s16(coef_block + 0 * DCTSIZE + 4);
|
||||
row1 = vld1_s16(coef_block + 1 * DCTSIZE + 4);
|
||||
row2 = vld1_s16(coef_block + 2 * DCTSIZE + 4);
|
||||
row3 = vld1_s16(coef_block + 3 * DCTSIZE + 4);
|
||||
row4 = vld1_s16(coef_block + 4 * DCTSIZE + 4);
|
||||
row5 = vld1_s16(coef_block + 5 * DCTSIZE + 4);
|
||||
row6 = vld1_s16(coef_block + 6 * DCTSIZE + 4);
|
||||
row7 = vld1_s16(coef_block + 7 * DCTSIZE + 4);
|
||||
|
||||
/* Load quantization table for right 4x8 block. */
|
||||
quant_row0 = vld1_s16(quantptr + 0 * DCTSIZE + 4);
|
||||
quant_row1 = vld1_s16(quantptr + 1 * DCTSIZE + 4);
|
||||
quant_row2 = vld1_s16(quantptr + 2 * DCTSIZE + 4);
|
||||
quant_row3 = vld1_s16(quantptr + 3 * DCTSIZE + 4);
|
||||
quant_row4 = vld1_s16(quantptr + 4 * DCTSIZE + 4);
|
||||
quant_row5 = vld1_s16(quantptr + 5 * DCTSIZE + 4);
|
||||
quant_row6 = vld1_s16(quantptr + 6 * DCTSIZE + 4);
|
||||
quant_row7 = vld1_s16(quantptr + 7 * DCTSIZE + 4);
|
||||
|
||||
/* Construct bitmap to test if DCT coefficients in right 4x8 block are 0. */
|
||||
bitmap = vorr_s16(row7, row6);
|
||||
bitmap = vorr_s16(bitmap, row5);
|
||||
bitmap = vorr_s16(bitmap, row4);
|
||||
bitmap_rows_4567 = vget_lane_s64(vreinterpret_s64_s16(bitmap), 0);
|
||||
bitmap = vorr_s16(bitmap, row3);
|
||||
bitmap = vorr_s16(bitmap, row2);
|
||||
bitmap = vorr_s16(bitmap, row1);
|
||||
int64_t right_ac_bitmap = vget_lane_s64(vreinterpret_s64_s16(bitmap), 0);
|
||||
|
||||
/* If this remains non-zero, a "regular" second pass will be performed. */
|
||||
int64_t right_ac_dc_bitmap = 1;
|
||||
|
||||
if (right_ac_bitmap == 0) {
|
||||
bitmap = vorr_s16(bitmap, row0);
|
||||
right_ac_dc_bitmap = vget_lane_s64(vreinterpret_s64_s16(bitmap), 0);
|
||||
|
||||
if (right_ac_dc_bitmap != 0) {
|
||||
int16x4_t dcval = vshl_n_s16(vmul_s16(row0, quant_row0), PASS1_BITS);
|
||||
int16x4x4_t quadrant = { { dcval, dcval, dcval, dcval } };
|
||||
/* Store 4x4 blocks to workspace, transposing in the process. */
|
||||
vst4_s16(workspace_l + 4 * DCTSIZE / 2, quadrant);
|
||||
vst4_s16(workspace_r + 4 * DCTSIZE / 2, quadrant);
|
||||
}
|
||||
} else {
|
||||
if (bitmap_rows_4567 == 0) {
|
||||
jsimd_idct_islow_pass1_sparse(row0, row1, row2, row3, quant_row0,
|
||||
quant_row1, quant_row2, quant_row3,
|
||||
workspace_l + 4 * DCTSIZE / 2,
|
||||
workspace_r + 4 * DCTSIZE / 2);
|
||||
} else {
|
||||
jsimd_idct_islow_pass1_regular(row0, row1, row2, row3, row4, row5,
|
||||
row6, row7, quant_row0, quant_row1,
|
||||
quant_row2, quant_row3, quant_row4,
|
||||
quant_row5, quant_row6, quant_row7,
|
||||
workspace_l + 4 * DCTSIZE / 2,
|
||||
workspace_r + 4 * DCTSIZE / 2);
|
||||
}
|
||||
}
|
||||
|
||||
/* Second pass: compute IDCT on rows in workspace. */
|
||||
|
||||
/* If all coefficients in right 4x8 block are 0, use "sparse" second pass. */
|
||||
if (right_ac_dc_bitmap == 0) {
|
||||
jsimd_idct_islow_pass2_sparse(workspace_l, output_buf, output_col, 0);
|
||||
jsimd_idct_islow_pass2_sparse(workspace_r, output_buf, output_col, 4);
|
||||
} else {
|
||||
jsimd_idct_islow_pass2_regular(workspace_l, output_buf, output_col, 0);
|
||||
jsimd_idct_islow_pass2_regular(workspace_r, output_buf, output_col, 4);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/* Perform dequantization and the first pass of the accurate inverse DCT on a
|
||||
* 4x8 block of coefficients. (To process the full 8x8 DCT block, this
|
||||
* function-- or some other optimized variant-- needs to be called for both the
|
||||
* left and right 4x8 blocks.)
|
||||
*
|
||||
* This "regular" version assumes that no optimization can be made to the IDCT
|
||||
* calculation, since no useful set of AC coefficients is all 0.
|
||||
*
|
||||
* The original C implementation of the accurate IDCT (jpeg_idct_slow()) can be
|
||||
* found in jidctint.c. Algorithmic changes made here are documented inline.
|
||||
*/
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass1_regular(int16x4_t row0,
|
||||
int16x4_t row1,
|
||||
int16x4_t row2,
|
||||
int16x4_t row3,
|
||||
int16x4_t row4,
|
||||
int16x4_t row5,
|
||||
int16x4_t row6,
|
||||
int16x4_t row7,
|
||||
int16x4_t quant_row0,
|
||||
int16x4_t quant_row1,
|
||||
int16x4_t quant_row2,
|
||||
int16x4_t quant_row3,
|
||||
int16x4_t quant_row4,
|
||||
int16x4_t quant_row5,
|
||||
int16x4_t quant_row6,
|
||||
int16x4_t quant_row7,
|
||||
int16_t *workspace_1,
|
||||
int16_t *workspace_2)
|
||||
{
|
||||
/* Load constants for IDCT computation. */
|
||||
#ifdef HAVE_VLD1_S16_X3
|
||||
const int16x4x3_t consts = vld1_s16_x3(jsimd_idct_islow_neon_consts);
|
||||
#else
|
||||
const int16x4_t consts1 = vld1_s16(jsimd_idct_islow_neon_consts);
|
||||
const int16x4_t consts2 = vld1_s16(jsimd_idct_islow_neon_consts + 4);
|
||||
const int16x4_t consts3 = vld1_s16(jsimd_idct_islow_neon_consts + 8);
|
||||
const int16x4x3_t consts = { { consts1, consts2, consts3 } };
|
||||
#endif
|
||||
|
||||
/* Even part */
|
||||
int16x4_t z2_s16 = vmul_s16(row2, quant_row2);
|
||||
int16x4_t z3_s16 = vmul_s16(row6, quant_row6);
|
||||
|
||||
int32x4_t tmp2 = vmull_lane_s16(z2_s16, consts.val[0], 1);
|
||||
int32x4_t tmp3 = vmull_lane_s16(z2_s16, consts.val[1], 2);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z3_s16, consts.val[2], 1);
|
||||
tmp3 = vmlal_lane_s16(tmp3, z3_s16, consts.val[0], 1);
|
||||
|
||||
z2_s16 = vmul_s16(row0, quant_row0);
|
||||
z3_s16 = vmul_s16(row4, quant_row4);
|
||||
|
||||
int32x4_t tmp0 = vshll_n_s16(vadd_s16(z2_s16, z3_s16), CONST_BITS);
|
||||
int32x4_t tmp1 = vshll_n_s16(vsub_s16(z2_s16, z3_s16), CONST_BITS);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp13 = vsubq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp11 = vaddq_s32(tmp1, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp1, tmp2);
|
||||
|
||||
/* Odd part */
|
||||
int16x4_t tmp0_s16 = vmul_s16(row7, quant_row7);
|
||||
int16x4_t tmp1_s16 = vmul_s16(row5, quant_row5);
|
||||
int16x4_t tmp2_s16 = vmul_s16(row3, quant_row3);
|
||||
int16x4_t tmp3_s16 = vmul_s16(row1, quant_row1);
|
||||
|
||||
z3_s16 = vadd_s16(tmp0_s16, tmp2_s16);
|
||||
int16x4_t z4_s16 = vadd_s16(tmp1_s16, tmp3_s16);
|
||||
|
||||
/* Implementation as per jpeg_idct_islow() in jidctint.c:
|
||||
* z5 = (z3 + z4) * 1.175875602;
|
||||
* z3 = z3 * -1.961570560; z4 = z4 * -0.390180644;
|
||||
* z3 += z5; z4 += z5;
|
||||
*
|
||||
* This implementation:
|
||||
* z3 = z3 * (1.175875602 - 1.961570560) + z4 * 1.175875602;
|
||||
* z4 = z3 * 1.175875602 + z4 * (1.175875602 - 0.390180644);
|
||||
*/
|
||||
|
||||
int32x4_t z3 = vmull_lane_s16(z3_s16, consts.val[2], 3);
|
||||
int32x4_t z4 = vmull_lane_s16(z3_s16, consts.val[1], 3);
|
||||
z3 = vmlal_lane_s16(z3, z4_s16, consts.val[1], 3);
|
||||
z4 = vmlal_lane_s16(z4, z4_s16, consts.val[2], 0);
|
||||
|
||||
/* Implementation as per jpeg_idct_islow() in jidctint.c:
|
||||
* z1 = tmp0 + tmp3; z2 = tmp1 + tmp2;
|
||||
* tmp0 = tmp0 * 0.298631336; tmp1 = tmp1 * 2.053119869;
|
||||
* tmp2 = tmp2 * 3.072711026; tmp3 = tmp3 * 1.501321110;
|
||||
* z1 = z1 * -0.899976223; z2 = z2 * -2.562915447;
|
||||
* tmp0 += z1 + z3; tmp1 += z2 + z4;
|
||||
* tmp2 += z2 + z3; tmp3 += z1 + z4;
|
||||
*
|
||||
* This implementation:
|
||||
* tmp0 = tmp0 * (0.298631336 - 0.899976223) + tmp3 * -0.899976223;
|
||||
* tmp1 = tmp1 * (2.053119869 - 2.562915447) + tmp2 * -2.562915447;
|
||||
* tmp2 = tmp1 * -2.562915447 + tmp2 * (3.072711026 - 2.562915447);
|
||||
* tmp3 = tmp0 * -0.899976223 + tmp3 * (1.501321110 - 0.899976223);
|
||||
* tmp0 += z3; tmp1 += z4;
|
||||
* tmp2 += z3; tmp3 += z4;
|
||||
*/
|
||||
|
||||
tmp0 = vmull_lane_s16(tmp0_s16, consts.val[0], 3);
|
||||
tmp1 = vmull_lane_s16(tmp1_s16, consts.val[1], 1);
|
||||
tmp2 = vmull_lane_s16(tmp2_s16, consts.val[2], 2);
|
||||
tmp3 = vmull_lane_s16(tmp3_s16, consts.val[1], 0);
|
||||
|
||||
tmp0 = vmlsl_lane_s16(tmp0, tmp3_s16, consts.val[0], 0);
|
||||
tmp1 = vmlsl_lane_s16(tmp1, tmp2_s16, consts.val[0], 2);
|
||||
tmp2 = vmlsl_lane_s16(tmp2, tmp1_s16, consts.val[0], 2);
|
||||
tmp3 = vmlsl_lane_s16(tmp3, tmp0_s16, consts.val[0], 0);
|
||||
|
||||
tmp0 = vaddq_s32(tmp0, z3);
|
||||
tmp1 = vaddq_s32(tmp1, z4);
|
||||
tmp2 = vaddq_s32(tmp2, z3);
|
||||
tmp3 = vaddq_s32(tmp3, z4);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
int16x4x4_t rows_0123 = { {
|
||||
vrshrn_n_s32(vaddq_s32(tmp10, tmp3), DESCALE_P1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp11, tmp2), DESCALE_P1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp12, tmp1), DESCALE_P1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp13, tmp0), DESCALE_P1)
|
||||
} };
|
||||
int16x4x4_t rows_4567 = { {
|
||||
vrshrn_n_s32(vsubq_s32(tmp13, tmp0), DESCALE_P1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp12, tmp1), DESCALE_P1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp11, tmp2), DESCALE_P1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp10, tmp3), DESCALE_P1)
|
||||
} };
|
||||
|
||||
/* Store 4x4 blocks to the intermediate workspace, ready for the second pass.
|
||||
* (VST4 transposes the blocks. We need to operate on rows in the next
|
||||
* pass.)
|
||||
*/
|
||||
vst4_s16(workspace_1, rows_0123);
|
||||
vst4_s16(workspace_2, rows_4567);
|
||||
}
|
||||
|
||||
|
||||
/* Perform dequantization and the first pass of the accurate inverse DCT on a
|
||||
* 4x8 block of coefficients.
|
||||
*
|
||||
* This "sparse" version assumes that the AC coefficients in rows 4-7 are all
|
||||
* 0. This simplifies the IDCT calculation, accelerating overall performance.
|
||||
*/
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass1_sparse(int16x4_t row0,
|
||||
int16x4_t row1,
|
||||
int16x4_t row2,
|
||||
int16x4_t row3,
|
||||
int16x4_t quant_row0,
|
||||
int16x4_t quant_row1,
|
||||
int16x4_t quant_row2,
|
||||
int16x4_t quant_row3,
|
||||
int16_t *workspace_1,
|
||||
int16_t *workspace_2)
|
||||
{
|
||||
/* Load constants for IDCT computation. */
|
||||
#ifdef HAVE_VLD1_S16_X3
|
||||
const int16x4x3_t consts = vld1_s16_x3(jsimd_idct_islow_neon_consts);
|
||||
#else
|
||||
const int16x4_t consts1 = vld1_s16(jsimd_idct_islow_neon_consts);
|
||||
const int16x4_t consts2 = vld1_s16(jsimd_idct_islow_neon_consts + 4);
|
||||
const int16x4_t consts3 = vld1_s16(jsimd_idct_islow_neon_consts + 8);
|
||||
const int16x4x3_t consts = { { consts1, consts2, consts3 } };
|
||||
#endif
|
||||
|
||||
/* Even part (z3 is all 0) */
|
||||
int16x4_t z2_s16 = vmul_s16(row2, quant_row2);
|
||||
|
||||
int32x4_t tmp2 = vmull_lane_s16(z2_s16, consts.val[0], 1);
|
||||
int32x4_t tmp3 = vmull_lane_s16(z2_s16, consts.val[1], 2);
|
||||
|
||||
z2_s16 = vmul_s16(row0, quant_row0);
|
||||
int32x4_t tmp0 = vshll_n_s16(z2_s16, CONST_BITS);
|
||||
int32x4_t tmp1 = vshll_n_s16(z2_s16, CONST_BITS);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp13 = vsubq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp11 = vaddq_s32(tmp1, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp1, tmp2);
|
||||
|
||||
/* Odd part (tmp0 and tmp1 are both all 0) */
|
||||
int16x4_t tmp2_s16 = vmul_s16(row3, quant_row3);
|
||||
int16x4_t tmp3_s16 = vmul_s16(row1, quant_row1);
|
||||
|
||||
int16x4_t z3_s16 = tmp2_s16;
|
||||
int16x4_t z4_s16 = tmp3_s16;
|
||||
|
||||
int32x4_t z3 = vmull_lane_s16(z3_s16, consts.val[2], 3);
|
||||
int32x4_t z4 = vmull_lane_s16(z3_s16, consts.val[1], 3);
|
||||
z3 = vmlal_lane_s16(z3, z4_s16, consts.val[1], 3);
|
||||
z4 = vmlal_lane_s16(z4, z4_s16, consts.val[2], 0);
|
||||
|
||||
tmp0 = vmlsl_lane_s16(z3, tmp3_s16, consts.val[0], 0);
|
||||
tmp1 = vmlsl_lane_s16(z4, tmp2_s16, consts.val[0], 2);
|
||||
tmp2 = vmlal_lane_s16(z3, tmp2_s16, consts.val[2], 2);
|
||||
tmp3 = vmlal_lane_s16(z4, tmp3_s16, consts.val[1], 0);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
int16x4x4_t rows_0123 = { {
|
||||
vrshrn_n_s32(vaddq_s32(tmp10, tmp3), DESCALE_P1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp11, tmp2), DESCALE_P1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp12, tmp1), DESCALE_P1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp13, tmp0), DESCALE_P1)
|
||||
} };
|
||||
int16x4x4_t rows_4567 = { {
|
||||
vrshrn_n_s32(vsubq_s32(tmp13, tmp0), DESCALE_P1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp12, tmp1), DESCALE_P1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp11, tmp2), DESCALE_P1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp10, tmp3), DESCALE_P1)
|
||||
} };
|
||||
|
||||
/* Store 4x4 blocks to the intermediate workspace, ready for the second pass.
|
||||
* (VST4 transposes the blocks. We need to operate on rows in the next
|
||||
* pass.)
|
||||
*/
|
||||
vst4_s16(workspace_1, rows_0123);
|
||||
vst4_s16(workspace_2, rows_4567);
|
||||
}
|
||||
|
||||
|
||||
/* Perform the second pass of the accurate inverse DCT on a 4x8 block of
|
||||
* coefficients. (To process the full 8x8 DCT block, this function-- or some
|
||||
* other optimized variant-- needs to be called for both the right and left 4x8
|
||||
* blocks.)
|
||||
*
|
||||
* This "regular" version assumes that no optimization can be made to the IDCT
|
||||
* calculation, since no useful set of coefficient values are all 0 after the
|
||||
* first pass.
|
||||
*
|
||||
* Again, the original C implementation of the accurate IDCT (jpeg_idct_slow())
|
||||
* can be found in jidctint.c. Algorithmic changes made here are documented
|
||||
* inline.
|
||||
*/
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass2_regular(int16_t *workspace,
|
||||
JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col,
|
||||
unsigned buf_offset)
|
||||
{
|
||||
/* Load constants for IDCT computation. */
|
||||
#ifdef HAVE_VLD1_S16_X3
|
||||
const int16x4x3_t consts = vld1_s16_x3(jsimd_idct_islow_neon_consts);
|
||||
#else
|
||||
const int16x4_t consts1 = vld1_s16(jsimd_idct_islow_neon_consts);
|
||||
const int16x4_t consts2 = vld1_s16(jsimd_idct_islow_neon_consts + 4);
|
||||
const int16x4_t consts3 = vld1_s16(jsimd_idct_islow_neon_consts + 8);
|
||||
const int16x4x3_t consts = { { consts1, consts2, consts3 } };
|
||||
#endif
|
||||
|
||||
/* Even part */
|
||||
int16x4_t z2_s16 = vld1_s16(workspace + 2 * DCTSIZE / 2);
|
||||
int16x4_t z3_s16 = vld1_s16(workspace + 6 * DCTSIZE / 2);
|
||||
|
||||
int32x4_t tmp2 = vmull_lane_s16(z2_s16, consts.val[0], 1);
|
||||
int32x4_t tmp3 = vmull_lane_s16(z2_s16, consts.val[1], 2);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z3_s16, consts.val[2], 1);
|
||||
tmp3 = vmlal_lane_s16(tmp3, z3_s16, consts.val[0], 1);
|
||||
|
||||
z2_s16 = vld1_s16(workspace + 0 * DCTSIZE / 2);
|
||||
z3_s16 = vld1_s16(workspace + 4 * DCTSIZE / 2);
|
||||
|
||||
int32x4_t tmp0 = vshll_n_s16(vadd_s16(z2_s16, z3_s16), CONST_BITS);
|
||||
int32x4_t tmp1 = vshll_n_s16(vsub_s16(z2_s16, z3_s16), CONST_BITS);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp13 = vsubq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp11 = vaddq_s32(tmp1, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp1, tmp2);
|
||||
|
||||
/* Odd part */
|
||||
int16x4_t tmp0_s16 = vld1_s16(workspace + 7 * DCTSIZE / 2);
|
||||
int16x4_t tmp1_s16 = vld1_s16(workspace + 5 * DCTSIZE / 2);
|
||||
int16x4_t tmp2_s16 = vld1_s16(workspace + 3 * DCTSIZE / 2);
|
||||
int16x4_t tmp3_s16 = vld1_s16(workspace + 1 * DCTSIZE / 2);
|
||||
|
||||
z3_s16 = vadd_s16(tmp0_s16, tmp2_s16);
|
||||
int16x4_t z4_s16 = vadd_s16(tmp1_s16, tmp3_s16);
|
||||
|
||||
/* Implementation as per jpeg_idct_islow() in jidctint.c:
|
||||
* z5 = (z3 + z4) * 1.175875602;
|
||||
* z3 = z3 * -1.961570560; z4 = z4 * -0.390180644;
|
||||
* z3 += z5; z4 += z5;
|
||||
*
|
||||
* This implementation:
|
||||
* z3 = z3 * (1.175875602 - 1.961570560) + z4 * 1.175875602;
|
||||
* z4 = z3 * 1.175875602 + z4 * (1.175875602 - 0.390180644);
|
||||
*/
|
||||
|
||||
int32x4_t z3 = vmull_lane_s16(z3_s16, consts.val[2], 3);
|
||||
int32x4_t z4 = vmull_lane_s16(z3_s16, consts.val[1], 3);
|
||||
z3 = vmlal_lane_s16(z3, z4_s16, consts.val[1], 3);
|
||||
z4 = vmlal_lane_s16(z4, z4_s16, consts.val[2], 0);
|
||||
|
||||
/* Implementation as per jpeg_idct_islow() in jidctint.c:
|
||||
* z1 = tmp0 + tmp3; z2 = tmp1 + tmp2;
|
||||
* tmp0 = tmp0 * 0.298631336; tmp1 = tmp1 * 2.053119869;
|
||||
* tmp2 = tmp2 * 3.072711026; tmp3 = tmp3 * 1.501321110;
|
||||
* z1 = z1 * -0.899976223; z2 = z2 * -2.562915447;
|
||||
* tmp0 += z1 + z3; tmp1 += z2 + z4;
|
||||
* tmp2 += z2 + z3; tmp3 += z1 + z4;
|
||||
*
|
||||
* This implementation:
|
||||
* tmp0 = tmp0 * (0.298631336 - 0.899976223) + tmp3 * -0.899976223;
|
||||
* tmp1 = tmp1 * (2.053119869 - 2.562915447) + tmp2 * -2.562915447;
|
||||
* tmp2 = tmp1 * -2.562915447 + tmp2 * (3.072711026 - 2.562915447);
|
||||
* tmp3 = tmp0 * -0.899976223 + tmp3 * (1.501321110 - 0.899976223);
|
||||
* tmp0 += z3; tmp1 += z4;
|
||||
* tmp2 += z3; tmp3 += z4;
|
||||
*/
|
||||
|
||||
tmp0 = vmull_lane_s16(tmp0_s16, consts.val[0], 3);
|
||||
tmp1 = vmull_lane_s16(tmp1_s16, consts.val[1], 1);
|
||||
tmp2 = vmull_lane_s16(tmp2_s16, consts.val[2], 2);
|
||||
tmp3 = vmull_lane_s16(tmp3_s16, consts.val[1], 0);
|
||||
|
||||
tmp0 = vmlsl_lane_s16(tmp0, tmp3_s16, consts.val[0], 0);
|
||||
tmp1 = vmlsl_lane_s16(tmp1, tmp2_s16, consts.val[0], 2);
|
||||
tmp2 = vmlsl_lane_s16(tmp2, tmp1_s16, consts.val[0], 2);
|
||||
tmp3 = vmlsl_lane_s16(tmp3, tmp0_s16, consts.val[0], 0);
|
||||
|
||||
tmp0 = vaddq_s32(tmp0, z3);
|
||||
tmp1 = vaddq_s32(tmp1, z4);
|
||||
tmp2 = vaddq_s32(tmp2, z3);
|
||||
tmp3 = vaddq_s32(tmp3, z4);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
int16x8_t cols_02_s16 = vcombine_s16(vaddhn_s32(tmp10, tmp3),
|
||||
vaddhn_s32(tmp12, tmp1));
|
||||
int16x8_t cols_13_s16 = vcombine_s16(vaddhn_s32(tmp11, tmp2),
|
||||
vaddhn_s32(tmp13, tmp0));
|
||||
int16x8_t cols_46_s16 = vcombine_s16(vsubhn_s32(tmp13, tmp0),
|
||||
vsubhn_s32(tmp11, tmp2));
|
||||
int16x8_t cols_57_s16 = vcombine_s16(vsubhn_s32(tmp12, tmp1),
|
||||
vsubhn_s32(tmp10, tmp3));
|
||||
/* Descale and narrow to 8-bit. */
|
||||
int8x8_t cols_02_s8 = vqrshrn_n_s16(cols_02_s16, DESCALE_P2 - 16);
|
||||
int8x8_t cols_13_s8 = vqrshrn_n_s16(cols_13_s16, DESCALE_P2 - 16);
|
||||
int8x8_t cols_46_s8 = vqrshrn_n_s16(cols_46_s16, DESCALE_P2 - 16);
|
||||
int8x8_t cols_57_s8 = vqrshrn_n_s16(cols_57_s16, DESCALE_P2 - 16);
|
||||
/* Clamp to range [0-255]. */
|
||||
uint8x8_t cols_02_u8 = vadd_u8(vreinterpret_u8_s8(cols_02_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
uint8x8_t cols_13_u8 = vadd_u8(vreinterpret_u8_s8(cols_13_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
uint8x8_t cols_46_u8 = vadd_u8(vreinterpret_u8_s8(cols_46_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
uint8x8_t cols_57_u8 = vadd_u8(vreinterpret_u8_s8(cols_57_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
|
||||
/* Transpose 4x8 block and store to memory. (Zipping adjacent columns
|
||||
* together allows us to store 16-bit elements.)
|
||||
*/
|
||||
uint8x8x2_t cols_01_23 = vzip_u8(cols_02_u8, cols_13_u8);
|
||||
uint8x8x2_t cols_45_67 = vzip_u8(cols_46_u8, cols_57_u8);
|
||||
uint16x4x4_t cols_01_23_45_67 = { {
|
||||
vreinterpret_u16_u8(cols_01_23.val[0]),
|
||||
vreinterpret_u16_u8(cols_01_23.val[1]),
|
||||
vreinterpret_u16_u8(cols_45_67.val[0]),
|
||||
vreinterpret_u16_u8(cols_45_67.val[1])
|
||||
} };
|
||||
|
||||
JSAMPROW outptr0 = output_buf[buf_offset + 0] + output_col;
|
||||
JSAMPROW outptr1 = output_buf[buf_offset + 1] + output_col;
|
||||
JSAMPROW outptr2 = output_buf[buf_offset + 2] + output_col;
|
||||
JSAMPROW outptr3 = output_buf[buf_offset + 3] + output_col;
|
||||
/* VST4 of 16-bit elements completes the transpose. */
|
||||
vst4_lane_u16((uint16_t *)outptr0, cols_01_23_45_67, 0);
|
||||
vst4_lane_u16((uint16_t *)outptr1, cols_01_23_45_67, 1);
|
||||
vst4_lane_u16((uint16_t *)outptr2, cols_01_23_45_67, 2);
|
||||
vst4_lane_u16((uint16_t *)outptr3, cols_01_23_45_67, 3);
|
||||
}
|
||||
|
||||
|
||||
/* Performs the second pass of the accurate inverse DCT on a 4x8 block
|
||||
* of coefficients.
|
||||
*
|
||||
* This "sparse" version assumes that the coefficient values (after the first
|
||||
* pass) in rows 4-7 are all 0. This simplifies the IDCT calculation,
|
||||
* accelerating overall performance.
|
||||
*/
|
||||
|
||||
static INLINE void jsimd_idct_islow_pass2_sparse(int16_t *workspace,
|
||||
JSAMPARRAY output_buf,
|
||||
JDIMENSION output_col,
|
||||
unsigned buf_offset)
|
||||
{
|
||||
/* Load constants for IDCT computation. */
|
||||
#ifdef HAVE_VLD1_S16_X3
|
||||
const int16x4x3_t consts = vld1_s16_x3(jsimd_idct_islow_neon_consts);
|
||||
#else
|
||||
const int16x4_t consts1 = vld1_s16(jsimd_idct_islow_neon_consts);
|
||||
const int16x4_t consts2 = vld1_s16(jsimd_idct_islow_neon_consts + 4);
|
||||
const int16x4_t consts3 = vld1_s16(jsimd_idct_islow_neon_consts + 8);
|
||||
const int16x4x3_t consts = { { consts1, consts2, consts3 } };
|
||||
#endif
|
||||
|
||||
/* Even part (z3 is all 0) */
|
||||
int16x4_t z2_s16 = vld1_s16(workspace + 2 * DCTSIZE / 2);
|
||||
|
||||
int32x4_t tmp2 = vmull_lane_s16(z2_s16, consts.val[0], 1);
|
||||
int32x4_t tmp3 = vmull_lane_s16(z2_s16, consts.val[1], 2);
|
||||
|
||||
z2_s16 = vld1_s16(workspace + 0 * DCTSIZE / 2);
|
||||
int32x4_t tmp0 = vshll_n_s16(z2_s16, CONST_BITS);
|
||||
int32x4_t tmp1 = vshll_n_s16(z2_s16, CONST_BITS);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp13 = vsubq_s32(tmp0, tmp3);
|
||||
int32x4_t tmp11 = vaddq_s32(tmp1, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp1, tmp2);
|
||||
|
||||
/* Odd part (tmp0 and tmp1 are both all 0) */
|
||||
int16x4_t tmp2_s16 = vld1_s16(workspace + 3 * DCTSIZE / 2);
|
||||
int16x4_t tmp3_s16 = vld1_s16(workspace + 1 * DCTSIZE / 2);
|
||||
|
||||
int16x4_t z3_s16 = tmp2_s16;
|
||||
int16x4_t z4_s16 = tmp3_s16;
|
||||
|
||||
int32x4_t z3 = vmull_lane_s16(z3_s16, consts.val[2], 3);
|
||||
z3 = vmlal_lane_s16(z3, z4_s16, consts.val[1], 3);
|
||||
int32x4_t z4 = vmull_lane_s16(z3_s16, consts.val[1], 3);
|
||||
z4 = vmlal_lane_s16(z4, z4_s16, consts.val[2], 0);
|
||||
|
||||
tmp0 = vmlsl_lane_s16(z3, tmp3_s16, consts.val[0], 0);
|
||||
tmp1 = vmlsl_lane_s16(z4, tmp2_s16, consts.val[0], 2);
|
||||
tmp2 = vmlal_lane_s16(z3, tmp2_s16, consts.val[2], 2);
|
||||
tmp3 = vmlal_lane_s16(z4, tmp3_s16, consts.val[1], 0);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
int16x8_t cols_02_s16 = vcombine_s16(vaddhn_s32(tmp10, tmp3),
|
||||
vaddhn_s32(tmp12, tmp1));
|
||||
int16x8_t cols_13_s16 = vcombine_s16(vaddhn_s32(tmp11, tmp2),
|
||||
vaddhn_s32(tmp13, tmp0));
|
||||
int16x8_t cols_46_s16 = vcombine_s16(vsubhn_s32(tmp13, tmp0),
|
||||
vsubhn_s32(tmp11, tmp2));
|
||||
int16x8_t cols_57_s16 = vcombine_s16(vsubhn_s32(tmp12, tmp1),
|
||||
vsubhn_s32(tmp10, tmp3));
|
||||
/* Descale and narrow to 8-bit. */
|
||||
int8x8_t cols_02_s8 = vqrshrn_n_s16(cols_02_s16, DESCALE_P2 - 16);
|
||||
int8x8_t cols_13_s8 = vqrshrn_n_s16(cols_13_s16, DESCALE_P2 - 16);
|
||||
int8x8_t cols_46_s8 = vqrshrn_n_s16(cols_46_s16, DESCALE_P2 - 16);
|
||||
int8x8_t cols_57_s8 = vqrshrn_n_s16(cols_57_s16, DESCALE_P2 - 16);
|
||||
/* Clamp to range [0-255]. */
|
||||
uint8x8_t cols_02_u8 = vadd_u8(vreinterpret_u8_s8(cols_02_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
uint8x8_t cols_13_u8 = vadd_u8(vreinterpret_u8_s8(cols_13_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
uint8x8_t cols_46_u8 = vadd_u8(vreinterpret_u8_s8(cols_46_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
uint8x8_t cols_57_u8 = vadd_u8(vreinterpret_u8_s8(cols_57_s8),
|
||||
vdup_n_u8(CENTERJSAMPLE));
|
||||
|
||||
/* Transpose 4x8 block and store to memory. (Zipping adjacent columns
|
||||
* together allows us to store 16-bit elements.)
|
||||
*/
|
||||
uint8x8x2_t cols_01_23 = vzip_u8(cols_02_u8, cols_13_u8);
|
||||
uint8x8x2_t cols_45_67 = vzip_u8(cols_46_u8, cols_57_u8);
|
||||
uint16x4x4_t cols_01_23_45_67 = { {
|
||||
vreinterpret_u16_u8(cols_01_23.val[0]),
|
||||
vreinterpret_u16_u8(cols_01_23.val[1]),
|
||||
vreinterpret_u16_u8(cols_45_67.val[0]),
|
||||
vreinterpret_u16_u8(cols_45_67.val[1])
|
||||
} };
|
||||
|
||||
JSAMPROW outptr0 = output_buf[buf_offset + 0] + output_col;
|
||||
JSAMPROW outptr1 = output_buf[buf_offset + 1] + output_col;
|
||||
JSAMPROW outptr2 = output_buf[buf_offset + 2] + output_col;
|
||||
JSAMPROW outptr3 = output_buf[buf_offset + 3] + output_col;
|
||||
/* VST4 of 16-bit elements completes the transpose. */
|
||||
vst4_lane_u16((uint16_t *)outptr0, cols_01_23_45_67, 0);
|
||||
vst4_lane_u16((uint16_t *)outptr1, cols_01_23_45_67, 1);
|
||||
vst4_lane_u16((uint16_t *)outptr2, cols_01_23_45_67, 2);
|
||||
vst4_lane_u16((uint16_t *)outptr3, cols_01_23_45_67, 3);
|
||||
}
|
||||
+486
@@ -0,0 +1,486 @@
|
||||
/*
|
||||
* jidctred-neon.c - reduced-size IDCT (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020, Arm Limited. All Rights Reserved.
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
#include "align.h"
|
||||
#include "neon-compat.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
#define CONST_BITS 13
|
||||
#define PASS1_BITS 2
|
||||
|
||||
#define F_0_211 1730
|
||||
#define F_0_509 4176
|
||||
#define F_0_601 4926
|
||||
#define F_0_720 5906
|
||||
#define F_0_765 6270
|
||||
#define F_0_850 6967
|
||||
#define F_0_899 7373
|
||||
#define F_1_061 8697
|
||||
#define F_1_272 10426
|
||||
#define F_1_451 11893
|
||||
#define F_1_847 15137
|
||||
#define F_2_172 17799
|
||||
#define F_2_562 20995
|
||||
#define F_3_624 29692
|
||||
|
||||
|
||||
/* jsimd_idct_2x2_neon() is an inverse DCT function that produces reduced-size
|
||||
* 2x2 output from an 8x8 DCT block. It uses the same calculations and
|
||||
* produces exactly the same output as IJG's original jpeg_idct_2x2() function
|
||||
* from jpeg-6b, which can be found in jidctred.c.
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.720959822 = 5906 * 2^-13
|
||||
* 0.850430095 = 6967 * 2^-13
|
||||
* 1.272758580 = 10426 * 2^-13
|
||||
* 3.624509785 = 29692 * 2^-13
|
||||
*
|
||||
* See jidctred.c for further details of the 2x2 IDCT algorithm. Where
|
||||
* possible, the variable names and comments here in jsimd_idct_2x2_neon()
|
||||
* match up with those in jpeg_idct_2x2().
|
||||
*/
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_idct_2x2_neon_consts[] = {
|
||||
-F_0_720, F_0_850, -F_1_272, F_3_624
|
||||
};
|
||||
|
||||
void jsimd_idct_2x2_neon(void *dct_table, JCOEFPTR coef_block,
|
||||
JSAMPARRAY output_buf, JDIMENSION output_col)
|
||||
{
|
||||
ISLOW_MULT_TYPE *quantptr = dct_table;
|
||||
|
||||
/* Load DCT coefficients. */
|
||||
int16x8_t row0 = vld1q_s16(coef_block + 0 * DCTSIZE);
|
||||
int16x8_t row1 = vld1q_s16(coef_block + 1 * DCTSIZE);
|
||||
int16x8_t row3 = vld1q_s16(coef_block + 3 * DCTSIZE);
|
||||
int16x8_t row5 = vld1q_s16(coef_block + 5 * DCTSIZE);
|
||||
int16x8_t row7 = vld1q_s16(coef_block + 7 * DCTSIZE);
|
||||
|
||||
/* Load quantization table values. */
|
||||
int16x8_t quant_row0 = vld1q_s16(quantptr + 0 * DCTSIZE);
|
||||
int16x8_t quant_row1 = vld1q_s16(quantptr + 1 * DCTSIZE);
|
||||
int16x8_t quant_row3 = vld1q_s16(quantptr + 3 * DCTSIZE);
|
||||
int16x8_t quant_row5 = vld1q_s16(quantptr + 5 * DCTSIZE);
|
||||
int16x8_t quant_row7 = vld1q_s16(quantptr + 7 * DCTSIZE);
|
||||
|
||||
/* Dequantize DCT coefficients. */
|
||||
row0 = vmulq_s16(row0, quant_row0);
|
||||
row1 = vmulq_s16(row1, quant_row1);
|
||||
row3 = vmulq_s16(row3, quant_row3);
|
||||
row5 = vmulq_s16(row5, quant_row5);
|
||||
row7 = vmulq_s16(row7, quant_row7);
|
||||
|
||||
/* Load IDCT conversion constants. */
|
||||
const int16x4_t consts = vld1_s16(jsimd_idct_2x2_neon_consts);
|
||||
|
||||
/* Pass 1: process columns from input, put results in vectors row0 and
|
||||
* row1.
|
||||
*/
|
||||
|
||||
/* Even part */
|
||||
int32x4_t tmp10_l = vshll_n_s16(vget_low_s16(row0), CONST_BITS + 2);
|
||||
int32x4_t tmp10_h = vshll_n_s16(vget_high_s16(row0), CONST_BITS + 2);
|
||||
|
||||
/* Odd part */
|
||||
int32x4_t tmp0_l = vmull_lane_s16(vget_low_s16(row1), consts, 3);
|
||||
tmp0_l = vmlal_lane_s16(tmp0_l, vget_low_s16(row3), consts, 2);
|
||||
tmp0_l = vmlal_lane_s16(tmp0_l, vget_low_s16(row5), consts, 1);
|
||||
tmp0_l = vmlal_lane_s16(tmp0_l, vget_low_s16(row7), consts, 0);
|
||||
int32x4_t tmp0_h = vmull_lane_s16(vget_high_s16(row1), consts, 3);
|
||||
tmp0_h = vmlal_lane_s16(tmp0_h, vget_high_s16(row3), consts, 2);
|
||||
tmp0_h = vmlal_lane_s16(tmp0_h, vget_high_s16(row5), consts, 1);
|
||||
tmp0_h = vmlal_lane_s16(tmp0_h, vget_high_s16(row7), consts, 0);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
row0 = vcombine_s16(vrshrn_n_s32(vaddq_s32(tmp10_l, tmp0_l), CONST_BITS),
|
||||
vrshrn_n_s32(vaddq_s32(tmp10_h, tmp0_h), CONST_BITS));
|
||||
row1 = vcombine_s16(vrshrn_n_s32(vsubq_s32(tmp10_l, tmp0_l), CONST_BITS),
|
||||
vrshrn_n_s32(vsubq_s32(tmp10_h, tmp0_h), CONST_BITS));
|
||||
|
||||
/* Transpose two rows, ready for second pass. */
|
||||
int16x8x2_t cols_0246_1357 = vtrnq_s16(row0, row1);
|
||||
int16x8_t cols_0246 = cols_0246_1357.val[0];
|
||||
int16x8_t cols_1357 = cols_0246_1357.val[1];
|
||||
/* Duplicate columns such that each is accessible in its own vector. */
|
||||
int32x4x2_t cols_1155_3377 = vtrnq_s32(vreinterpretq_s32_s16(cols_1357),
|
||||
vreinterpretq_s32_s16(cols_1357));
|
||||
int16x8_t cols_1155 = vreinterpretq_s16_s32(cols_1155_3377.val[0]);
|
||||
int16x8_t cols_3377 = vreinterpretq_s16_s32(cols_1155_3377.val[1]);
|
||||
|
||||
/* Pass 2: process two rows, store to output array. */
|
||||
|
||||
/* Even part: we're only interested in col0; the top half of tmp10 is "don't
|
||||
* care."
|
||||
*/
|
||||
int32x4_t tmp10 = vshll_n_s16(vget_low_s16(cols_0246), CONST_BITS + 2);
|
||||
|
||||
/* Odd part: we're only interested in the bottom half of tmp0. */
|
||||
int32x4_t tmp0 = vmull_lane_s16(vget_low_s16(cols_1155), consts, 3);
|
||||
tmp0 = vmlal_lane_s16(tmp0, vget_low_s16(cols_3377), consts, 2);
|
||||
tmp0 = vmlal_lane_s16(tmp0, vget_high_s16(cols_1155), consts, 1);
|
||||
tmp0 = vmlal_lane_s16(tmp0, vget_high_s16(cols_3377), consts, 0);
|
||||
|
||||
/* Final output stage: descale and clamp to range [0-255]. */
|
||||
int16x8_t output_s16 = vcombine_s16(vaddhn_s32(tmp10, tmp0),
|
||||
vsubhn_s32(tmp10, tmp0));
|
||||
output_s16 = vrsraq_n_s16(vdupq_n_s16(CENTERJSAMPLE), output_s16,
|
||||
CONST_BITS + PASS1_BITS + 3 + 2 - 16);
|
||||
/* Narrow to 8-bit and convert to unsigned. */
|
||||
uint8x8_t output_u8 = vqmovun_s16(output_s16);
|
||||
|
||||
/* Store 2x2 block to memory. */
|
||||
vst1_lane_u8(output_buf[0] + output_col, output_u8, 0);
|
||||
vst1_lane_u8(output_buf[1] + output_col, output_u8, 1);
|
||||
vst1_lane_u8(output_buf[0] + output_col + 1, output_u8, 4);
|
||||
vst1_lane_u8(output_buf[1] + output_col + 1, output_u8, 5);
|
||||
}
|
||||
|
||||
|
||||
/* jsimd_idct_4x4_neon() is an inverse DCT function that produces reduced-size
|
||||
* 4x4 output from an 8x8 DCT block. It uses the same calculations and
|
||||
* produces exactly the same output as IJG's original jpeg_idct_4x4() function
|
||||
* from jpeg-6b, which can be found in jidctred.c.
|
||||
*
|
||||
* Scaled integer constants are used to avoid floating-point arithmetic:
|
||||
* 0.211164243 = 1730 * 2^-13
|
||||
* 0.509795579 = 4176 * 2^-13
|
||||
* 0.601344887 = 4926 * 2^-13
|
||||
* 0.765366865 = 6270 * 2^-13
|
||||
* 0.899976223 = 7373 * 2^-13
|
||||
* 1.061594337 = 8697 * 2^-13
|
||||
* 1.451774981 = 11893 * 2^-13
|
||||
* 1.847759065 = 15137 * 2^-13
|
||||
* 2.172734803 = 17799 * 2^-13
|
||||
* 2.562915447 = 20995 * 2^-13
|
||||
*
|
||||
* See jidctred.c for further details of the 4x4 IDCT algorithm. Where
|
||||
* possible, the variable names and comments here in jsimd_idct_4x4_neon()
|
||||
* match up with those in jpeg_idct_4x4().
|
||||
*/
|
||||
|
||||
ALIGN(16) static const int16_t jsimd_idct_4x4_neon_consts[] = {
|
||||
F_1_847, -F_0_765, -F_0_211, F_1_451,
|
||||
-F_2_172, F_1_061, -F_0_509, -F_0_601,
|
||||
F_0_899, F_2_562, 0, 0
|
||||
};
|
||||
|
||||
void jsimd_idct_4x4_neon(void *dct_table, JCOEFPTR coef_block,
|
||||
JSAMPARRAY output_buf, JDIMENSION output_col)
|
||||
{
|
||||
ISLOW_MULT_TYPE *quantptr = dct_table;
|
||||
|
||||
/* Load DCT coefficients. */
|
||||
int16x8_t row0 = vld1q_s16(coef_block + 0 * DCTSIZE);
|
||||
int16x8_t row1 = vld1q_s16(coef_block + 1 * DCTSIZE);
|
||||
int16x8_t row2 = vld1q_s16(coef_block + 2 * DCTSIZE);
|
||||
int16x8_t row3 = vld1q_s16(coef_block + 3 * DCTSIZE);
|
||||
int16x8_t row5 = vld1q_s16(coef_block + 5 * DCTSIZE);
|
||||
int16x8_t row6 = vld1q_s16(coef_block + 6 * DCTSIZE);
|
||||
int16x8_t row7 = vld1q_s16(coef_block + 7 * DCTSIZE);
|
||||
|
||||
/* Load quantization table values for DC coefficients. */
|
||||
int16x8_t quant_row0 = vld1q_s16(quantptr + 0 * DCTSIZE);
|
||||
/* Dequantize DC coefficients. */
|
||||
row0 = vmulq_s16(row0, quant_row0);
|
||||
|
||||
/* Construct bitmap to test if all AC coefficients are 0. */
|
||||
int16x8_t bitmap = vorrq_s16(row1, row2);
|
||||
bitmap = vorrq_s16(bitmap, row3);
|
||||
bitmap = vorrq_s16(bitmap, row5);
|
||||
bitmap = vorrq_s16(bitmap, row6);
|
||||
bitmap = vorrq_s16(bitmap, row7);
|
||||
|
||||
int64_t left_ac_bitmap = vgetq_lane_s64(vreinterpretq_s64_s16(bitmap), 0);
|
||||
int64_t right_ac_bitmap = vgetq_lane_s64(vreinterpretq_s64_s16(bitmap), 1);
|
||||
|
||||
/* Load constants for IDCT computation. */
|
||||
#ifdef HAVE_VLD1_S16_X3
|
||||
const int16x4x3_t consts = vld1_s16_x3(jsimd_idct_4x4_neon_consts);
|
||||
#else
|
||||
/* GCC does not currently support the intrinsic vld1_<type>_x3(). */
|
||||
const int16x4_t consts1 = vld1_s16(jsimd_idct_4x4_neon_consts);
|
||||
const int16x4_t consts2 = vld1_s16(jsimd_idct_4x4_neon_consts + 4);
|
||||
const int16x4_t consts3 = vld1_s16(jsimd_idct_4x4_neon_consts + 8);
|
||||
const int16x4x3_t consts = { { consts1, consts2, consts3 } };
|
||||
#endif
|
||||
|
||||
if (left_ac_bitmap == 0 && right_ac_bitmap == 0) {
|
||||
/* All AC coefficients are zero.
|
||||
* Compute DC values and duplicate into row vectors 0, 1, 2, and 3.
|
||||
*/
|
||||
int16x8_t dcval = vshlq_n_s16(row0, PASS1_BITS);
|
||||
row0 = dcval;
|
||||
row1 = dcval;
|
||||
row2 = dcval;
|
||||
row3 = dcval;
|
||||
} else if (left_ac_bitmap == 0) {
|
||||
/* AC coefficients are zero for columns 0, 1, 2, and 3.
|
||||
* Compute DC values for these columns.
|
||||
*/
|
||||
int16x4_t dcval = vshl_n_s16(vget_low_s16(row0), PASS1_BITS);
|
||||
|
||||
/* Commence regular IDCT computation for columns 4, 5, 6, and 7. */
|
||||
|
||||
/* Load quantization table. */
|
||||
int16x4_t quant_row1 = vld1_s16(quantptr + 1 * DCTSIZE + 4);
|
||||
int16x4_t quant_row2 = vld1_s16(quantptr + 2 * DCTSIZE + 4);
|
||||
int16x4_t quant_row3 = vld1_s16(quantptr + 3 * DCTSIZE + 4);
|
||||
int16x4_t quant_row5 = vld1_s16(quantptr + 5 * DCTSIZE + 4);
|
||||
int16x4_t quant_row6 = vld1_s16(quantptr + 6 * DCTSIZE + 4);
|
||||
int16x4_t quant_row7 = vld1_s16(quantptr + 7 * DCTSIZE + 4);
|
||||
|
||||
/* Even part */
|
||||
int32x4_t tmp0 = vshll_n_s16(vget_high_s16(row0), CONST_BITS + 1);
|
||||
|
||||
int16x4_t z2 = vmul_s16(vget_high_s16(row2), quant_row2);
|
||||
int16x4_t z3 = vmul_s16(vget_high_s16(row6), quant_row6);
|
||||
|
||||
int32x4_t tmp2 = vmull_lane_s16(z2, consts.val[0], 0);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z3, consts.val[0], 1);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp0, tmp2);
|
||||
|
||||
/* Odd part */
|
||||
int16x4_t z1 = vmul_s16(vget_high_s16(row7), quant_row7);
|
||||
z2 = vmul_s16(vget_high_s16(row5), quant_row5);
|
||||
z3 = vmul_s16(vget_high_s16(row3), quant_row3);
|
||||
int16x4_t z4 = vmul_s16(vget_high_s16(row1), quant_row1);
|
||||
|
||||
tmp0 = vmull_lane_s16(z1, consts.val[0], 2);
|
||||
tmp0 = vmlal_lane_s16(tmp0, z2, consts.val[0], 3);
|
||||
tmp0 = vmlal_lane_s16(tmp0, z3, consts.val[1], 0);
|
||||
tmp0 = vmlal_lane_s16(tmp0, z4, consts.val[1], 1);
|
||||
|
||||
tmp2 = vmull_lane_s16(z1, consts.val[1], 2);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z2, consts.val[1], 3);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z3, consts.val[2], 0);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z4, consts.val[2], 1);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
row0 = vcombine_s16(dcval, vrshrn_n_s32(vaddq_s32(tmp10, tmp2),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
row3 = vcombine_s16(dcval, vrshrn_n_s32(vsubq_s32(tmp10, tmp2),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
row1 = vcombine_s16(dcval, vrshrn_n_s32(vaddq_s32(tmp12, tmp0),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
row2 = vcombine_s16(dcval, vrshrn_n_s32(vsubq_s32(tmp12, tmp0),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
} else if (right_ac_bitmap == 0) {
|
||||
/* AC coefficients are zero for columns 4, 5, 6, and 7.
|
||||
* Compute DC values for these columns.
|
||||
*/
|
||||
int16x4_t dcval = vshl_n_s16(vget_high_s16(row0), PASS1_BITS);
|
||||
|
||||
/* Commence regular IDCT computation for columns 0, 1, 2, and 3. */
|
||||
|
||||
/* Load quantization table. */
|
||||
int16x4_t quant_row1 = vld1_s16(quantptr + 1 * DCTSIZE);
|
||||
int16x4_t quant_row2 = vld1_s16(quantptr + 2 * DCTSIZE);
|
||||
int16x4_t quant_row3 = vld1_s16(quantptr + 3 * DCTSIZE);
|
||||
int16x4_t quant_row5 = vld1_s16(quantptr + 5 * DCTSIZE);
|
||||
int16x4_t quant_row6 = vld1_s16(quantptr + 6 * DCTSIZE);
|
||||
int16x4_t quant_row7 = vld1_s16(quantptr + 7 * DCTSIZE);
|
||||
|
||||
/* Even part */
|
||||
int32x4_t tmp0 = vshll_n_s16(vget_low_s16(row0), CONST_BITS + 1);
|
||||
|
||||
int16x4_t z2 = vmul_s16(vget_low_s16(row2), quant_row2);
|
||||
int16x4_t z3 = vmul_s16(vget_low_s16(row6), quant_row6);
|
||||
|
||||
int32x4_t tmp2 = vmull_lane_s16(z2, consts.val[0], 0);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z3, consts.val[0], 1);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp0, tmp2);
|
||||
|
||||
/* Odd part */
|
||||
int16x4_t z1 = vmul_s16(vget_low_s16(row7), quant_row7);
|
||||
z2 = vmul_s16(vget_low_s16(row5), quant_row5);
|
||||
z3 = vmul_s16(vget_low_s16(row3), quant_row3);
|
||||
int16x4_t z4 = vmul_s16(vget_low_s16(row1), quant_row1);
|
||||
|
||||
tmp0 = vmull_lane_s16(z1, consts.val[0], 2);
|
||||
tmp0 = vmlal_lane_s16(tmp0, z2, consts.val[0], 3);
|
||||
tmp0 = vmlal_lane_s16(tmp0, z3, consts.val[1], 0);
|
||||
tmp0 = vmlal_lane_s16(tmp0, z4, consts.val[1], 1);
|
||||
|
||||
tmp2 = vmull_lane_s16(z1, consts.val[1], 2);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z2, consts.val[1], 3);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z3, consts.val[2], 0);
|
||||
tmp2 = vmlal_lane_s16(tmp2, z4, consts.val[2], 1);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
row0 = vcombine_s16(vrshrn_n_s32(vaddq_s32(tmp10, tmp2),
|
||||
CONST_BITS - PASS1_BITS + 1), dcval);
|
||||
row3 = vcombine_s16(vrshrn_n_s32(vsubq_s32(tmp10, tmp2),
|
||||
CONST_BITS - PASS1_BITS + 1), dcval);
|
||||
row1 = vcombine_s16(vrshrn_n_s32(vaddq_s32(tmp12, tmp0),
|
||||
CONST_BITS - PASS1_BITS + 1), dcval);
|
||||
row2 = vcombine_s16(vrshrn_n_s32(vsubq_s32(tmp12, tmp0),
|
||||
CONST_BITS - PASS1_BITS + 1), dcval);
|
||||
} else {
|
||||
/* All AC coefficients are non-zero; full IDCT calculation required. */
|
||||
int16x8_t quant_row1 = vld1q_s16(quantptr + 1 * DCTSIZE);
|
||||
int16x8_t quant_row2 = vld1q_s16(quantptr + 2 * DCTSIZE);
|
||||
int16x8_t quant_row3 = vld1q_s16(quantptr + 3 * DCTSIZE);
|
||||
int16x8_t quant_row5 = vld1q_s16(quantptr + 5 * DCTSIZE);
|
||||
int16x8_t quant_row6 = vld1q_s16(quantptr + 6 * DCTSIZE);
|
||||
int16x8_t quant_row7 = vld1q_s16(quantptr + 7 * DCTSIZE);
|
||||
|
||||
/* Even part */
|
||||
int32x4_t tmp0_l = vshll_n_s16(vget_low_s16(row0), CONST_BITS + 1);
|
||||
int32x4_t tmp0_h = vshll_n_s16(vget_high_s16(row0), CONST_BITS + 1);
|
||||
|
||||
int16x8_t z2 = vmulq_s16(row2, quant_row2);
|
||||
int16x8_t z3 = vmulq_s16(row6, quant_row6);
|
||||
|
||||
int32x4_t tmp2_l = vmull_lane_s16(vget_low_s16(z2), consts.val[0], 0);
|
||||
int32x4_t tmp2_h = vmull_lane_s16(vget_high_s16(z2), consts.val[0], 0);
|
||||
tmp2_l = vmlal_lane_s16(tmp2_l, vget_low_s16(z3), consts.val[0], 1);
|
||||
tmp2_h = vmlal_lane_s16(tmp2_h, vget_high_s16(z3), consts.val[0], 1);
|
||||
|
||||
int32x4_t tmp10_l = vaddq_s32(tmp0_l, tmp2_l);
|
||||
int32x4_t tmp10_h = vaddq_s32(tmp0_h, tmp2_h);
|
||||
int32x4_t tmp12_l = vsubq_s32(tmp0_l, tmp2_l);
|
||||
int32x4_t tmp12_h = vsubq_s32(tmp0_h, tmp2_h);
|
||||
|
||||
/* Odd part */
|
||||
int16x8_t z1 = vmulq_s16(row7, quant_row7);
|
||||
z2 = vmulq_s16(row5, quant_row5);
|
||||
z3 = vmulq_s16(row3, quant_row3);
|
||||
int16x8_t z4 = vmulq_s16(row1, quant_row1);
|
||||
|
||||
tmp0_l = vmull_lane_s16(vget_low_s16(z1), consts.val[0], 2);
|
||||
tmp0_l = vmlal_lane_s16(tmp0_l, vget_low_s16(z2), consts.val[0], 3);
|
||||
tmp0_l = vmlal_lane_s16(tmp0_l, vget_low_s16(z3), consts.val[1], 0);
|
||||
tmp0_l = vmlal_lane_s16(tmp0_l, vget_low_s16(z4), consts.val[1], 1);
|
||||
tmp0_h = vmull_lane_s16(vget_high_s16(z1), consts.val[0], 2);
|
||||
tmp0_h = vmlal_lane_s16(tmp0_h, vget_high_s16(z2), consts.val[0], 3);
|
||||
tmp0_h = vmlal_lane_s16(tmp0_h, vget_high_s16(z3), consts.val[1], 0);
|
||||
tmp0_h = vmlal_lane_s16(tmp0_h, vget_high_s16(z4), consts.val[1], 1);
|
||||
|
||||
tmp2_l = vmull_lane_s16(vget_low_s16(z1), consts.val[1], 2);
|
||||
tmp2_l = vmlal_lane_s16(tmp2_l, vget_low_s16(z2), consts.val[1], 3);
|
||||
tmp2_l = vmlal_lane_s16(tmp2_l, vget_low_s16(z3), consts.val[2], 0);
|
||||
tmp2_l = vmlal_lane_s16(tmp2_l, vget_low_s16(z4), consts.val[2], 1);
|
||||
tmp2_h = vmull_lane_s16(vget_high_s16(z1), consts.val[1], 2);
|
||||
tmp2_h = vmlal_lane_s16(tmp2_h, vget_high_s16(z2), consts.val[1], 3);
|
||||
tmp2_h = vmlal_lane_s16(tmp2_h, vget_high_s16(z3), consts.val[2], 0);
|
||||
tmp2_h = vmlal_lane_s16(tmp2_h, vget_high_s16(z4), consts.val[2], 1);
|
||||
|
||||
/* Final output stage: descale and narrow to 16-bit. */
|
||||
row0 = vcombine_s16(vrshrn_n_s32(vaddq_s32(tmp10_l, tmp2_l),
|
||||
CONST_BITS - PASS1_BITS + 1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp10_h, tmp2_h),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
row3 = vcombine_s16(vrshrn_n_s32(vsubq_s32(tmp10_l, tmp2_l),
|
||||
CONST_BITS - PASS1_BITS + 1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp10_h, tmp2_h),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
row1 = vcombine_s16(vrshrn_n_s32(vaddq_s32(tmp12_l, tmp0_l),
|
||||
CONST_BITS - PASS1_BITS + 1),
|
||||
vrshrn_n_s32(vaddq_s32(tmp12_h, tmp0_h),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
row2 = vcombine_s16(vrshrn_n_s32(vsubq_s32(tmp12_l, tmp0_l),
|
||||
CONST_BITS - PASS1_BITS + 1),
|
||||
vrshrn_n_s32(vsubq_s32(tmp12_h, tmp0_h),
|
||||
CONST_BITS - PASS1_BITS + 1));
|
||||
}
|
||||
|
||||
/* Transpose 8x4 block to perform IDCT on rows in second pass. */
|
||||
int16x8x2_t row_01 = vtrnq_s16(row0, row1);
|
||||
int16x8x2_t row_23 = vtrnq_s16(row2, row3);
|
||||
|
||||
int32x4x2_t cols_0426 = vtrnq_s32(vreinterpretq_s32_s16(row_01.val[0]),
|
||||
vreinterpretq_s32_s16(row_23.val[0]));
|
||||
int32x4x2_t cols_1537 = vtrnq_s32(vreinterpretq_s32_s16(row_01.val[1]),
|
||||
vreinterpretq_s32_s16(row_23.val[1]));
|
||||
|
||||
int16x4_t col0 = vreinterpret_s16_s32(vget_low_s32(cols_0426.val[0]));
|
||||
int16x4_t col1 = vreinterpret_s16_s32(vget_low_s32(cols_1537.val[0]));
|
||||
int16x4_t col2 = vreinterpret_s16_s32(vget_low_s32(cols_0426.val[1]));
|
||||
int16x4_t col3 = vreinterpret_s16_s32(vget_low_s32(cols_1537.val[1]));
|
||||
int16x4_t col5 = vreinterpret_s16_s32(vget_high_s32(cols_1537.val[0]));
|
||||
int16x4_t col6 = vreinterpret_s16_s32(vget_high_s32(cols_0426.val[1]));
|
||||
int16x4_t col7 = vreinterpret_s16_s32(vget_high_s32(cols_1537.val[1]));
|
||||
|
||||
/* Commence second pass of IDCT. */
|
||||
|
||||
/* Even part */
|
||||
int32x4_t tmp0 = vshll_n_s16(col0, CONST_BITS + 1);
|
||||
int32x4_t tmp2 = vmull_lane_s16(col2, consts.val[0], 0);
|
||||
tmp2 = vmlal_lane_s16(tmp2, col6, consts.val[0], 1);
|
||||
|
||||
int32x4_t tmp10 = vaddq_s32(tmp0, tmp2);
|
||||
int32x4_t tmp12 = vsubq_s32(tmp0, tmp2);
|
||||
|
||||
/* Odd part */
|
||||
tmp0 = vmull_lane_s16(col7, consts.val[0], 2);
|
||||
tmp0 = vmlal_lane_s16(tmp0, col5, consts.val[0], 3);
|
||||
tmp0 = vmlal_lane_s16(tmp0, col3, consts.val[1], 0);
|
||||
tmp0 = vmlal_lane_s16(tmp0, col1, consts.val[1], 1);
|
||||
|
||||
tmp2 = vmull_lane_s16(col7, consts.val[1], 2);
|
||||
tmp2 = vmlal_lane_s16(tmp2, col5, consts.val[1], 3);
|
||||
tmp2 = vmlal_lane_s16(tmp2, col3, consts.val[2], 0);
|
||||
tmp2 = vmlal_lane_s16(tmp2, col1, consts.val[2], 1);
|
||||
|
||||
/* Final output stage: descale and clamp to range [0-255]. */
|
||||
int16x8_t output_cols_02 = vcombine_s16(vaddhn_s32(tmp10, tmp2),
|
||||
vsubhn_s32(tmp12, tmp0));
|
||||
int16x8_t output_cols_13 = vcombine_s16(vaddhn_s32(tmp12, tmp0),
|
||||
vsubhn_s32(tmp10, tmp2));
|
||||
output_cols_02 = vrsraq_n_s16(vdupq_n_s16(CENTERJSAMPLE), output_cols_02,
|
||||
CONST_BITS + PASS1_BITS + 3 + 1 - 16);
|
||||
output_cols_13 = vrsraq_n_s16(vdupq_n_s16(CENTERJSAMPLE), output_cols_13,
|
||||
CONST_BITS + PASS1_BITS + 3 + 1 - 16);
|
||||
/* Narrow to 8-bit and convert to unsigned while zipping 8-bit elements.
|
||||
* An interleaving store completes the transpose.
|
||||
*/
|
||||
uint8x8x2_t output_0123 = vzip_u8(vqmovun_s16(output_cols_02),
|
||||
vqmovun_s16(output_cols_13));
|
||||
uint16x4x2_t output_01_23 = { {
|
||||
vreinterpret_u16_u8(output_0123.val[0]),
|
||||
vreinterpret_u16_u8(output_0123.val[1])
|
||||
} };
|
||||
|
||||
/* Store 4x4 block to memory. */
|
||||
JSAMPROW outptr0 = output_buf[0] + output_col;
|
||||
JSAMPROW outptr1 = output_buf[1] + output_col;
|
||||
JSAMPROW outptr2 = output_buf[2] + output_col;
|
||||
JSAMPROW outptr3 = output_buf[3] + output_col;
|
||||
vst2_lane_u16((uint16_t *)outptr0, output_01_23, 0);
|
||||
vst2_lane_u16((uint16_t *)outptr1, output_01_23, 1);
|
||||
vst2_lane_u16((uint16_t *)outptr2, output_01_23, 2);
|
||||
vst2_lane_u16((uint16_t *)outptr3, output_01_23, 3);
|
||||
}
|
||||
+193
@@ -0,0 +1,193 @@
|
||||
/*
|
||||
* jquanti-neon.c - sample data conversion and quantization (Arm Neon)
|
||||
*
|
||||
* Copyright (C) 2020-2021, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#define JPEG_INTERNALS
|
||||
#include "../../jinclude.h"
|
||||
#include "../../jpeglib.h"
|
||||
#include "../../jsimd.h"
|
||||
#include "../../jdct.h"
|
||||
#include "../../jsimddct.h"
|
||||
#include "../jsimd.h"
|
||||
|
||||
#include <arm_neon.h>
|
||||
|
||||
|
||||
/* After downsampling, the resulting sample values are in the range [0, 255],
|
||||
* but the Discrete Cosine Transform (DCT) operates on values centered around
|
||||
* 0.
|
||||
*
|
||||
* To prepare sample values for the DCT, load samples into a DCT workspace,
|
||||
* subtracting CENTERJSAMPLE (128). The samples, now in the range [-128, 127],
|
||||
* are also widened from 8- to 16-bit.
|
||||
*
|
||||
* The equivalent scalar C function convsamp() can be found in jcdctmgr.c.
|
||||
*/
|
||||
|
||||
void jsimd_convsamp_neon(JSAMPARRAY sample_data, JDIMENSION start_col,
|
||||
DCTELEM *workspace)
|
||||
{
|
||||
uint8x8_t samp_row0 = vld1_u8(sample_data[0] + start_col);
|
||||
uint8x8_t samp_row1 = vld1_u8(sample_data[1] + start_col);
|
||||
uint8x8_t samp_row2 = vld1_u8(sample_data[2] + start_col);
|
||||
uint8x8_t samp_row3 = vld1_u8(sample_data[3] + start_col);
|
||||
uint8x8_t samp_row4 = vld1_u8(sample_data[4] + start_col);
|
||||
uint8x8_t samp_row5 = vld1_u8(sample_data[5] + start_col);
|
||||
uint8x8_t samp_row6 = vld1_u8(sample_data[6] + start_col);
|
||||
uint8x8_t samp_row7 = vld1_u8(sample_data[7] + start_col);
|
||||
|
||||
int16x8_t row0 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row0, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row1 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row1, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row2 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row2, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row3 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row3, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row4 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row4, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row5 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row5, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row6 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row6, vdup_n_u8(CENTERJSAMPLE)));
|
||||
int16x8_t row7 =
|
||||
vreinterpretq_s16_u16(vsubl_u8(samp_row7, vdup_n_u8(CENTERJSAMPLE)));
|
||||
|
||||
vst1q_s16(workspace + 0 * DCTSIZE, row0);
|
||||
vst1q_s16(workspace + 1 * DCTSIZE, row1);
|
||||
vst1q_s16(workspace + 2 * DCTSIZE, row2);
|
||||
vst1q_s16(workspace + 3 * DCTSIZE, row3);
|
||||
vst1q_s16(workspace + 4 * DCTSIZE, row4);
|
||||
vst1q_s16(workspace + 5 * DCTSIZE, row5);
|
||||
vst1q_s16(workspace + 6 * DCTSIZE, row6);
|
||||
vst1q_s16(workspace + 7 * DCTSIZE, row7);
|
||||
}
|
||||
|
||||
|
||||
/* After the DCT, the resulting array of coefficient values needs to be divided
|
||||
* by an array of quantization values.
|
||||
*
|
||||
* To avoid a slow division operation, the DCT coefficients are multiplied by
|
||||
* the (scaled) reciprocals of the quantization values and then right-shifted.
|
||||
*
|
||||
* The equivalent scalar C function quantize() can be found in jcdctmgr.c.
|
||||
*/
|
||||
|
||||
void jsimd_quantize_neon(JCOEFPTR coef_block, DCTELEM *divisors,
|
||||
DCTELEM *workspace)
|
||||
{
|
||||
JCOEFPTR out_ptr = coef_block;
|
||||
UDCTELEM *recip_ptr = (UDCTELEM *)divisors;
|
||||
UDCTELEM *corr_ptr = (UDCTELEM *)divisors + DCTSIZE2;
|
||||
DCTELEM *shift_ptr = divisors + 3 * DCTSIZE2;
|
||||
int i;
|
||||
|
||||
#if defined(__clang__) && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
#pragma unroll
|
||||
#endif
|
||||
for (i = 0; i < DCTSIZE; i += DCTSIZE / 2) {
|
||||
/* Load DCT coefficients. */
|
||||
int16x8_t row0 = vld1q_s16(workspace + (i + 0) * DCTSIZE);
|
||||
int16x8_t row1 = vld1q_s16(workspace + (i + 1) * DCTSIZE);
|
||||
int16x8_t row2 = vld1q_s16(workspace + (i + 2) * DCTSIZE);
|
||||
int16x8_t row3 = vld1q_s16(workspace + (i + 3) * DCTSIZE);
|
||||
/* Load reciprocals of quantization values. */
|
||||
uint16x8_t recip0 = vld1q_u16(recip_ptr + (i + 0) * DCTSIZE);
|
||||
uint16x8_t recip1 = vld1q_u16(recip_ptr + (i + 1) * DCTSIZE);
|
||||
uint16x8_t recip2 = vld1q_u16(recip_ptr + (i + 2) * DCTSIZE);
|
||||
uint16x8_t recip3 = vld1q_u16(recip_ptr + (i + 3) * DCTSIZE);
|
||||
uint16x8_t corr0 = vld1q_u16(corr_ptr + (i + 0) * DCTSIZE);
|
||||
uint16x8_t corr1 = vld1q_u16(corr_ptr + (i + 1) * DCTSIZE);
|
||||
uint16x8_t corr2 = vld1q_u16(corr_ptr + (i + 2) * DCTSIZE);
|
||||
uint16x8_t corr3 = vld1q_u16(corr_ptr + (i + 3) * DCTSIZE);
|
||||
int16x8_t shift0 = vld1q_s16(shift_ptr + (i + 0) * DCTSIZE);
|
||||
int16x8_t shift1 = vld1q_s16(shift_ptr + (i + 1) * DCTSIZE);
|
||||
int16x8_t shift2 = vld1q_s16(shift_ptr + (i + 2) * DCTSIZE);
|
||||
int16x8_t shift3 = vld1q_s16(shift_ptr + (i + 3) * DCTSIZE);
|
||||
|
||||
/* Extract sign from coefficients. */
|
||||
int16x8_t sign_row0 = vshrq_n_s16(row0, 15);
|
||||
int16x8_t sign_row1 = vshrq_n_s16(row1, 15);
|
||||
int16x8_t sign_row2 = vshrq_n_s16(row2, 15);
|
||||
int16x8_t sign_row3 = vshrq_n_s16(row3, 15);
|
||||
/* Get absolute value of DCT coefficients. */
|
||||
uint16x8_t abs_row0 = vreinterpretq_u16_s16(vabsq_s16(row0));
|
||||
uint16x8_t abs_row1 = vreinterpretq_u16_s16(vabsq_s16(row1));
|
||||
uint16x8_t abs_row2 = vreinterpretq_u16_s16(vabsq_s16(row2));
|
||||
uint16x8_t abs_row3 = vreinterpretq_u16_s16(vabsq_s16(row3));
|
||||
/* Add correction. */
|
||||
abs_row0 = vaddq_u16(abs_row0, corr0);
|
||||
abs_row1 = vaddq_u16(abs_row1, corr1);
|
||||
abs_row2 = vaddq_u16(abs_row2, corr2);
|
||||
abs_row3 = vaddq_u16(abs_row3, corr3);
|
||||
|
||||
/* Multiply DCT coefficients by quantization reciprocals. */
|
||||
int32x4_t row0_l = vreinterpretq_s32_u32(vmull_u16(vget_low_u16(abs_row0),
|
||||
vget_low_u16(recip0)));
|
||||
int32x4_t row0_h = vreinterpretq_s32_u32(vmull_u16(vget_high_u16(abs_row0),
|
||||
vget_high_u16(recip0)));
|
||||
int32x4_t row1_l = vreinterpretq_s32_u32(vmull_u16(vget_low_u16(abs_row1),
|
||||
vget_low_u16(recip1)));
|
||||
int32x4_t row1_h = vreinterpretq_s32_u32(vmull_u16(vget_high_u16(abs_row1),
|
||||
vget_high_u16(recip1)));
|
||||
int32x4_t row2_l = vreinterpretq_s32_u32(vmull_u16(vget_low_u16(abs_row2),
|
||||
vget_low_u16(recip2)));
|
||||
int32x4_t row2_h = vreinterpretq_s32_u32(vmull_u16(vget_high_u16(abs_row2),
|
||||
vget_high_u16(recip2)));
|
||||
int32x4_t row3_l = vreinterpretq_s32_u32(vmull_u16(vget_low_u16(abs_row3),
|
||||
vget_low_u16(recip3)));
|
||||
int32x4_t row3_h = vreinterpretq_s32_u32(vmull_u16(vget_high_u16(abs_row3),
|
||||
vget_high_u16(recip3)));
|
||||
/* Narrow back to 16-bit. */
|
||||
row0 = vcombine_s16(vshrn_n_s32(row0_l, 16), vshrn_n_s32(row0_h, 16));
|
||||
row1 = vcombine_s16(vshrn_n_s32(row1_l, 16), vshrn_n_s32(row1_h, 16));
|
||||
row2 = vcombine_s16(vshrn_n_s32(row2_l, 16), vshrn_n_s32(row2_h, 16));
|
||||
row3 = vcombine_s16(vshrn_n_s32(row3_l, 16), vshrn_n_s32(row3_h, 16));
|
||||
|
||||
/* Since VSHR only supports an immediate as its second argument, negate the
|
||||
* shift value and shift left.
|
||||
*/
|
||||
row0 = vreinterpretq_s16_u16(vshlq_u16(vreinterpretq_u16_s16(row0),
|
||||
vnegq_s16(shift0)));
|
||||
row1 = vreinterpretq_s16_u16(vshlq_u16(vreinterpretq_u16_s16(row1),
|
||||
vnegq_s16(shift1)));
|
||||
row2 = vreinterpretq_s16_u16(vshlq_u16(vreinterpretq_u16_s16(row2),
|
||||
vnegq_s16(shift2)));
|
||||
row3 = vreinterpretq_s16_u16(vshlq_u16(vreinterpretq_u16_s16(row3),
|
||||
vnegq_s16(shift3)));
|
||||
|
||||
/* Restore sign to original product. */
|
||||
row0 = veorq_s16(row0, sign_row0);
|
||||
row0 = vsubq_s16(row0, sign_row0);
|
||||
row1 = veorq_s16(row1, sign_row1);
|
||||
row1 = vsubq_s16(row1, sign_row1);
|
||||
row2 = veorq_s16(row2, sign_row2);
|
||||
row2 = vsubq_s16(row2, sign_row2);
|
||||
row3 = veorq_s16(row3, sign_row3);
|
||||
row3 = vsubq_s16(row3, sign_row3);
|
||||
|
||||
/* Store quantized coefficients to memory. */
|
||||
vst1q_s16(out_ptr + (i + 0) * DCTSIZE, row0);
|
||||
vst1q_s16(out_ptr + (i + 1) * DCTSIZE, row1);
|
||||
vst1q_s16(out_ptr + (i + 2) * DCTSIZE, row2);
|
||||
vst1q_s16(out_ptr + (i + 3) * DCTSIZE, row3);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
/*
|
||||
* Copyright (C) 2020, D. R. Commander. All Rights Reserved.
|
||||
* Copyright (C) 2020-2021, Arm Limited. All Rights Reserved.
|
||||
*
|
||||
* This software is provided 'as-is', without any express or implied
|
||||
* warranty. In no event will the authors be held liable for any damages
|
||||
* arising from the use of this software.
|
||||
*
|
||||
* Permission is granted to anyone to use this software for any purpose,
|
||||
* including commercial applications, and to alter it and redistribute it
|
||||
* freely, subject to the following restrictions:
|
||||
*
|
||||
* 1. The origin of this software must not be misrepresented; you must not
|
||||
* claim that you wrote the original software. If you use this software
|
||||
* in a product, an acknowledgment in the product documentation would be
|
||||
* appreciated but is not required.
|
||||
* 2. Altered source versions must be plainly marked as such, and must not be
|
||||
* misrepresented as being the original software.
|
||||
* 3. This notice may not be removed or altered from any source distribution.
|
||||
*/
|
||||
|
||||
#cmakedefine HAVE_VLD1_S16_X3
|
||||
#cmakedefine HAVE_VLD1_U16_X2
|
||||
#cmakedefine HAVE_VLD1Q_U8_X4
|
||||
|
||||
/* Define compiler-independent count-leading-zeros and byte-swap macros */
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#define BUILTIN_CLZ(x) _CountLeadingZeros(x)
|
||||
#define BUILTIN_CLZLL(x) _CountLeadingZeros64(x)
|
||||
#define BUILTIN_BSWAP64(x) _byteswap_uint64(x)
|
||||
#elif defined(__clang__) || defined(__GNUC__)
|
||||
#define BUILTIN_CLZ(x) __builtin_clz(x)
|
||||
#define BUILTIN_CLZLL(x) __builtin_clzll(x)
|
||||
#define BUILTIN_BSWAP64(x) __builtin_bswap64(x)
|
||||
#else
|
||||
#error "Unknown compiler"
|
||||
#endif
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user