Fix alpha coverage for lerp_565 stage.
authorbungeman <bungeman@google.com>
Wed, 10 May 2017 17:50:12 +0000 (13:50 -0400)
committerSkia Commit-Bot <skia-commit-bot@chromium.org>
Wed, 10 May 2017 18:20:25 +0000 (18:20 +0000)
Like three gray with alpha blends and taking the max alpha.

Change-Id: I104c84f784979030744127f8f66905ad9d1bdf0e
Reviewed-on: https://skia-review.googlesource.com/15898
Commit-Queue: Ben Wagner <bungeman@google.com>
Reviewed-by: Mike Klein <mtklein@chromium.org>
gm/lcdtext.cpp
src/jumper/SkJumper_generated.S
src/jumper/SkJumper_generated_win.S
src/jumper/SkJumper_stages.cpp

index 41e83d6..a48856f 100644 (file)
@@ -146,13 +146,8 @@ DEF_SIMPLE_GM(savelayer_lcdtext, canvas, 620, 260) {
     for (auto preserve : gPreserveLCDText) {
         preserve ? canvas->saveLayerPreserveLCDTextRequests(nullptr, nullptr)
                  : canvas->saveLayer(nullptr, nullptr);
-        if (preserve) {
-            SkPaint noLCD = paint;
-            noLCD.setLCDRenderText(false);
-            canvas->drawString("LCD not supported", 30, 60, noLCD);
-        } else {
-            canvas->drawString("Hamburgefons", 30, 60, paint);
-        }
+
+        canvas->drawString("Hamburgefons", 30, 60, paint);
 
         SkPaint p;
         p.setColor(0xFFCCCCCC);
index eb0d913..590452b 100644 (file)
@@ -1653,41 +1653,49 @@ FUNCTION(_sk_lerp_565_aarch64)
 _sk_lerp_565_aarch64:
   .long  0xa8c10c28                          // ldp           x8, x3, [x1], #16
   .long  0xd37ff809                          // lsl           x9, x0, #1
-  .long  0x4f072710                          // movi          v16.4s, #0xf8, lsl #8
-  .long  0x4ea4d413                          // fsub          v19.4s, v0.4s, v4.4s
+  .long  0x4f072711                          // movi          v17.4s, #0xf8, lsl #8
+  .long  0x4ea4d416                          // fsub          v22.4s, v0.4s, v4.4s
   .long  0xf9400108                          // ldr           x8, [x8]
-  .long  0xfc696903                          // ldr           d3, [x8, x9]
+  .long  0x4f0007f4                          // movi          v20.4s, #0x1f
+  .long  0x4ea7d463                          // fsub          v3.4s, v3.4s, v7.4s
+  .long  0xfc696910                          // ldr           d16, [x8, x9]
   .long  0x52a6f088                          // mov           w8, #0x37840000
   .long  0x72842108                          // movk          w8, #0x2108
-  .long  0x4e040d11                          // dup           v17.4s, w8
-  .long  0x2f10a463                          // uxtl          v3.4s, v3.4h
-  .long  0x321b17e8                          // orr           w8, wzr, #0x7e0
-  .long  0x4e301c60                          // and           v0.16b, v3.16b, v16.16b
   .long  0x4e040d12                          // dup           v18.4s, w8
+  .long  0x321b17e8                          // orr           w8, wzr, #0x7e0
+  .long  0x4e040d13                          // dup           v19.4s, w8
   .long  0x52a74048                          // mov           w8, #0x3a020000
-  .long  0x4e21d800                          // scvtf         v0.4s, v0.4s
+  .long  0x2f10a600                          // uxtl          v0.4s, v16.4h
   .long  0x72810428                          // movk          w8, #0x821
-  .long  0x6e31dc10                          // fmul          v16.4s, v0.4s, v17.4s
-  .long  0x4ea41c80                          // mov           v0.16b, v4.16b
-  .long  0x4e33ce00                          // fmla          v0.4s, v16.4s, v19.4s
-  .long  0x4f0007f0                          // movi          v16.4s, #0x1f
-  .long  0x4e040d11                          // dup           v17.4s, w8
+  .long  0x4e311c10                          // and           v16.16b, v0.16b, v17.16b
+  .long  0x4e040d15                          // dup           v21.4s, w8
   .long  0x52a7a088                          // mov           w8, #0x3d040000
-  .long  0x4e321c72                          // and           v18.16b, v3.16b, v18.16b
+  .long  0x4e331c11                          // and           v17.16b, v0.16b, v19.16b
+  .long  0x4e341c13                          // and           v19.16b, v0.16b, v20.16b
+  .long  0x4ea5d434                          // fsub          v20.4s, v1.4s, v5.4s
+  .long  0x4e21da01                          // scvtf         v1.4s, v16.4s
   .long  0x72842108                          // movk          w8, #0x2108
-  .long  0x4e301c63                          // and           v3.16b, v3.16b, v16.16b
-  .long  0x4ea6d450                          // fsub          v16.4s, v2.4s, v6.4s
-  .long  0x4e21da42                          // scvtf         v2.4s, v18.4s
-  .long  0x6e31dc51                          // fmul          v17.4s, v2.4s, v17.4s
-  .long  0x4e040d02                          // dup           v2.4s, w8
-  .long  0x4e21d863                          // scvtf         v3.4s, v3.4s
-  .long  0x4ea5d433                          // fsub          v19.4s, v1.4s, v5.4s
+  .long  0x6e32dc30                          // fmul          v16.4s, v1.4s, v18.4s
+  .long  0x4ea6d452                          // fsub          v18.4s, v2.4s, v6.4s
+  .long  0x4e21da22                          // scvtf         v2.4s, v17.4s
   .long  0x4ea51ca1                          // mov           v1.16b, v5.16b
-  .long  0x6e22dc63                          // fmul          v3.4s, v3.4s, v2.4s
+  .long  0x6e35dc51                          // fmul          v17.4s, v2.4s, v21.4s
+  .long  0x4e040d02                          // dup           v2.4s, w8
+  .long  0x4e21da73                          // scvtf         v19.4s, v19.4s
+  .long  0x6e22de73                          // fmul          v19.4s, v19.4s, v2.4s
   .long  0x4ea61cc2                          // mov           v2.16b, v6.16b
-  .long  0x4e33ce21                          // fmla          v1.4s, v17.4s, v19.4s
-  .long  0x4e30cc62                          // fmla          v2.4s, v3.4s, v16.4s
-  .long  0x4f03f603                          // fmov          v3.4s, #1.000000000000000000e+00
+  .long  0x4ea71cf5                          // mov           v21.16b, v7.16b
+  .long  0x4e34ce21                          // fmla          v1.4s, v17.4s, v20.4s
+  .long  0x4ea71cf4                          // mov           v20.16b, v7.16b
+  .long  0x4e32ce62                          // fmla          v2.4s, v19.4s, v18.4s
+  .long  0x4ea71cf2                          // mov           v18.16b, v7.16b
+  .long  0x4e23ce35                          // fmla          v21.4s, v17.4s, v3.4s
+  .long  0x4e23ce74                          // fmla          v20.4s, v19.4s, v3.4s
+  .long  0x4ea41c80                          // mov           v0.16b, v4.16b
+  .long  0x4e23ce12                          // fmla          v18.4s, v16.4s, v3.4s
+  .long  0x4e34f6a3                          // fmax          v3.4s, v21.4s, v20.4s
+  .long  0x4e36ce00                          // fmla          v0.4s, v16.4s, v22.4s
+  .long  0x4e23f643                          // fmax          v3.4s, v18.4s, v3.4s
   .long  0xd61f0060                          // br            x3
 
 HIDDEN _sk_load_tables_aarch64
@@ -2652,9 +2660,9 @@ FUNCTION(_sk_gather_i8_aarch64)
 _sk_gather_i8_aarch64:
   .long  0xaa0103e8                          // mov           x8, x1
   .long  0xf8408429                          // ldr           x9, [x1], #8
-  .long  0xb4000069                          // cbz           x9, 2394 <sk_gather_i8_aarch64+0x14>
+  .long  0xb4000069                          // cbz           x9, 23b4 <sk_gather_i8_aarch64+0x14>
   .long  0xaa0903ea                          // mov           x10, x9
-  .long  0x14000003                          // b             239c <sk_gather_i8_aarch64+0x1c>
+  .long  0x14000003                          // b             23bc <sk_gather_i8_aarch64+0x1c>
   .long  0xf940050a                          // ldr           x10, [x8, #8]
   .long  0x91004101                          // add           x1, x8, #0x10
   .long  0xf8410548                          // ldr           x8, [x10], #16
@@ -3503,7 +3511,7 @@ _sk_linear_gradient_aarch64:
   .long  0x4d40c902                          // ld1r          {v2.4s}, [x8]
   .long  0xf9400128                          // ldr           x8, [x9]
   .long  0x4d40c943                          // ld1r          {v3.4s}, [x10]
-  .long  0xb40006c8                          // cbz           x8, 2f68 <sk_linear_gradient_aarch64+0x100>
+  .long  0xb40006c8                          // cbz           x8, 2f88 <sk_linear_gradient_aarch64+0x100>
   .long  0x6dbf23e9                          // stp           d9, d8, [sp, #-16]!
   .long  0xf9400529                          // ldr           x9, [x9, #8]
   .long  0x6f00e413                          // movi          v19.2d, #0x0
@@ -3554,9 +3562,9 @@ _sk_linear_gradient_aarch64:
   .long  0xd1000508                          // sub           x8, x8, #0x1
   .long  0x6e771fd0                          // bsl           v16.16b, v30.16b, v23.16b
   .long  0x91009129                          // add           x9, x9, #0x24
-  .long  0xb5fffaa8                          // cbnz          x8, 2eb0 <sk_linear_gradient_aarch64+0x48>
+  .long  0xb5fffaa8                          // cbnz          x8, 2ed0 <sk_linear_gradient_aarch64+0x48>
   .long  0x6cc123e9                          // ldp           d9, d8, [sp], #16
-  .long  0x14000005                          // b             2f78 <sk_linear_gradient_aarch64+0x110>
+  .long  0x14000005                          // b             2f98 <sk_linear_gradient_aarch64+0x110>
   .long  0x6f00e414                          // movi          v20.2d, #0x0
   .long  0x6f00e412                          // movi          v18.2d, #0x0
   .long  0x6f00e411                          // movi          v17.2d, #0x0
@@ -5801,48 +5809,56 @@ FUNCTION(_sk_lerp_565_vfp4)
 _sk_lerp_565_vfp4:
   .long  0xe24dd004                          // sub           sp, sp, #4
   .long  0xe8911008                          // ldm           r1, {r3, ip}
-  .long  0xf3c72218                          // vmov.i32      d18, #63488
-  .long  0xf2c1101f                          // vmov.i32      d17, #31
-  .long  0xf2603d04                          // vsub.f32      d19, d0, d4
-  .long  0xe2811008                          // add           r1, r1, #8
+  .long  0xf2c1201f                          // vmov.i32      d18, #31
+  .long  0xf3c71218                          // vmov.i32      d17, #63488
+  .long  0xeddf3b28                          // vldr          d19, [pc, #160]
+  .long  0xf2679117                          // vorr          d25, d7, d7
   .long  0xe5933000                          // ldr           r3, [r3]
-  .long  0xf2616d05                          // vsub.f32      d22, d1, d5
-  .long  0xf2240114                          // vorr          d0, d4, d4
+  .long  0xf2617d05                          // vsub.f32      d23, d1, d5
   .long  0xf2251115                          // vorr          d1, d5, d5
+  .long  0xe2811008                          // add           r1, r1, #8
+  .long  0xf2606d04                          // vsub.f32      d22, d0, d4
   .long  0xe7933080                          // ldr           r3, [r3, r0, lsl #1]
-  .long  0xf2873f10                          // vmov.f32      d3, #1
+  .long  0xf2628d06                          // vsub.f32      d24, d2, d6
+  .long  0xf2240114                          // vorr          d0, d4, d4
   .long  0xe58d3000                          // str           r3, [sp]
   .long  0xe1a0300d                          // mov           r3, sp
   .long  0xf4e3083f                          // vld1.32       {d16[0]}, [r3 :32]
   .long  0xe3a03e7e                          // mov           r3, #2016
+  .long  0xf2262116                          // vorr          d2, d6, d6
   .long  0xf3d04a30                          // vmovl.u16     q10, d16
   .long  0xee803b90                          // vdup.32       d16, r3
   .long  0xf24421b2                          // vand          d18, d20, d18
-  .long  0xf24411b1                          // vand          d17, d20, d17
-  .long  0xeddf5b12                          // vldr          d21, [pc, #72]
   .long  0xf24401b0                          // vand          d16, d20, d16
-  .long  0xeddf4b0e                          // vldr          d20, [pc, #56]
   .long  0xf3fb2622                          // vcvt.f32.s32  d18, d18
   .long  0xf3fb0620                          // vcvt.f32.s32  d16, d16
+  .long  0xf24411b1                          // vand          d17, d20, d17
+  .long  0xeddf4b14                          // vldr          d20, [pc, #80]
+  .long  0xf2635d07                          // vsub.f32      d21, d3, d7
   .long  0xf3fb1621                          // vcvt.f32.s32  d17, d17
-  .long  0xf3422db4                          // vmul.f32      d18, d18, d20
-  .long  0xeddf4b0d                          // vldr          d20, [pc, #52]
-  .long  0xf3400db5                          // vmul.f32      d16, d16, d21
-  .long  0xf2625d06                          // vsub.f32      d21, d2, d6
-  .long  0xf3411db4                          // vmul.f32      d17, d17, d20
-  .long  0xf2262116                          // vorr          d2, d6, d6
-  .long  0xf2030cb2                          // vfma.f32      d0, d19, d18
-  .long  0xf2061cb0                          // vfma.f32      d1, d22, d16
-  .long  0xf2052cb1                          // vfma.f32      d2, d21, d17
+  .long  0xf3422db3                          // vmul.f32      d18, d18, d19
+  .long  0xeddf3b12                          // vldr          d19, [pc, #72]
+  .long  0xf3400db4                          // vmul.f32      d16, d16, d20
+  .long  0xf2674117                          // vorr          d20, d7, d7
+  .long  0xf3411db3                          // vmul.f32      d17, d17, d19
+  .long  0xf2673117                          // vorr          d19, d7, d7
+  .long  0xf2453cb2                          // vfma.f32      d19, d21, d18
+  .long  0xf2454cb0                          // vfma.f32      d20, d21, d16
+  .long  0xf2459cb1                          // vfma.f32      d25, d21, d17
+  .long  0xf2071cb0                          // vfma.f32      d1, d23, d16
+  .long  0xf2060cb1                          // vfma.f32      d0, d22, d17
+  .long  0xf2082cb2                          // vfma.f32      d2, d24, d18
+  .long  0xf2440fa3                          // vmax.f32      d16, d20, d19
+  .long  0xf2093fa0                          // vmax.f32      d3, d25, d16
   .long  0xe28dd004                          // add           sp, sp, #4
   .long  0xe12fff1c                          // bx            ip
   .long  0xe320f000                          // nop           {0}
-  .long  0x37842108                          // .word         0x37842108
-  .long  0x37842108                          // .word         0x37842108
-  .long  0x3a020821                          // .word         0x3a020821
-  .long  0x3a020821                          // .word         0x3a020821
   .long  0x3d042108                          // .word         0x3d042108
   .long  0x3d042108                          // .word         0x3d042108
+  .long  0x3a020821                          // .word         0x3a020821
+  .long  0x3a020821                          // .word         0x3a020821
+  .long  0x37842108                          // .word         0x37842108
+  .long  0x37842108                          // .word         0x37842108
 
 HIDDEN _sk_load_tables_vfp4
 .globl _sk_load_tables_vfp4
@@ -7840,7 +7856,7 @@ _sk_linear_gradient_vfp4:
   .long  0xe494c00c                          // ldr           ip, [r4], #12
   .long  0xf4a41c9f                          // vld1.32       {d1[]}, [r4 :32]
   .long  0xe35c0000                          // cmp           ip, #0
-  .long  0x0a000036                          // beq           3558 <sk_linear_gradient_vfp4+0x110>
+  .long  0x0a000036                          // beq           3578 <sk_linear_gradient_vfp4+0x110>
   .long  0xe59e3004                          // ldr           r3, [lr, #4]
   .long  0xf2c01010                          // vmov.i32      d17, #0
   .long  0xf2c07010                          // vmov.i32      d23, #0
@@ -7890,12 +7906,12 @@ _sk_linear_gradient_vfp4:
   .long  0xf26371b3                          // vorr          d23, d19, d19
   .long  0xf26481b4                          // vorr          d24, d20, d20
   .long  0xf26561b5                          // vorr          d22, d21, d21
-  .long  0x1affffd3                          // bne           3494 <sk_linear_gradient_vfp4+0x4c>
+  .long  0x1affffd3                          // bne           34b4 <sk_linear_gradient_vfp4+0x4c>
   .long  0xf26c01bc                          // vorr          d16, d28, d28
   .long  0xf22b11bb                          // vorr          d1, d27, d27
   .long  0xf22a21ba                          // vorr          d2, d26, d26
   .long  0xf22931b9                          // vorr          d3, d25, d25
-  .long  0xea000003                          // b             3568 <sk_linear_gradient_vfp4+0x120>
+  .long  0xea000003                          // b             3588 <sk_linear_gradient_vfp4+0x120>
   .long  0xf2c05010                          // vmov.i32      d21, #0
   .long  0xf2c04010                          // vmov.i32      d20, #0
   .long  0xf2c03010                          // vmov.i32      d19, #0
@@ -8445,14 +8461,14 @@ _sk_seed_shader_hsw:
   .byte  197,249,110,199                     // vmovd         %edi,%xmm0
   .byte  196,226,125,88,192                  // vpbroadcastd  %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,61,68,0,0         // vbroadcastss  0x443d(%rip),%ymm1        # 4500 <_sk_callback_hsw+0x125>
+  .byte  196,226,125,24,13,81,68,0,0         // vbroadcastss  0x4451(%rip),%ymm1        # 4514 <_sk_callback_hsw+0x125>
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
   .byte  197,252,88,2                        // vaddps        (%rdx),%ymm0,%ymm0
   .byte  196,226,125,24,16                   // vbroadcastss  (%rax),%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  197,236,88,201                      // vaddps        %ymm1,%ymm2,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,21,33,68,0,0         // vbroadcastss  0x4421(%rip),%ymm2        # 4504 <_sk_callback_hsw+0x129>
+  .byte  196,226,125,24,21,53,68,0,0         // vbroadcastss  0x4435(%rip),%ymm2        # 4518 <_sk_callback_hsw+0x129>
   .byte  197,228,87,219                      // vxorps        %ymm3,%ymm3,%ymm3
   .byte  197,220,87,228                      // vxorps        %ymm4,%ymm4,%ymm4
   .byte  197,212,87,237                      // vxorps        %ymm5,%ymm5,%ymm5
@@ -8473,13 +8489,13 @@ _sk_dither_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  196,66,125,88,8                     // vpbroadcastd  (%r8),%ymm9
   .byte  196,65,61,239,201                   // vpxor         %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,88,21,224,67,0,0         // vpbroadcastd  0x43e0(%rip),%ymm10        # 4508 <_sk_callback_hsw+0x12d>
+  .byte  196,98,125,88,21,244,67,0,0         // vpbroadcastd  0x43f4(%rip),%ymm10        # 451c <_sk_callback_hsw+0x12d>
   .byte  196,65,53,219,218                   // vpand         %ymm10,%ymm9,%ymm11
   .byte  196,193,37,114,243,5                // vpslld        $0x5,%ymm11,%ymm11
   .byte  196,65,61,219,210                   // vpand         %ymm10,%ymm8,%ymm10
   .byte  196,193,45,114,242,4                // vpslld        $0x4,%ymm10,%ymm10
-  .byte  196,98,125,88,37,197,67,0,0         // vpbroadcastd  0x43c5(%rip),%ymm12        # 450c <_sk_callback_hsw+0x131>
-  .byte  196,98,125,88,45,192,67,0,0         // vpbroadcastd  0x43c0(%rip),%ymm13        # 4510 <_sk_callback_hsw+0x135>
+  .byte  196,98,125,88,37,217,67,0,0         // vpbroadcastd  0x43d9(%rip),%ymm12        # 4520 <_sk_callback_hsw+0x131>
+  .byte  196,98,125,88,45,212,67,0,0         // vpbroadcastd  0x43d4(%rip),%ymm13        # 4524 <_sk_callback_hsw+0x135>
   .byte  196,65,53,219,245                   // vpand         %ymm13,%ymm9,%ymm14
   .byte  196,193,13,114,246,2                // vpslld        $0x2,%ymm14,%ymm14
   .byte  196,65,61,219,237                   // vpand         %ymm13,%ymm8,%ymm13
@@ -8494,8 +8510,8 @@ _sk_dither_hsw:
   .byte  196,65,61,235,194                   // vpor          %ymm10,%ymm8,%ymm8
   .byte  196,65,61,235,193                   // vpor          %ymm9,%ymm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,114,67,0,0         // vbroadcastss  0x4372(%rip),%ymm9        # 4514 <_sk_callback_hsw+0x139>
-  .byte  196,98,125,24,21,109,67,0,0         // vbroadcastss  0x436d(%rip),%ymm10        # 4518 <_sk_callback_hsw+0x13d>
+  .byte  196,98,125,24,13,134,67,0,0         // vbroadcastss  0x4386(%rip),%ymm9        # 4528 <_sk_callback_hsw+0x139>
+  .byte  196,98,125,24,21,129,67,0,0         // vbroadcastss  0x4381(%rip),%ymm10        # 452c <_sk_callback_hsw+0x13d>
   .byte  196,66,61,184,209                   // vfmadd231ps   %ymm9,%ymm8,%ymm10
   .byte  196,98,125,24,64,8                  // vbroadcastss  0x8(%rax),%ymm8
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
@@ -8533,7 +8549,7 @@ HIDDEN _sk_srcatop_hsw
 FUNCTION(_sk_srcatop_hsw)
 _sk_srcatop_hsw:
   .byte  197,252,89,199                      // vmulps        %ymm7,%ymm0,%ymm0
-  .byte  196,98,125,24,5,19,67,0,0           // vbroadcastss  0x4313(%rip),%ymm8        # 451c <_sk_callback_hsw+0x141>
+  .byte  196,98,125,24,5,39,67,0,0           // vbroadcastss  0x4327(%rip),%ymm8        # 4530 <_sk_callback_hsw+0x141>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,226,61,184,196                  // vfmadd231ps   %ymm4,%ymm8,%ymm0
   .byte  197,244,89,207                      // vmulps        %ymm7,%ymm1,%ymm1
@@ -8549,7 +8565,7 @@ HIDDEN _sk_dstatop_hsw
 .globl _sk_dstatop_hsw
 FUNCTION(_sk_dstatop_hsw)
 _sk_dstatop_hsw:
-  .byte  196,98,125,24,5,230,66,0,0          // vbroadcastss  0x42e6(%rip),%ymm8        # 4520 <_sk_callback_hsw+0x145>
+  .byte  196,98,125,24,5,250,66,0,0          // vbroadcastss  0x42fa(%rip),%ymm8        # 4534 <_sk_callback_hsw+0x145>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  196,226,101,184,196                 // vfmadd231ps   %ymm4,%ymm3,%ymm0
@@ -8588,7 +8604,7 @@ HIDDEN _sk_srcout_hsw
 .globl _sk_srcout_hsw
 FUNCTION(_sk_srcout_hsw)
 _sk_srcout_hsw:
-  .byte  196,98,125,24,5,141,66,0,0          // vbroadcastss  0x428d(%rip),%ymm8        # 4524 <_sk_callback_hsw+0x149>
+  .byte  196,98,125,24,5,161,66,0,0          // vbroadcastss  0x42a1(%rip),%ymm8        # 4538 <_sk_callback_hsw+0x149>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -8601,7 +8617,7 @@ HIDDEN _sk_dstout_hsw
 .globl _sk_dstout_hsw
 FUNCTION(_sk_dstout_hsw)
 _sk_dstout_hsw:
-  .byte  196,226,125,24,5,112,66,0,0         // vbroadcastss  0x4270(%rip),%ymm0        # 4528 <_sk_callback_hsw+0x14d>
+  .byte  196,226,125,24,5,132,66,0,0         // vbroadcastss  0x4284(%rip),%ymm0        # 453c <_sk_callback_hsw+0x14d>
   .byte  197,252,92,219                      // vsubps        %ymm3,%ymm0,%ymm3
   .byte  197,228,89,196                      // vmulps        %ymm4,%ymm3,%ymm0
   .byte  197,228,89,205                      // vmulps        %ymm5,%ymm3,%ymm1
@@ -8614,7 +8630,7 @@ HIDDEN _sk_srcover_hsw
 .globl _sk_srcover_hsw
 FUNCTION(_sk_srcover_hsw)
 _sk_srcover_hsw:
-  .byte  196,98,125,24,5,83,66,0,0           // vbroadcastss  0x4253(%rip),%ymm8        # 452c <_sk_callback_hsw+0x151>
+  .byte  196,98,125,24,5,103,66,0,0          // vbroadcastss  0x4267(%rip),%ymm8        # 4540 <_sk_callback_hsw+0x151>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,93,184,192                  // vfmadd231ps   %ymm8,%ymm4,%ymm0
   .byte  196,194,85,184,200                  // vfmadd231ps   %ymm8,%ymm5,%ymm1
@@ -8627,7 +8643,7 @@ HIDDEN _sk_dstover_hsw
 .globl _sk_dstover_hsw
 FUNCTION(_sk_dstover_hsw)
 _sk_dstover_hsw:
-  .byte  196,98,125,24,5,50,66,0,0           // vbroadcastss  0x4232(%rip),%ymm8        # 4530 <_sk_callback_hsw+0x155>
+  .byte  196,98,125,24,5,70,66,0,0           // vbroadcastss  0x4246(%rip),%ymm8        # 4544 <_sk_callback_hsw+0x155>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  196,226,61,168,196                  // vfmadd213ps   %ymm4,%ymm8,%ymm0
   .byte  196,226,61,168,205                  // vfmadd213ps   %ymm5,%ymm8,%ymm1
@@ -8651,7 +8667,7 @@ HIDDEN _sk_multiply_hsw
 .globl _sk_multiply_hsw
 FUNCTION(_sk_multiply_hsw)
 _sk_multiply_hsw:
-  .byte  196,98,125,24,5,253,65,0,0          // vbroadcastss  0x41fd(%rip),%ymm8        # 4534 <_sk_callback_hsw+0x159>
+  .byte  196,98,125,24,5,17,66,0,0           // vbroadcastss  0x4211(%rip),%ymm8        # 4548 <_sk_callback_hsw+0x159>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,208                       // vmulps        %ymm0,%ymm9,%ymm10
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -8699,7 +8715,7 @@ HIDDEN _sk_xor__hsw
 .globl _sk_xor__hsw
 FUNCTION(_sk_xor__hsw)
 _sk_xor__hsw:
-  .byte  196,98,125,24,5,120,65,0,0          // vbroadcastss  0x4178(%rip),%ymm8        # 4538 <_sk_callback_hsw+0x15d>
+  .byte  196,98,125,24,5,140,65,0,0          // vbroadcastss  0x418c(%rip),%ymm8        # 454c <_sk_callback_hsw+0x15d>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -8733,7 +8749,7 @@ _sk_darken_hsw:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,95,209                  // vmaxps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,0,65,0,0            // vbroadcastss  0x4100(%rip),%ymm8        # 453c <_sk_callback_hsw+0x161>
+  .byte  196,98,125,24,5,20,65,0,0           // vbroadcastss  0x4114(%rip),%ymm8        # 4550 <_sk_callback_hsw+0x161>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -8758,7 +8774,7 @@ _sk_lighten_hsw:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,175,64,0,0          // vbroadcastss  0x40af(%rip),%ymm8        # 4540 <_sk_callback_hsw+0x165>
+  .byte  196,98,125,24,5,195,64,0,0          // vbroadcastss  0x40c3(%rip),%ymm8        # 4554 <_sk_callback_hsw+0x165>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -8786,7 +8802,7 @@ _sk_difference_hsw:
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,82,64,0,0           // vbroadcastss  0x4052(%rip),%ymm8        # 4544 <_sk_callback_hsw+0x169>
+  .byte  196,98,125,24,5,102,64,0,0          // vbroadcastss  0x4066(%rip),%ymm8        # 4558 <_sk_callback_hsw+0x169>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -8808,7 +8824,7 @@ _sk_exclusion_hsw:
   .byte  197,236,89,214                      // vmulps        %ymm6,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,16,64,0,0           // vbroadcastss  0x4010(%rip),%ymm8        # 4548 <_sk_callback_hsw+0x16d>
+  .byte  196,98,125,24,5,36,64,0,0           // vbroadcastss  0x4024(%rip),%ymm8        # 455c <_sk_callback_hsw+0x16d>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -8818,7 +8834,7 @@ HIDDEN _sk_colorburn_hsw
 .globl _sk_colorburn_hsw
 FUNCTION(_sk_colorburn_hsw)
 _sk_colorburn_hsw:
-  .byte  196,98,125,24,5,254,63,0,0          // vbroadcastss  0x3ffe(%rip),%ymm8        # 454c <_sk_callback_hsw+0x171>
+  .byte  196,98,125,24,5,18,64,0,0           // vbroadcastss  0x4012(%rip),%ymm8        # 4560 <_sk_callback_hsw+0x171>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,216                       // vmulps        %ymm0,%ymm9,%ymm11
   .byte  196,65,44,87,210                    // vxorps        %ymm10,%ymm10,%ymm10
@@ -8876,7 +8892,7 @@ HIDDEN _sk_colordodge_hsw
 FUNCTION(_sk_colordodge_hsw)
 _sk_colordodge_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
-  .byte  196,98,125,24,13,9,63,0,0           // vbroadcastss  0x3f09(%rip),%ymm9        # 4550 <_sk_callback_hsw+0x175>
+  .byte  196,98,125,24,13,29,63,0,0          // vbroadcastss  0x3f1d(%rip),%ymm9        # 4564 <_sk_callback_hsw+0x175>
   .byte  197,52,92,215                       // vsubps        %ymm7,%ymm9,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,52,92,203                       // vsubps        %ymm3,%ymm9,%ymm9
@@ -8929,7 +8945,7 @@ HIDDEN _sk_hardlight_hsw
 .globl _sk_hardlight_hsw
 FUNCTION(_sk_hardlight_hsw)
 _sk_hardlight_hsw:
-  .byte  196,98,125,24,5,42,62,0,0           // vbroadcastss  0x3e2a(%rip),%ymm8        # 4554 <_sk_callback_hsw+0x179>
+  .byte  196,98,125,24,5,62,62,0,0           // vbroadcastss  0x3e3e(%rip),%ymm8        # 4568 <_sk_callback_hsw+0x179>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -8980,7 +8996,7 @@ HIDDEN _sk_overlay_hsw
 .globl _sk_overlay_hsw
 FUNCTION(_sk_overlay_hsw)
 _sk_overlay_hsw:
-  .byte  196,98,125,24,5,98,61,0,0           // vbroadcastss  0x3d62(%rip),%ymm8        # 4558 <_sk_callback_hsw+0x17d>
+  .byte  196,98,125,24,5,118,61,0,0          // vbroadcastss  0x3d76(%rip),%ymm8        # 456c <_sk_callback_hsw+0x17d>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -9041,10 +9057,10 @@ _sk_softlight_hsw:
   .byte  196,65,20,88,197                    // vaddps        %ymm13,%ymm13,%ymm8
   .byte  196,65,60,88,192                    // vaddps        %ymm8,%ymm8,%ymm8
   .byte  196,66,61,168,192                   // vfmadd213ps   %ymm8,%ymm8,%ymm8
-  .byte  196,98,125,24,29,109,60,0,0         // vbroadcastss  0x3c6d(%rip),%ymm11        # 4560 <_sk_callback_hsw+0x185>
+  .byte  196,98,125,24,29,129,60,0,0         // vbroadcastss  0x3c81(%rip),%ymm11        # 4574 <_sk_callback_hsw+0x185>
   .byte  196,65,20,88,227                    // vaddps        %ymm11,%ymm13,%ymm12
   .byte  196,65,28,89,192                    // vmulps        %ymm8,%ymm12,%ymm8
-  .byte  196,98,125,24,37,94,60,0,0          // vbroadcastss  0x3c5e(%rip),%ymm12        # 4564 <_sk_callback_hsw+0x189>
+  .byte  196,98,125,24,37,114,60,0,0         // vbroadcastss  0x3c72(%rip),%ymm12        # 4578 <_sk_callback_hsw+0x189>
   .byte  196,66,21,184,196                   // vfmadd231ps   %ymm12,%ymm13,%ymm8
   .byte  196,65,124,82,245                   // vrsqrtps      %ymm13,%ymm14
   .byte  196,65,124,83,246                   // vrcpps        %ymm14,%ymm14
@@ -9054,7 +9070,7 @@ _sk_softlight_hsw:
   .byte  197,4,194,255,2                     // vcmpleps      %ymm7,%ymm15,%ymm15
   .byte  196,67,13,74,240,240                // vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   .byte  197,116,88,249                      // vaddps        %ymm1,%ymm1,%ymm15
-  .byte  196,98,125,24,5,33,60,0,0           // vbroadcastss  0x3c21(%rip),%ymm8        # 455c <_sk_callback_hsw+0x181>
+  .byte  196,98,125,24,5,53,60,0,0           // vbroadcastss  0x3c35(%rip),%ymm8        # 4570 <_sk_callback_hsw+0x181>
   .byte  196,65,60,92,237                    // vsubps        %ymm13,%ymm8,%ymm13
   .byte  197,132,92,195                      // vsubps        %ymm3,%ymm15,%ymm0
   .byte  196,98,125,168,235                  // vfmadd213ps   %ymm3,%ymm0,%ymm13
@@ -9137,7 +9153,7 @@ FUNCTION(_sk_hue_hsw)
 _sk_hue_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,208,0                // vcmpeqps      %ymm8,%ymm3,%ymm10
-  .byte  196,98,125,24,13,184,58,0,0         // vbroadcastss  0x3ab8(%rip),%ymm9        # 4568 <_sk_callback_hsw+0x18d>
+  .byte  196,98,125,24,13,204,58,0,0         // vbroadcastss  0x3acc(%rip),%ymm9        # 457c <_sk_callback_hsw+0x18d>
   .byte  197,52,94,219                       // vdivps        %ymm3,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,172,89,192                      // vmulps        %ymm0,%ymm10,%ymm0
@@ -9166,11 +9182,11 @@ _sk_hue_hsw:
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  196,193,108,94,212                  // vdivps        %ymm12,%ymm2,%ymm2
   .byte  196,195,109,74,208,208              // vblendvps     %ymm13,%ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,21,45,58,0,0          // vbroadcastss  0x3a2d(%rip),%ymm10        # 456c <_sk_callback_hsw+0x191>
-  .byte  196,98,125,24,29,40,58,0,0          // vbroadcastss  0x3a28(%rip),%ymm11        # 4570 <_sk_callback_hsw+0x195>
+  .byte  196,98,125,24,21,65,58,0,0          // vbroadcastss  0x3a41(%rip),%ymm10        # 4580 <_sk_callback_hsw+0x191>
+  .byte  196,98,125,24,29,60,58,0,0          // vbroadcastss  0x3a3c(%rip),%ymm11        # 4584 <_sk_callback_hsw+0x195>
   .byte  196,65,84,89,227                    // vmulps        %ymm11,%ymm5,%ymm12
   .byte  196,66,93,184,226                   // vfmadd231ps   %ymm10,%ymm4,%ymm12
-  .byte  196,98,125,24,45,25,58,0,0          // vbroadcastss  0x3a19(%rip),%ymm13        # 4574 <_sk_callback_hsw+0x199>
+  .byte  196,98,125,24,45,45,58,0,0          // vbroadcastss  0x3a2d(%rip),%ymm13        # 4588 <_sk_callback_hsw+0x199>
   .byte  196,66,77,184,229                   // vfmadd231ps   %ymm13,%ymm6,%ymm12
   .byte  196,65,116,89,243                   // vmulps        %ymm11,%ymm1,%ymm14
   .byte  196,66,125,184,242                  // vfmadd231ps   %ymm10,%ymm0,%ymm14
@@ -9238,7 +9254,7 @@ FUNCTION(_sk_saturation_hsw)
 _sk_saturation_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,68,194,208,0                 // vcmpeqps      %ymm8,%ymm7,%ymm10
-  .byte  196,98,125,24,13,241,56,0,0         // vbroadcastss  0x38f1(%rip),%ymm9        # 4578 <_sk_callback_hsw+0x19d>
+  .byte  196,98,125,24,13,5,57,0,0           // vbroadcastss  0x3905(%rip),%ymm9        # 458c <_sk_callback_hsw+0x19d>
   .byte  197,52,94,223                       // vdivps        %ymm7,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,44,89,220                       // vmulps        %ymm4,%ymm10,%ymm11
@@ -9267,11 +9283,11 @@ _sk_saturation_hsw:
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  197,252,94,194                      // vdivps        %ymm2,%ymm0,%ymm0
   .byte  196,195,125,74,192,208              // vblendvps     %ymm13,%ymm8,%ymm0,%ymm0
-  .byte  196,226,125,24,21,109,56,0,0        // vbroadcastss  0x386d(%rip),%ymm2        # 457c <_sk_callback_hsw+0x1a1>
-  .byte  196,226,125,24,13,104,56,0,0        // vbroadcastss  0x3868(%rip),%ymm1        # 4580 <_sk_callback_hsw+0x1a5>
+  .byte  196,226,125,24,21,129,56,0,0        // vbroadcastss  0x3881(%rip),%ymm2        # 4590 <_sk_callback_hsw+0x1a1>
+  .byte  196,226,125,24,13,124,56,0,0        // vbroadcastss  0x387c(%rip),%ymm1        # 4594 <_sk_callback_hsw+0x1a5>
   .byte  197,84,89,209                       // vmulps        %ymm1,%ymm5,%ymm10
   .byte  196,98,93,184,210                   // vfmadd231ps   %ymm2,%ymm4,%ymm10
-  .byte  196,98,125,24,45,90,56,0,0          // vbroadcastss  0x385a(%rip),%ymm13        # 4584 <_sk_callback_hsw+0x1a9>
+  .byte  196,98,125,24,45,110,56,0,0         // vbroadcastss  0x386e(%rip),%ymm13        # 4598 <_sk_callback_hsw+0x1a9>
   .byte  196,66,77,184,213                   // vfmadd231ps   %ymm13,%ymm6,%ymm10
   .byte  197,28,89,241                       // vmulps        %ymm1,%ymm12,%ymm14
   .byte  196,98,37,184,242                   // vfmadd231ps   %ymm2,%ymm11,%ymm14
@@ -9339,17 +9355,17 @@ FUNCTION(_sk_color_hsw)
 _sk_color_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,208,0                // vcmpeqps      %ymm8,%ymm3,%ymm10
-  .byte  196,98,125,24,13,44,55,0,0          // vbroadcastss  0x372c(%rip),%ymm9        # 4588 <_sk_callback_hsw+0x1ad>
+  .byte  196,98,125,24,13,64,55,0,0          // vbroadcastss  0x3740(%rip),%ymm9        # 459c <_sk_callback_hsw+0x1ad>
   .byte  197,52,94,219                       // vdivps        %ymm3,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,172,89,192                      // vmulps        %ymm0,%ymm10,%ymm0
   .byte  197,172,89,201                      // vmulps        %ymm1,%ymm10,%ymm1
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
-  .byte  196,98,125,24,21,17,55,0,0          // vbroadcastss  0x3711(%rip),%ymm10        # 458c <_sk_callback_hsw+0x1b1>
-  .byte  196,98,125,24,29,12,55,0,0          // vbroadcastss  0x370c(%rip),%ymm11        # 4590 <_sk_callback_hsw+0x1b5>
+  .byte  196,98,125,24,21,37,55,0,0          // vbroadcastss  0x3725(%rip),%ymm10        # 45a0 <_sk_callback_hsw+0x1b1>
+  .byte  196,98,125,24,29,32,55,0,0          // vbroadcastss  0x3720(%rip),%ymm11        # 45a4 <_sk_callback_hsw+0x1b5>
   .byte  196,65,84,89,227                    // vmulps        %ymm11,%ymm5,%ymm12
   .byte  196,66,93,184,226                   // vfmadd231ps   %ymm10,%ymm4,%ymm12
-  .byte  196,98,125,24,45,253,54,0,0         // vbroadcastss  0x36fd(%rip),%ymm13        # 4594 <_sk_callback_hsw+0x1b9>
+  .byte  196,98,125,24,45,17,55,0,0          // vbroadcastss  0x3711(%rip),%ymm13        # 45a8 <_sk_callback_hsw+0x1b9>
   .byte  196,66,77,184,229                   // vfmadd231ps   %ymm13,%ymm6,%ymm12
   .byte  196,65,116,89,243                   // vmulps        %ymm11,%ymm1,%ymm14
   .byte  196,66,125,184,242                  // vfmadd231ps   %ymm10,%ymm0,%ymm14
@@ -9417,17 +9433,17 @@ FUNCTION(_sk_luminosity_hsw)
 _sk_luminosity_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,68,194,208,0                 // vcmpeqps      %ymm8,%ymm7,%ymm10
-  .byte  196,98,125,24,13,213,53,0,0         // vbroadcastss  0x35d5(%rip),%ymm9        # 4598 <_sk_callback_hsw+0x1bd>
+  .byte  196,98,125,24,13,233,53,0,0         // vbroadcastss  0x35e9(%rip),%ymm9        # 45ac <_sk_callback_hsw+0x1bd>
   .byte  197,52,94,223                       // vdivps        %ymm7,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,44,89,220                       // vmulps        %ymm4,%ymm10,%ymm11
   .byte  197,44,89,229                       // vmulps        %ymm5,%ymm10,%ymm12
   .byte  197,44,89,214                       // vmulps        %ymm6,%ymm10,%ymm10
-  .byte  196,98,125,24,45,186,53,0,0         // vbroadcastss  0x35ba(%rip),%ymm13        # 459c <_sk_callback_hsw+0x1c1>
-  .byte  196,98,125,24,53,181,53,0,0         // vbroadcastss  0x35b5(%rip),%ymm14        # 45a0 <_sk_callback_hsw+0x1c5>
+  .byte  196,98,125,24,45,206,53,0,0         // vbroadcastss  0x35ce(%rip),%ymm13        # 45b0 <_sk_callback_hsw+0x1c1>
+  .byte  196,98,125,24,53,201,53,0,0         // vbroadcastss  0x35c9(%rip),%ymm14        # 45b4 <_sk_callback_hsw+0x1c5>
   .byte  196,193,116,89,206                  // vmulps        %ymm14,%ymm1,%ymm1
   .byte  196,226,21,168,193                  // vfmadd213ps   %ymm1,%ymm13,%ymm0
-  .byte  196,98,125,24,61,166,53,0,0         // vbroadcastss  0x35a6(%rip),%ymm15        # 45a4 <_sk_callback_hsw+0x1c9>
+  .byte  196,98,125,24,61,186,53,0,0         // vbroadcastss  0x35ba(%rip),%ymm15        # 45b8 <_sk_callback_hsw+0x1c9>
   .byte  196,226,5,168,208                   // vfmadd213ps   %ymm0,%ymm15,%ymm2
   .byte  196,193,28,89,198                   // vmulps        %ymm14,%ymm12,%ymm0
   .byte  196,194,37,184,197                  // vfmadd231ps   %ymm13,%ymm11,%ymm0
@@ -9505,7 +9521,7 @@ HIDDEN _sk_clamp_1_hsw
 .globl _sk_clamp_1_hsw
 FUNCTION(_sk_clamp_1_hsw)
 _sk_clamp_1_hsw:
-  .byte  196,98,125,24,5,104,52,0,0          // vbroadcastss  0x3468(%rip),%ymm8        # 45a8 <_sk_callback_hsw+0x1cd>
+  .byte  196,98,125,24,5,124,52,0,0          // vbroadcastss  0x347c(%rip),%ymm8        # 45bc <_sk_callback_hsw+0x1cd>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
@@ -9517,7 +9533,7 @@ HIDDEN _sk_clamp_a_hsw
 .globl _sk_clamp_a_hsw
 FUNCTION(_sk_clamp_a_hsw)
 _sk_clamp_a_hsw:
-  .byte  196,98,125,24,5,75,52,0,0           // vbroadcastss  0x344b(%rip),%ymm8        # 45ac <_sk_callback_hsw+0x1d1>
+  .byte  196,98,125,24,5,95,52,0,0           // vbroadcastss  0x345f(%rip),%ymm8        # 45c0 <_sk_callback_hsw+0x1d1>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  197,252,93,195                      // vminps        %ymm3,%ymm0,%ymm0
   .byte  197,244,93,203                      // vminps        %ymm3,%ymm1,%ymm1
@@ -9603,7 +9619,7 @@ FUNCTION(_sk_unpremul_hsw)
 _sk_unpremul_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,200,0                // vcmpeqps      %ymm8,%ymm3,%ymm9
-  .byte  196,98,125,24,21,147,51,0,0         // vbroadcastss  0x3393(%rip),%ymm10        # 45b0 <_sk_callback_hsw+0x1d5>
+  .byte  196,98,125,24,21,167,51,0,0         // vbroadcastss  0x33a7(%rip),%ymm10        # 45c4 <_sk_callback_hsw+0x1d5>
   .byte  197,44,94,211                       // vdivps        %ymm3,%ymm10,%ymm10
   .byte  196,67,45,74,192,144                // vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
@@ -9616,16 +9632,16 @@ HIDDEN _sk_from_srgb_hsw
 .globl _sk_from_srgb_hsw
 FUNCTION(_sk_from_srgb_hsw)
 _sk_from_srgb_hsw:
-  .byte  196,98,125,24,5,116,51,0,0          // vbroadcastss  0x3374(%rip),%ymm8        # 45b4 <_sk_callback_hsw+0x1d9>
+  .byte  196,98,125,24,5,136,51,0,0          // vbroadcastss  0x3388(%rip),%ymm8        # 45c8 <_sk_callback_hsw+0x1d9>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  197,124,89,208                      // vmulps        %ymm0,%ymm0,%ymm10
-  .byte  196,98,125,24,29,102,51,0,0         // vbroadcastss  0x3366(%rip),%ymm11        # 45b8 <_sk_callback_hsw+0x1dd>
-  .byte  196,98,125,24,37,97,51,0,0          // vbroadcastss  0x3361(%rip),%ymm12        # 45bc <_sk_callback_hsw+0x1e1>
+  .byte  196,98,125,24,29,122,51,0,0         // vbroadcastss  0x337a(%rip),%ymm11        # 45cc <_sk_callback_hsw+0x1dd>
+  .byte  196,98,125,24,37,117,51,0,0         // vbroadcastss  0x3375(%rip),%ymm12        # 45d0 <_sk_callback_hsw+0x1e1>
   .byte  196,65,124,40,236                   // vmovaps       %ymm12,%ymm13
   .byte  196,66,125,168,235                  // vfmadd213ps   %ymm11,%ymm0,%ymm13
-  .byte  196,98,125,24,53,82,51,0,0          // vbroadcastss  0x3352(%rip),%ymm14        # 45c0 <_sk_callback_hsw+0x1e5>
+  .byte  196,98,125,24,53,102,51,0,0         // vbroadcastss  0x3366(%rip),%ymm14        # 45d4 <_sk_callback_hsw+0x1e5>
   .byte  196,66,45,168,238                   // vfmadd213ps   %ymm14,%ymm10,%ymm13
-  .byte  196,98,125,24,21,72,51,0,0          // vbroadcastss  0x3348(%rip),%ymm10        # 45c4 <_sk_callback_hsw+0x1e9>
+  .byte  196,98,125,24,21,92,51,0,0          // vbroadcastss  0x335c(%rip),%ymm10        # 45d8 <_sk_callback_hsw+0x1e9>
   .byte  196,193,124,194,194,1               // vcmpltps      %ymm10,%ymm0,%ymm0
   .byte  196,195,21,74,193,0                 // vblendvps     %ymm0,%ymm9,%ymm13,%ymm0
   .byte  196,65,116,89,200                   // vmulps        %ymm8,%ymm1,%ymm9
@@ -9651,16 +9667,16 @@ _sk_to_srgb_hsw:
   .byte  197,124,82,192                      // vrsqrtps      %ymm0,%ymm8
   .byte  196,65,124,83,200                   // vrcpps        %ymm8,%ymm9
   .byte  196,65,124,82,208                   // vrsqrtps      %ymm8,%ymm10
-  .byte  196,98,125,24,5,226,50,0,0          // vbroadcastss  0x32e2(%rip),%ymm8        # 45c8 <_sk_callback_hsw+0x1ed>
+  .byte  196,98,125,24,5,246,50,0,0          // vbroadcastss  0x32f6(%rip),%ymm8        # 45dc <_sk_callback_hsw+0x1ed>
   .byte  196,65,124,89,216                   // vmulps        %ymm8,%ymm0,%ymm11
-  .byte  196,98,125,24,37,216,50,0,0         // vbroadcastss  0x32d8(%rip),%ymm12        # 45cc <_sk_callback_hsw+0x1f1>
-  .byte  196,98,125,24,45,211,50,0,0         // vbroadcastss  0x32d3(%rip),%ymm13        # 45d0 <_sk_callback_hsw+0x1f5>
+  .byte  196,98,125,24,37,236,50,0,0         // vbroadcastss  0x32ec(%rip),%ymm12        # 45e0 <_sk_callback_hsw+0x1f1>
+  .byte  196,98,125,24,45,231,50,0,0         // vbroadcastss  0x32e7(%rip),%ymm13        # 45e4 <_sk_callback_hsw+0x1f5>
   .byte  196,66,21,168,204                   // vfmadd213ps   %ymm12,%ymm13,%ymm9
-  .byte  196,98,125,24,53,201,50,0,0         // vbroadcastss  0x32c9(%rip),%ymm14        # 45d4 <_sk_callback_hsw+0x1f9>
+  .byte  196,98,125,24,53,221,50,0,0         // vbroadcastss  0x32dd(%rip),%ymm14        # 45e8 <_sk_callback_hsw+0x1f9>
   .byte  196,66,13,184,202                   // vfmadd231ps   %ymm10,%ymm14,%ymm9
-  .byte  196,98,125,24,21,191,50,0,0         // vbroadcastss  0x32bf(%rip),%ymm10        # 45d8 <_sk_callback_hsw+0x1fd>
+  .byte  196,98,125,24,21,211,50,0,0         // vbroadcastss  0x32d3(%rip),%ymm10        # 45ec <_sk_callback_hsw+0x1fd>
   .byte  196,65,44,93,201                    // vminps        %ymm9,%ymm10,%ymm9
-  .byte  196,98,125,24,61,181,50,0,0         // vbroadcastss  0x32b5(%rip),%ymm15        # 45dc <_sk_callback_hsw+0x201>
+  .byte  196,98,125,24,61,201,50,0,0         // vbroadcastss  0x32c9(%rip),%ymm15        # 45f0 <_sk_callback_hsw+0x201>
   .byte  196,193,124,194,199,1               // vcmpltps      %ymm15,%ymm0,%ymm0
   .byte  196,195,53,74,195,0                 // vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   .byte  197,124,82,201                      // vrsqrtps      %ymm1,%ymm9
@@ -9693,26 +9709,26 @@ _sk_rgb_to_hsl_hsw:
   .byte  197,124,93,201                      // vminps        %ymm1,%ymm0,%ymm9
   .byte  197,52,93,202                       // vminps        %ymm2,%ymm9,%ymm9
   .byte  196,65,60,92,209                    // vsubps        %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,29,47,50,0,0          // vbroadcastss  0x322f(%rip),%ymm11        # 45e0 <_sk_callback_hsw+0x205>
+  .byte  196,98,125,24,29,67,50,0,0          // vbroadcastss  0x3243(%rip),%ymm11        # 45f4 <_sk_callback_hsw+0x205>
   .byte  196,65,36,94,218                    // vdivps        %ymm10,%ymm11,%ymm11
   .byte  197,116,92,226                      // vsubps        %ymm2,%ymm1,%ymm12
   .byte  197,116,194,234,1                   // vcmpltps      %ymm2,%ymm1,%ymm13
-  .byte  196,98,125,24,53,28,50,0,0          // vbroadcastss  0x321c(%rip),%ymm14        # 45e4 <_sk_callback_hsw+0x209>
+  .byte  196,98,125,24,53,48,50,0,0          // vbroadcastss  0x3230(%rip),%ymm14        # 45f8 <_sk_callback_hsw+0x209>
   .byte  196,65,4,87,255                     // vxorps        %ymm15,%ymm15,%ymm15
   .byte  196,67,5,74,238,208                 // vblendvps     %ymm13,%ymm14,%ymm15,%ymm13
   .byte  196,66,37,168,229                   // vfmadd213ps   %ymm13,%ymm11,%ymm12
   .byte  197,236,92,208                      // vsubps        %ymm0,%ymm2,%ymm2
   .byte  197,124,92,233                      // vsubps        %ymm1,%ymm0,%ymm13
-  .byte  196,98,125,24,53,3,50,0,0           // vbroadcastss  0x3203(%rip),%ymm14        # 45ec <_sk_callback_hsw+0x211>
+  .byte  196,98,125,24,53,23,50,0,0          // vbroadcastss  0x3217(%rip),%ymm14        # 4600 <_sk_callback_hsw+0x211>
   .byte  196,66,37,168,238                   // vfmadd213ps   %ymm14,%ymm11,%ymm13
-  .byte  196,98,125,24,53,241,49,0,0         // vbroadcastss  0x31f1(%rip),%ymm14        # 45e8 <_sk_callback_hsw+0x20d>
+  .byte  196,98,125,24,53,5,50,0,0           // vbroadcastss  0x3205(%rip),%ymm14        # 45fc <_sk_callback_hsw+0x20d>
   .byte  196,194,37,168,214                  // vfmadd213ps   %ymm14,%ymm11,%ymm2
   .byte  197,188,194,201,0                   // vcmpeqps      %ymm1,%ymm8,%ymm1
   .byte  196,227,21,74,202,16                // vblendvps     %ymm1,%ymm2,%ymm13,%ymm1
   .byte  197,188,194,192,0                   // vcmpeqps      %ymm0,%ymm8,%ymm0
   .byte  196,195,117,74,196,0                // vblendvps     %ymm0,%ymm12,%ymm1,%ymm0
   .byte  196,193,60,88,201                   // vaddps        %ymm9,%ymm8,%ymm1
-  .byte  196,98,125,24,29,212,49,0,0         // vbroadcastss  0x31d4(%rip),%ymm11        # 45f4 <_sk_callback_hsw+0x219>
+  .byte  196,98,125,24,29,232,49,0,0         // vbroadcastss  0x31e8(%rip),%ymm11        # 4608 <_sk_callback_hsw+0x219>
   .byte  196,193,116,89,211                  // vmulps        %ymm11,%ymm1,%ymm2
   .byte  197,36,194,218,1                    // vcmpltps      %ymm2,%ymm11,%ymm11
   .byte  196,65,12,92,224                    // vsubps        %ymm8,%ymm14,%ymm12
@@ -9722,7 +9738,7 @@ _sk_rgb_to_hsl_hsw:
   .byte  197,172,94,201                      // vdivps        %ymm1,%ymm10,%ymm1
   .byte  196,195,125,74,199,128              // vblendvps     %ymm8,%ymm15,%ymm0,%ymm0
   .byte  196,195,117,74,207,128              // vblendvps     %ymm8,%ymm15,%ymm1,%ymm1
-  .byte  196,98,125,24,5,151,49,0,0          // vbroadcastss  0x3197(%rip),%ymm8        # 45f0 <_sk_callback_hsw+0x215>
+  .byte  196,98,125,24,5,171,49,0,0          // vbroadcastss  0x31ab(%rip),%ymm8        # 4604 <_sk_callback_hsw+0x215>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -9739,30 +9755,30 @@ _sk_hsl_to_rgb_hsw:
   .byte  197,252,17,92,36,128                // vmovups       %ymm3,-0x80(%rsp)
   .byte  197,252,40,233                      // vmovaps       %ymm1,%ymm5
   .byte  197,252,40,224                      // vmovaps       %ymm0,%ymm4
-  .byte  196,98,125,24,5,100,49,0,0          // vbroadcastss  0x3164(%rip),%ymm8        # 45f8 <_sk_callback_hsw+0x21d>
+  .byte  196,98,125,24,5,120,49,0,0          // vbroadcastss  0x3178(%rip),%ymm8        # 460c <_sk_callback_hsw+0x21d>
   .byte  197,60,194,202,2                    // vcmpleps      %ymm2,%ymm8,%ymm9
   .byte  197,84,89,210                       // vmulps        %ymm2,%ymm5,%ymm10
   .byte  196,65,84,92,218                    // vsubps        %ymm10,%ymm5,%ymm11
   .byte  196,67,45,74,203,144                // vblendvps     %ymm9,%ymm11,%ymm10,%ymm9
   .byte  197,52,88,210                       // vaddps        %ymm2,%ymm9,%ymm10
-  .byte  196,98,125,24,13,71,49,0,0          // vbroadcastss  0x3147(%rip),%ymm9        # 45fc <_sk_callback_hsw+0x221>
+  .byte  196,98,125,24,13,91,49,0,0          // vbroadcastss  0x315b(%rip),%ymm9        # 4610 <_sk_callback_hsw+0x221>
   .byte  196,66,109,170,202                  // vfmsub213ps   %ymm10,%ymm2,%ymm9
-  .byte  196,98,125,24,29,61,49,0,0          // vbroadcastss  0x313d(%rip),%ymm11        # 4600 <_sk_callback_hsw+0x225>
+  .byte  196,98,125,24,29,81,49,0,0          // vbroadcastss  0x3151(%rip),%ymm11        # 4614 <_sk_callback_hsw+0x225>
   .byte  196,65,92,88,219                    // vaddps        %ymm11,%ymm4,%ymm11
   .byte  196,67,125,8,227,1                  // vroundps      $0x1,%ymm11,%ymm12
   .byte  196,65,36,92,252                    // vsubps        %ymm12,%ymm11,%ymm15
   .byte  196,65,44,92,217                    // vsubps        %ymm9,%ymm10,%ymm11
-  .byte  196,98,125,24,45,39,49,0,0          // vbroadcastss  0x3127(%rip),%ymm13        # 4608 <_sk_callback_hsw+0x22d>
+  .byte  196,98,125,24,45,59,49,0,0          // vbroadcastss  0x313b(%rip),%ymm13        # 461c <_sk_callback_hsw+0x22d>
   .byte  196,193,4,89,197                    // vmulps        %ymm13,%ymm15,%ymm0
-  .byte  196,98,125,24,53,29,49,0,0          // vbroadcastss  0x311d(%rip),%ymm14        # 460c <_sk_callback_hsw+0x231>
+  .byte  196,98,125,24,53,49,49,0,0          // vbroadcastss  0x3131(%rip),%ymm14        # 4620 <_sk_callback_hsw+0x231>
   .byte  197,12,92,224                       // vsubps        %ymm0,%ymm14,%ymm12
   .byte  196,66,37,168,225                   // vfmadd213ps   %ymm9,%ymm11,%ymm12
-  .byte  196,226,125,24,29,3,49,0,0          // vbroadcastss  0x3103(%rip),%ymm3        # 4604 <_sk_callback_hsw+0x229>
+  .byte  196,226,125,24,29,23,49,0,0         // vbroadcastss  0x3117(%rip),%ymm3        # 4618 <_sk_callback_hsw+0x229>
   .byte  196,193,100,194,255,2               // vcmpleps      %ymm15,%ymm3,%ymm7
   .byte  196,195,29,74,249,112               // vblendvps     %ymm7,%ymm9,%ymm12,%ymm7
   .byte  196,65,60,194,231,2                 // vcmpleps      %ymm15,%ymm8,%ymm12
   .byte  196,227,45,74,255,192               // vblendvps     %ymm12,%ymm7,%ymm10,%ymm7
-  .byte  196,98,125,24,37,238,48,0,0         // vbroadcastss  0x30ee(%rip),%ymm12        # 4610 <_sk_callback_hsw+0x235>
+  .byte  196,98,125,24,37,2,49,0,0           // vbroadcastss  0x3102(%rip),%ymm12        # 4624 <_sk_callback_hsw+0x235>
   .byte  196,65,28,194,255,2                 // vcmpleps      %ymm15,%ymm12,%ymm15
   .byte  196,194,37,168,193                  // vfmadd213ps   %ymm9,%ymm11,%ymm0
   .byte  196,99,125,74,255,240               // vblendvps     %ymm15,%ymm7,%ymm0,%ymm15
@@ -9778,7 +9794,7 @@ _sk_hsl_to_rgb_hsw:
   .byte  197,156,194,192,2                   // vcmpleps      %ymm0,%ymm12,%ymm0
   .byte  196,194,37,168,249                  // vfmadd213ps   %ymm9,%ymm11,%ymm7
   .byte  196,227,69,74,201,0                 // vblendvps     %ymm0,%ymm1,%ymm7,%ymm1
-  .byte  196,226,125,24,5,154,48,0,0         // vbroadcastss  0x309a(%rip),%ymm0        # 4614 <_sk_callback_hsw+0x239>
+  .byte  196,226,125,24,5,174,48,0,0         // vbroadcastss  0x30ae(%rip),%ymm0        # 4628 <_sk_callback_hsw+0x239>
   .byte  197,220,88,192                      // vaddps        %ymm0,%ymm4,%ymm0
   .byte  196,227,125,8,224,1                 // vroundps      $0x1,%ymm0,%ymm4
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
@@ -9832,7 +9848,7 @@ _sk_scale_u8_hsw:
   .byte  197,122,126,0                       // vmovq         (%rax),%xmm8
   .byte  196,66,125,49,192                   // vpmovzxbd     %xmm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,218,47,0,0         // vbroadcastss  0x2fda(%rip),%ymm9        # 4618 <_sk_callback_hsw+0x23d>
+  .byte  196,98,125,24,13,238,47,0,0         // vbroadcastss  0x2fee(%rip),%ymm9        # 462c <_sk_callback_hsw+0x23d>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -9884,7 +9900,7 @@ _sk_lerp_u8_hsw:
   .byte  197,122,126,0                       // vmovq         (%rax),%xmm8
   .byte  196,66,125,49,192                   // vpmovzxbd     %xmm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,71,47,0,0          // vbroadcastss  0x2f47(%rip),%ymm9        # 461c <_sk_callback_hsw+0x241>
+  .byte  196,98,125,24,13,91,47,0,0          // vbroadcastss  0x2f5b(%rip),%ymm9        # 4630 <_sk_callback_hsw+0x241>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
   .byte  196,226,61,168,196                  // vfmadd213ps   %ymm4,%ymm8,%ymm0
@@ -9917,72 +9933,78 @@ _sk_lerp_565_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,149,0,0,0                    // jne           17cd <_sk_lerp_565_hsw+0xa3>
-  .byte  196,193,122,111,28,122              // vmovdqu       (%r10,%rdi,2),%xmm3
-  .byte  196,226,125,51,219                  // vpmovzxwd     %xmm3,%ymm3
-  .byte  196,98,125,88,5,212,46,0,0          // vpbroadcastd  0x2ed4(%rip),%ymm8        # 4620 <_sk_callback_hsw+0x245>
-  .byte  196,65,101,219,192                  // vpand         %ymm8,%ymm3,%ymm8
-  .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,197,46,0,0         // vbroadcastss  0x2ec5(%rip),%ymm9        # 4624 <_sk_callback_hsw+0x249>
-  .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,88,13,187,46,0,0         // vpbroadcastd  0x2ebb(%rip),%ymm9        # 4628 <_sk_callback_hsw+0x24d>
-  .byte  196,65,101,219,201                  // vpand         %ymm9,%ymm3,%ymm9
+  .byte  15,133,169,0,0,0                    // jne           17e1 <_sk_lerp_565_hsw+0xb7>
+  .byte  196,65,122,111,4,122                // vmovdqu       (%r10,%rdi,2),%xmm8
+  .byte  196,66,125,51,192                   // vpmovzxwd     %xmm8,%ymm8
+  .byte  196,98,125,88,13,232,46,0,0         // vpbroadcastd  0x2ee8(%rip),%ymm9        # 4634 <_sk_callback_hsw+0x245>
+  .byte  196,65,61,219,201                   // vpand         %ymm9,%ymm8,%ymm9
   .byte  196,65,124,91,201                   // vcvtdq2ps     %ymm9,%ymm9
-  .byte  196,98,125,24,21,172,46,0,0         // vbroadcastss  0x2eac(%rip),%ymm10        # 462c <_sk_callback_hsw+0x251>
+  .byte  196,98,125,24,21,217,46,0,0         // vbroadcastss  0x2ed9(%rip),%ymm10        # 4638 <_sk_callback_hsw+0x249>
   .byte  196,65,52,89,202                    // vmulps        %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,88,21,162,46,0,0         // vpbroadcastd  0x2ea2(%rip),%ymm10        # 4630 <_sk_callback_hsw+0x255>
-  .byte  196,193,101,219,218                 // vpand         %ymm10,%ymm3,%ymm3
-  .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,21,148,46,0,0         // vbroadcastss  0x2e94(%rip),%ymm10        # 4634 <_sk_callback_hsw+0x259>
-  .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
+  .byte  196,98,125,88,21,207,46,0,0         // vpbroadcastd  0x2ecf(%rip),%ymm10        # 463c <_sk_callback_hsw+0x24d>
+  .byte  196,65,61,219,210                   // vpand         %ymm10,%ymm8,%ymm10
+  .byte  196,65,124,91,210                   // vcvtdq2ps     %ymm10,%ymm10
+  .byte  196,98,125,24,29,192,46,0,0         // vbroadcastss  0x2ec0(%rip),%ymm11        # 4640 <_sk_callback_hsw+0x251>
+  .byte  196,65,44,89,211                    // vmulps        %ymm11,%ymm10,%ymm10
+  .byte  196,98,125,88,29,182,46,0,0         // vpbroadcastd  0x2eb6(%rip),%ymm11        # 4644 <_sk_callback_hsw+0x255>
+  .byte  196,65,61,219,195                   // vpand         %ymm11,%ymm8,%ymm8
+  .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
+  .byte  196,98,125,24,29,167,46,0,0         // vbroadcastss  0x2ea7(%rip),%ymm11        # 4648 <_sk_callback_hsw+0x259>
+  .byte  196,65,60,89,195                    // vmulps        %ymm11,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
-  .byte  196,226,61,168,196                  // vfmadd213ps   %ymm4,%ymm8,%ymm0
+  .byte  196,226,53,168,196                  // vfmadd213ps   %ymm4,%ymm9,%ymm0
   .byte  197,244,92,205                      // vsubps        %ymm5,%ymm1,%ymm1
-  .byte  196,226,53,168,205                  // vfmadd213ps   %ymm5,%ymm9,%ymm1
+  .byte  196,226,45,168,205                  // vfmadd213ps   %ymm5,%ymm10,%ymm1
   .byte  197,236,92,214                      // vsubps        %ymm6,%ymm2,%ymm2
-  .byte  196,226,101,168,214                 // vfmadd213ps   %ymm6,%ymm3,%ymm2
+  .byte  196,226,61,168,214                  // vfmadd213ps   %ymm6,%ymm8,%ymm2
+  .byte  197,228,92,223                      // vsubps        %ymm7,%ymm3,%ymm3
+  .byte  196,98,101,168,207                  // vfmadd213ps   %ymm7,%ymm3,%ymm9
+  .byte  196,98,101,168,215                  // vfmadd213ps   %ymm7,%ymm3,%ymm10
+  .byte  196,98,101,168,199                  // vfmadd213ps   %ymm7,%ymm3,%ymm8
+  .byte  196,193,44,95,216                   // vmaxps        %ymm8,%ymm10,%ymm3
+  .byte  197,180,95,219                      // vmaxps        %ymm3,%ymm9,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,109,46,0,0        // vbroadcastss  0x2e6d(%rip),%ymm3        # 4638 <_sk_callback_hsw+0x25d>
   .byte  255,224                             // jmpq          *%rax
   .byte  65,137,200                          // mov           %ecx,%r8d
   .byte  65,128,224,7                        // and           $0x7,%r8b
-  .byte  197,225,239,219                     // vpxor         %xmm3,%xmm3,%xmm3
+  .byte  196,65,57,239,192                   // vpxor         %xmm8,%xmm8,%xmm8
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,89,255,255,255               // ja            173e <_sk_lerp_565_hsw+0x14>
+  .byte  15,135,68,255,255,255               // ja            173e <_sk_lerp_565_hsw+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,76,0,0,0                  // lea           0x4c(%rip),%r9        # 183c <_sk_lerp_565_hsw+0x112>
+  .byte  76,141,13,75,0,0,0                  // lea           0x4b(%rip),%r9        # 1850 <_sk_lerp_565_hsw+0x126>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
-  .byte  197,225,239,219                     // vpxor         %xmm3,%xmm3,%xmm3
-  .byte  196,193,97,196,92,122,12,6          // vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  196,193,97,196,92,122,10,5          // vpinsrw       $0x5,0xa(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  196,193,97,196,92,122,8,4           // vpinsrw       $0x4,0x8(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  196,193,97,196,92,122,6,3           // vpinsrw       $0x3,0x6(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  196,193,97,196,92,122,4,2           // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  196,193,97,196,92,122,2,1           // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  196,193,97,196,28,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm3,%xmm3
-  .byte  233,5,255,255,255                   // jmpq          173e <_sk_lerp_565_hsw+0x14>
-  .byte  15,31,0                             // nopl          (%rax)
-  .byte  241                                 // icebp
+  .byte  196,65,57,239,192                   // vpxor         %xmm8,%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,12,6           // vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,10,5           // vpinsrw       $0x5,0xa(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,8,4            // vpinsrw       $0x4,0x8(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,6,3            // vpinsrw       $0x3,0x6(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,4,2            // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,2,1            // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  196,65,57,196,4,122,0               // vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
+  .byte  233,239,254,255,255                 // jmpq          173e <_sk_lerp_565_hsw+0x14>
+  .byte  144                                 // nop
+  .byte  243,255                             // repz          (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
+  .byte  235,255                             // jmp           1855 <_sk_lerp_565_hsw+0x12b>
   .byte  255                                 // (bad)
-  .byte  233,255,255,255,225                 // jmpq          ffffffffe2001844 <_sk_callback_hsw+0xffffffffe1ffd469>
+  .byte  255,227                             // jmpq          *%rbx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  217,255                             // fcos
+  .byte  219,255                             // (bad)
   .byte  255                                 // (bad)
-  .byte  255,209                             // callq         *%rcx
+  .byte  255,211                             // callq         *%rbx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,201                             // dec           %ecx
+  .byte  255,203                             // dec           %ebx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  189                                 // .byte         0xbd
+  .byte  190                                 // .byte         0xbe
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
@@ -9996,23 +10018,23 @@ _sk_load_tables_hsw:
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  76,3,8                              // add           (%rax),%r9
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,105                             // jne           18d6 <_sk_load_tables_hsw+0x7e>
+  .byte  117,105                             // jne           18ea <_sk_load_tables_hsw+0x7e>
   .byte  196,193,126,111,25                  // vmovdqu       (%r9),%ymm3
-  .byte  197,229,219,13,102,48,0,0           // vpand         0x3066(%rip),%ymm3,%ymm1        # 48e0 <_sk_callback_hsw+0x505>
+  .byte  197,229,219,13,114,48,0,0           // vpand         0x3072(%rip),%ymm3,%ymm1        # 4900 <_sk_callback_hsw+0x511>
   .byte  196,65,61,118,192                   // vpcmpeqd      %ymm8,%ymm8,%ymm8
   .byte  72,139,72,8                         // mov           0x8(%rax),%rcx
   .byte  76,139,72,16                        // mov           0x10(%rax),%r9
   .byte  197,237,118,210                     // vpcmpeqd      %ymm2,%ymm2,%ymm2
   .byte  196,226,109,146,4,137               // vgatherdps    %ymm2,(%rcx,%ymm1,4),%ymm0
-  .byte  196,226,101,0,21,102,48,0,0         // vpshufb       0x3066(%rip),%ymm3,%ymm2        # 4900 <_sk_callback_hsw+0x525>
+  .byte  196,226,101,0,21,114,48,0,0         // vpshufb       0x3072(%rip),%ymm3,%ymm2        # 4920 <_sk_callback_hsw+0x531>
   .byte  196,65,53,118,201                   // vpcmpeqd      %ymm9,%ymm9,%ymm9
   .byte  196,194,53,146,12,145               // vgatherdps    %ymm9,(%r9,%ymm2,4),%ymm1
   .byte  72,139,64,24                        // mov           0x18(%rax),%rax
-  .byte  196,98,101,0,13,110,48,0,0          // vpshufb       0x306e(%rip),%ymm3,%ymm9        # 4920 <_sk_callback_hsw+0x545>
+  .byte  196,98,101,0,13,122,48,0,0          // vpshufb       0x307a(%rip),%ymm3,%ymm9        # 4940 <_sk_callback_hsw+0x551>
   .byte  196,162,61,146,20,136               // vgatherdps    %ymm8,(%rax,%ymm9,4),%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,114,45,0,0          // vbroadcastss  0x2d72(%rip),%ymm8        # 463c <_sk_callback_hsw+0x261>
+  .byte  196,98,125,24,5,110,45,0,0          // vbroadcastss  0x2d6e(%rip),%ymm8        # 464c <_sk_callback_hsw+0x25d>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,137,193                          // mov           %r8,%rcx
@@ -10025,7 +10047,7 @@ _sk_load_tables_hsw:
   .byte  196,193,249,110,194                 // vmovq         %r10,%xmm0
   .byte  196,226,125,33,192                  // vpmovsxbd     %xmm0,%ymm0
   .byte  196,194,125,140,25                  // vpmaskmovd    (%r9),%ymm0,%ymm3
-  .byte  233,115,255,255,255                 // jmpq          1872 <_sk_load_tables_hsw+0x1a>
+  .byte  233,115,255,255,255                 // jmpq          1886 <_sk_load_tables_hsw+0x1a>
 
 HIDDEN _sk_load_tables_u16_be_hsw
 .globl _sk_load_tables_u16_be_hsw
@@ -10035,7 +10057,7 @@ _sk_load_tables_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,201,0,0,0                    // jne           19de <_sk_load_tables_u16_be_hsw+0xdf>
+  .byte  15,133,201,0,0,0                    // jne           19f2 <_sk_load_tables_u16_be_hsw+0xdf>
   .byte  196,1,121,16,4,72                   // vmovupd       (%r8,%r9,2),%xmm8
   .byte  196,129,121,16,84,72,16             // vmovupd       0x10(%r8,%r9,2),%xmm2
   .byte  196,129,121,16,92,72,32             // vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -10051,7 +10073,7 @@ _sk_load_tables_u16_be_hsw:
   .byte  197,185,108,200                     // vpunpcklqdq   %xmm0,%xmm8,%xmm1
   .byte  197,185,109,208                     // vpunpckhqdq   %xmm0,%xmm8,%xmm2
   .byte  197,49,108,195                      // vpunpcklqdq   %xmm3,%xmm9,%xmm8
-  .byte  197,121,111,21,250,48,0,0           // vmovdqa       0x30fa(%rip),%xmm10        # 4a60 <_sk_callback_hsw+0x685>
+  .byte  197,121,111,21,6,49,0,0             // vmovdqa       0x3106(%rip),%xmm10        # 4a80 <_sk_callback_hsw+0x691>
   .byte  196,193,113,219,194                 // vpand         %xmm10,%xmm1,%xmm0
   .byte  196,226,125,51,200                  // vpmovzxwd     %xmm0,%ymm1
   .byte  196,65,37,118,219                   // vpcmpeqd      %ymm11,%ymm11,%ymm11
@@ -10073,36 +10095,36 @@ _sk_load_tables_u16_be_hsw:
   .byte  197,185,235,219                     // vpor          %xmm3,%xmm8,%xmm3
   .byte  196,226,125,51,219                  // vpmovzxwd     %xmm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,107,44,0,0          // vbroadcastss  0x2c6b(%rip),%ymm8        # 4640 <_sk_callback_hsw+0x265>
+  .byte  196,98,125,24,5,103,44,0,0          // vbroadcastss  0x2c67(%rip),%ymm8        # 4650 <_sk_callback_hsw+0x261>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
   .byte  196,1,123,16,4,72                   // vmovsd        (%r8,%r9,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            1a44 <_sk_load_tables_u16_be_hsw+0x145>
+  .byte  116,85                              // je            1a58 <_sk_load_tables_u16_be_hsw+0x145>
   .byte  196,1,57,22,68,72,8                 // vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            1a44 <_sk_load_tables_u16_be_hsw+0x145>
+  .byte  114,72                              // jb            1a58 <_sk_load_tables_u16_be_hsw+0x145>
   .byte  196,129,123,16,84,72,16             // vmovsd        0x10(%r8,%r9,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            1a51 <_sk_load_tables_u16_be_hsw+0x152>
+  .byte  116,72                              // je            1a65 <_sk_load_tables_u16_be_hsw+0x152>
   .byte  196,129,105,22,84,72,24             // vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            1a51 <_sk_load_tables_u16_be_hsw+0x152>
+  .byte  114,59                              // jb            1a65 <_sk_load_tables_u16_be_hsw+0x152>
   .byte  196,129,123,16,92,72,32             // vmovsd        0x20(%r8,%r9,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,9,255,255,255                // je            1930 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  15,132,9,255,255,255                // je            1944 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  196,129,97,22,92,72,40              // vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,248,254,255,255              // jb            1930 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  15,130,248,254,255,255              // jb            1944 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  196,1,122,126,76,72,48              // vmovq         0x30(%r8,%r9,2),%xmm9
-  .byte  233,236,254,255,255                 // jmpq          1930 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  233,236,254,255,255                 // jmpq          1944 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,223,254,255,255                 // jmpq          1930 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  233,223,254,255,255                 // jmpq          1944 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,214,254,255,255                 // jmpq          1930 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  233,214,254,255,255                 // jmpq          1944 <_sk_load_tables_u16_be_hsw+0x31>
 
 HIDDEN _sk_load_tables_rgb_u16_be_hsw
 .globl _sk_load_tables_rgb_u16_be_hsw
@@ -10112,7 +10134,7 @@ _sk_load_tables_rgb_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,127                       // lea           (%rdi,%rdi,2),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,193,0,0,0                    // jne           1b2d <_sk_load_tables_rgb_u16_be_hsw+0xd3>
+  .byte  15,133,193,0,0,0                    // jne           1b41 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
   .byte  196,129,122,111,4,72                // vmovdqu       (%r8,%r9,2),%xmm0
   .byte  196,129,122,111,84,72,12            // vmovdqu       0xc(%r8,%r9,2),%xmm2
   .byte  196,129,122,111,76,72,24            // vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -10133,7 +10155,7 @@ _sk_load_tables_rgb_u16_be_hsw:
   .byte  197,185,108,218                     // vpunpcklqdq   %xmm2,%xmm8,%xmm3
   .byte  197,185,109,210                     // vpunpckhqdq   %xmm2,%xmm8,%xmm2
   .byte  197,121,108,193                     // vpunpcklqdq   %xmm1,%xmm0,%xmm8
-  .byte  197,121,111,13,154,47,0,0           // vmovdqa       0x2f9a(%rip),%xmm9        # 4a70 <_sk_callback_hsw+0x695>
+  .byte  197,121,111,13,166,47,0,0           // vmovdqa       0x2fa6(%rip),%xmm9        # 4a90 <_sk_callback_hsw+0x6a1>
   .byte  196,193,97,219,193                  // vpand         %xmm9,%xmm3,%xmm0
   .byte  196,226,125,51,200                  // vpmovzxwd     %xmm0,%ymm1
   .byte  197,229,118,219                     // vpcmpeqd      %ymm3,%ymm3,%ymm3
@@ -10150,41 +10172,41 @@ _sk_load_tables_rgb_u16_be_hsw:
   .byte  196,98,125,51,194                   // vpmovzxwd     %xmm2,%ymm8
   .byte  196,162,101,146,20,128              // vgatherdps    %ymm3,(%rax,%ymm8,4),%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,25,43,0,0         // vbroadcastss  0x2b19(%rip),%ymm3        # 4644 <_sk_callback_hsw+0x269>
+  .byte  196,226,125,24,29,21,43,0,0         // vbroadcastss  0x2b15(%rip),%ymm3        # 4654 <_sk_callback_hsw+0x265>
   .byte  255,224                             // jmpq          *%rax
   .byte  196,129,121,110,4,72                // vmovd         (%r8,%r9,2),%xmm0
   .byte  196,129,121,196,68,72,4,2           // vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           1b46 <_sk_load_tables_rgb_u16_be_hsw+0xec>
-  .byte  233,90,255,255,255                  // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  117,5                               // jne           1b5a <_sk_load_tables_rgb_u16_be_hsw+0xec>
+  .byte  233,90,255,255,255                  // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   .byte  196,129,121,110,76,72,6             // vmovd         0x6(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,68,72,10,2            // vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            1b75 <_sk_load_tables_rgb_u16_be_hsw+0x11b>
+  .byte  114,26                              // jb            1b89 <_sk_load_tables_rgb_u16_be_hsw+0x11b>
   .byte  196,129,121,110,76,72,12            // vmovd         0xc(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,84,72,16,2          // vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           1b7a <_sk_load_tables_rgb_u16_be_hsw+0x120>
-  .byte  233,43,255,255,255                  // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  .byte  233,38,255,255,255                  // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           1b8e <_sk_load_tables_rgb_u16_be_hsw+0x120>
+  .byte  233,43,255,255,255                  // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,38,255,255,255                  // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   .byte  196,129,121,110,76,72,18            // vmovd         0x12(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,76,72,22,2            // vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            1ba9 <_sk_load_tables_rgb_u16_be_hsw+0x14f>
+  .byte  114,26                              // jb            1bbd <_sk_load_tables_rgb_u16_be_hsw+0x14f>
   .byte  196,129,121,110,76,72,24            // vmovd         0x18(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,76,72,28,2          // vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           1bae <_sk_load_tables_rgb_u16_be_hsw+0x154>
-  .byte  233,247,254,255,255                 // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  .byte  233,242,254,255,255                 // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           1bc2 <_sk_load_tables_rgb_u16_be_hsw+0x154>
+  .byte  233,247,254,255,255                 // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,242,254,255,255                 // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   .byte  196,129,121,110,92,72,30            // vmovd         0x1e(%r8,%r9,2),%xmm3
   .byte  196,1,97,196,92,72,34,2             // vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            1bd7 <_sk_load_tables_rgb_u16_be_hsw+0x17d>
+  .byte  114,20                              // jb            1beb <_sk_load_tables_rgb_u16_be_hsw+0x17d>
   .byte  196,129,121,110,92,72,36            // vmovd         0x24(%r8,%r9,2),%xmm3
   .byte  196,129,97,196,92,72,40,2           // vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  .byte  233,201,254,255,255                 // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  .byte  233,196,254,255,255                 // jmpq          1aa0 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,201,254,255,255                 // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,196,254,255,255                 // jmpq          1ab4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
 
 HIDDEN _sk_byte_tables_hsw
 .globl _sk_byte_tables_hsw
@@ -10197,7 +10219,7 @@ _sk_byte_tables_hsw:
   .byte  65,84                               // push          %r12
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,87,42,0,0           // vbroadcastss  0x2a57(%rip),%ymm8        # 4648 <_sk_callback_hsw+0x26d>
+  .byte  196,98,125,24,5,83,42,0,0           // vbroadcastss  0x2a53(%rip),%ymm8        # 4658 <_sk_callback_hsw+0x269>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,195,249,22,192,1                // vpextrq       $0x1,%xmm0,%r8
@@ -10234,7 +10256,7 @@ _sk_byte_tables_hsw:
   .byte  196,227,121,32,197,7                // vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,168,41,0,0         // vbroadcastss  0x29a8(%rip),%ymm9        # 464c <_sk_callback_hsw+0x271>
+  .byte  196,98,125,24,13,164,41,0,0         // vbroadcastss  0x29a4(%rip),%ymm9        # 465c <_sk_callback_hsw+0x26d>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -10395,7 +10417,7 @@ _sk_byte_tables_rgb_hsw:
   .byte  196,227,121,32,197,7                // vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,225,38,0,0         // vbroadcastss  0x26e1(%rip),%ymm9        # 4650 <_sk_callback_hsw+0x275>
+  .byte  196,98,125,24,13,221,38,0,0         // vbroadcastss  0x26dd(%rip),%ymm9        # 4660 <_sk_callback_hsw+0x271>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -10558,33 +10580,33 @@ _sk_parametric_r_hsw:
   .byte  196,66,125,168,211                  // vfmadd213ps   %ymm11,%ymm0,%ymm10
   .byte  196,226,125,24,0                    // vbroadcastss  (%rax),%ymm0
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,148,36,0,0         // vbroadcastss  0x2494(%rip),%ymm12        # 4654 <_sk_callback_hsw+0x279>
-  .byte  196,98,125,24,45,143,36,0,0         // vbroadcastss  0x248f(%rip),%ymm13        # 4658 <_sk_callback_hsw+0x27d>
+  .byte  196,98,125,24,37,144,36,0,0         // vbroadcastss  0x2490(%rip),%ymm12        # 4664 <_sk_callback_hsw+0x275>
+  .byte  196,98,125,24,45,139,36,0,0         // vbroadcastss  0x248b(%rip),%ymm13        # 4668 <_sk_callback_hsw+0x279>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,133,36,0,0         // vbroadcastss  0x2485(%rip),%ymm13        # 465c <_sk_callback_hsw+0x281>
+  .byte  196,98,125,24,45,129,36,0,0         // vbroadcastss  0x2481(%rip),%ymm13        # 466c <_sk_callback_hsw+0x27d>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,123,36,0,0         // vbroadcastss  0x247b(%rip),%ymm13        # 4660 <_sk_callback_hsw+0x285>
+  .byte  196,98,125,24,45,119,36,0,0         // vbroadcastss  0x2477(%rip),%ymm13        # 4670 <_sk_callback_hsw+0x281>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,113,36,0,0         // vbroadcastss  0x2471(%rip),%ymm11        # 4664 <_sk_callback_hsw+0x289>
+  .byte  196,98,125,24,29,109,36,0,0         // vbroadcastss  0x246d(%rip),%ymm11        # 4674 <_sk_callback_hsw+0x285>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,103,36,0,0         // vbroadcastss  0x2467(%rip),%ymm12        # 4668 <_sk_callback_hsw+0x28d>
+  .byte  196,98,125,24,37,99,36,0,0          // vbroadcastss  0x2463(%rip),%ymm12        # 4678 <_sk_callback_hsw+0x289>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,93,36,0,0          // vbroadcastss  0x245d(%rip),%ymm12        # 466c <_sk_callback_hsw+0x291>
+  .byte  196,98,125,24,37,89,36,0,0          // vbroadcastss  0x2459(%rip),%ymm12        # 467c <_sk_callback_hsw+0x28d>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  196,99,125,8,208,1                  // vroundps      $0x1,%ymm0,%ymm10
   .byte  196,65,124,92,210                   // vsubps        %ymm10,%ymm0,%ymm10
-  .byte  196,98,125,24,29,62,36,0,0          // vbroadcastss  0x243e(%rip),%ymm11        # 4670 <_sk_callback_hsw+0x295>
+  .byte  196,98,125,24,29,58,36,0,0          // vbroadcastss  0x243a(%rip),%ymm11        # 4680 <_sk_callback_hsw+0x291>
   .byte  196,193,124,88,195                  // vaddps        %ymm11,%ymm0,%ymm0
-  .byte  196,98,125,24,29,52,36,0,0          // vbroadcastss  0x2434(%rip),%ymm11        # 4674 <_sk_callback_hsw+0x299>
+  .byte  196,98,125,24,29,48,36,0,0          // vbroadcastss  0x2430(%rip),%ymm11        # 4684 <_sk_callback_hsw+0x295>
   .byte  196,98,45,172,216                   // vfnmadd213ps  %ymm0,%ymm10,%ymm11
-  .byte  196,226,125,24,5,42,36,0,0          // vbroadcastss  0x242a(%rip),%ymm0        # 4678 <_sk_callback_hsw+0x29d>
+  .byte  196,226,125,24,5,38,36,0,0          // vbroadcastss  0x2426(%rip),%ymm0        # 4688 <_sk_callback_hsw+0x299>
   .byte  196,193,124,92,194                  // vsubps        %ymm10,%ymm0,%ymm0
-  .byte  196,98,125,24,21,32,36,0,0          // vbroadcastss  0x2420(%rip),%ymm10        # 467c <_sk_callback_hsw+0x2a1>
+  .byte  196,98,125,24,21,28,36,0,0          // vbroadcastss  0x241c(%rip),%ymm10        # 468c <_sk_callback_hsw+0x29d>
   .byte  197,172,94,192                      // vdivps        %ymm0,%ymm10,%ymm0
   .byte  197,164,88,192                      // vaddps        %ymm0,%ymm11,%ymm0
-  .byte  196,98,125,24,21,19,36,0,0          // vbroadcastss  0x2413(%rip),%ymm10        # 4680 <_sk_callback_hsw+0x2a5>
+  .byte  196,98,125,24,21,15,36,0,0          // vbroadcastss  0x240f(%rip),%ymm10        # 4690 <_sk_callback_hsw+0x2a1>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -10592,7 +10614,7 @@ _sk_parametric_r_hsw:
   .byte  196,195,125,74,193,128              // vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,124,95,192                  // vmaxps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,234,35,0,0          // vbroadcastss  0x23ea(%rip),%ymm8        # 4684 <_sk_callback_hsw+0x2a9>
+  .byte  196,98,125,24,5,230,35,0,0          // vbroadcastss  0x23e6(%rip),%ymm8        # 4694 <_sk_callback_hsw+0x2a5>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10612,33 +10634,33 @@ _sk_parametric_g_hsw:
   .byte  196,66,117,168,211                  // vfmadd213ps   %ymm11,%ymm1,%ymm10
   .byte  196,226,125,24,8                    // vbroadcastss  (%rax),%ymm1
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,162,35,0,0         // vbroadcastss  0x23a2(%rip),%ymm12        # 4688 <_sk_callback_hsw+0x2ad>
-  .byte  196,98,125,24,45,157,35,0,0         // vbroadcastss  0x239d(%rip),%ymm13        # 468c <_sk_callback_hsw+0x2b1>
+  .byte  196,98,125,24,37,158,35,0,0         // vbroadcastss  0x239e(%rip),%ymm12        # 4698 <_sk_callback_hsw+0x2a9>
+  .byte  196,98,125,24,45,153,35,0,0         // vbroadcastss  0x2399(%rip),%ymm13        # 469c <_sk_callback_hsw+0x2ad>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,147,35,0,0         // vbroadcastss  0x2393(%rip),%ymm13        # 4690 <_sk_callback_hsw+0x2b5>
+  .byte  196,98,125,24,45,143,35,0,0         // vbroadcastss  0x238f(%rip),%ymm13        # 46a0 <_sk_callback_hsw+0x2b1>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,137,35,0,0         // vbroadcastss  0x2389(%rip),%ymm13        # 4694 <_sk_callback_hsw+0x2b9>
+  .byte  196,98,125,24,45,133,35,0,0         // vbroadcastss  0x2385(%rip),%ymm13        # 46a4 <_sk_callback_hsw+0x2b5>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,127,35,0,0         // vbroadcastss  0x237f(%rip),%ymm11        # 4698 <_sk_callback_hsw+0x2bd>
+  .byte  196,98,125,24,29,123,35,0,0         // vbroadcastss  0x237b(%rip),%ymm11        # 46a8 <_sk_callback_hsw+0x2b9>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,117,35,0,0         // vbroadcastss  0x2375(%rip),%ymm12        # 469c <_sk_callback_hsw+0x2c1>
+  .byte  196,98,125,24,37,113,35,0,0         // vbroadcastss  0x2371(%rip),%ymm12        # 46ac <_sk_callback_hsw+0x2bd>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,107,35,0,0         // vbroadcastss  0x236b(%rip),%ymm12        # 46a0 <_sk_callback_hsw+0x2c5>
+  .byte  196,98,125,24,37,103,35,0,0         // vbroadcastss  0x2367(%rip),%ymm12        # 46b0 <_sk_callback_hsw+0x2c1>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  196,99,125,8,209,1                  // vroundps      $0x1,%ymm1,%ymm10
   .byte  196,65,116,92,210                   // vsubps        %ymm10,%ymm1,%ymm10
-  .byte  196,98,125,24,29,76,35,0,0          // vbroadcastss  0x234c(%rip),%ymm11        # 46a4 <_sk_callback_hsw+0x2c9>
+  .byte  196,98,125,24,29,72,35,0,0          // vbroadcastss  0x2348(%rip),%ymm11        # 46b4 <_sk_callback_hsw+0x2c5>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,66,35,0,0          // vbroadcastss  0x2342(%rip),%ymm11        # 46a8 <_sk_callback_hsw+0x2cd>
+  .byte  196,98,125,24,29,62,35,0,0          // vbroadcastss  0x233e(%rip),%ymm11        # 46b8 <_sk_callback_hsw+0x2c9>
   .byte  196,98,45,172,217                   // vfnmadd213ps  %ymm1,%ymm10,%ymm11
-  .byte  196,226,125,24,13,56,35,0,0         // vbroadcastss  0x2338(%rip),%ymm1        # 46ac <_sk_callback_hsw+0x2d1>
+  .byte  196,226,125,24,13,52,35,0,0         // vbroadcastss  0x2334(%rip),%ymm1        # 46bc <_sk_callback_hsw+0x2cd>
   .byte  196,193,116,92,202                  // vsubps        %ymm10,%ymm1,%ymm1
-  .byte  196,98,125,24,21,46,35,0,0          // vbroadcastss  0x232e(%rip),%ymm10        # 46b0 <_sk_callback_hsw+0x2d5>
+  .byte  196,98,125,24,21,42,35,0,0          // vbroadcastss  0x232a(%rip),%ymm10        # 46c0 <_sk_callback_hsw+0x2d1>
   .byte  197,172,94,201                      // vdivps        %ymm1,%ymm10,%ymm1
   .byte  197,164,88,201                      // vaddps        %ymm1,%ymm11,%ymm1
-  .byte  196,98,125,24,21,33,35,0,0          // vbroadcastss  0x2321(%rip),%ymm10        # 46b4 <_sk_callback_hsw+0x2d9>
+  .byte  196,98,125,24,21,29,35,0,0          // vbroadcastss  0x231d(%rip),%ymm10        # 46c4 <_sk_callback_hsw+0x2d5>
   .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -10646,7 +10668,7 @@ _sk_parametric_g_hsw:
   .byte  196,195,117,74,201,128              // vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,116,95,200                  // vmaxps        %ymm8,%ymm1,%ymm1
-  .byte  196,98,125,24,5,248,34,0,0          // vbroadcastss  0x22f8(%rip),%ymm8        # 46b8 <_sk_callback_hsw+0x2dd>
+  .byte  196,98,125,24,5,244,34,0,0          // vbroadcastss  0x22f4(%rip),%ymm8        # 46c8 <_sk_callback_hsw+0x2d9>
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10666,33 +10688,33 @@ _sk_parametric_b_hsw:
   .byte  196,66,109,168,211                  // vfmadd213ps   %ymm11,%ymm2,%ymm10
   .byte  196,226,125,24,16                   // vbroadcastss  (%rax),%ymm2
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,176,34,0,0         // vbroadcastss  0x22b0(%rip),%ymm12        # 46bc <_sk_callback_hsw+0x2e1>
-  .byte  196,98,125,24,45,171,34,0,0         // vbroadcastss  0x22ab(%rip),%ymm13        # 46c0 <_sk_callback_hsw+0x2e5>
+  .byte  196,98,125,24,37,172,34,0,0         // vbroadcastss  0x22ac(%rip),%ymm12        # 46cc <_sk_callback_hsw+0x2dd>
+  .byte  196,98,125,24,45,167,34,0,0         // vbroadcastss  0x22a7(%rip),%ymm13        # 46d0 <_sk_callback_hsw+0x2e1>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,161,34,0,0         // vbroadcastss  0x22a1(%rip),%ymm13        # 46c4 <_sk_callback_hsw+0x2e9>
+  .byte  196,98,125,24,45,157,34,0,0         // vbroadcastss  0x229d(%rip),%ymm13        # 46d4 <_sk_callback_hsw+0x2e5>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,151,34,0,0         // vbroadcastss  0x2297(%rip),%ymm13        # 46c8 <_sk_callback_hsw+0x2ed>
+  .byte  196,98,125,24,45,147,34,0,0         // vbroadcastss  0x2293(%rip),%ymm13        # 46d8 <_sk_callback_hsw+0x2e9>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,141,34,0,0         // vbroadcastss  0x228d(%rip),%ymm11        # 46cc <_sk_callback_hsw+0x2f1>
+  .byte  196,98,125,24,29,137,34,0,0         // vbroadcastss  0x2289(%rip),%ymm11        # 46dc <_sk_callback_hsw+0x2ed>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,131,34,0,0         // vbroadcastss  0x2283(%rip),%ymm12        # 46d0 <_sk_callback_hsw+0x2f5>
+  .byte  196,98,125,24,37,127,34,0,0         // vbroadcastss  0x227f(%rip),%ymm12        # 46e0 <_sk_callback_hsw+0x2f1>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,121,34,0,0         // vbroadcastss  0x2279(%rip),%ymm12        # 46d4 <_sk_callback_hsw+0x2f9>
+  .byte  196,98,125,24,37,117,34,0,0         // vbroadcastss  0x2275(%rip),%ymm12        # 46e4 <_sk_callback_hsw+0x2f5>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  196,99,125,8,210,1                  // vroundps      $0x1,%ymm2,%ymm10
   .byte  196,65,108,92,210                   // vsubps        %ymm10,%ymm2,%ymm10
-  .byte  196,98,125,24,29,90,34,0,0          // vbroadcastss  0x225a(%rip),%ymm11        # 46d8 <_sk_callback_hsw+0x2fd>
+  .byte  196,98,125,24,29,86,34,0,0          // vbroadcastss  0x2256(%rip),%ymm11        # 46e8 <_sk_callback_hsw+0x2f9>
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
-  .byte  196,98,125,24,29,80,34,0,0          // vbroadcastss  0x2250(%rip),%ymm11        # 46dc <_sk_callback_hsw+0x301>
+  .byte  196,98,125,24,29,76,34,0,0          // vbroadcastss  0x224c(%rip),%ymm11        # 46ec <_sk_callback_hsw+0x2fd>
   .byte  196,98,45,172,218                   // vfnmadd213ps  %ymm2,%ymm10,%ymm11
-  .byte  196,226,125,24,21,70,34,0,0         // vbroadcastss  0x2246(%rip),%ymm2        # 46e0 <_sk_callback_hsw+0x305>
+  .byte  196,226,125,24,21,66,34,0,0         // vbroadcastss  0x2242(%rip),%ymm2        # 46f0 <_sk_callback_hsw+0x301>
   .byte  196,193,108,92,210                  // vsubps        %ymm10,%ymm2,%ymm2
-  .byte  196,98,125,24,21,60,34,0,0          // vbroadcastss  0x223c(%rip),%ymm10        # 46e4 <_sk_callback_hsw+0x309>
+  .byte  196,98,125,24,21,56,34,0,0          // vbroadcastss  0x2238(%rip),%ymm10        # 46f4 <_sk_callback_hsw+0x305>
   .byte  197,172,94,210                      // vdivps        %ymm2,%ymm10,%ymm2
   .byte  197,164,88,210                      // vaddps        %ymm2,%ymm11,%ymm2
-  .byte  196,98,125,24,21,47,34,0,0          // vbroadcastss  0x222f(%rip),%ymm10        # 46e8 <_sk_callback_hsw+0x30d>
+  .byte  196,98,125,24,21,43,34,0,0          // vbroadcastss  0x222b(%rip),%ymm10        # 46f8 <_sk_callback_hsw+0x309>
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  197,253,91,210                      // vcvtps2dq     %ymm2,%ymm2
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -10700,7 +10722,7 @@ _sk_parametric_b_hsw:
   .byte  196,195,109,74,209,128              // vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,108,95,208                  // vmaxps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,6,34,0,0            // vbroadcastss  0x2206(%rip),%ymm8        # 46ec <_sk_callback_hsw+0x311>
+  .byte  196,98,125,24,5,2,34,0,0            // vbroadcastss  0x2202(%rip),%ymm8        # 46fc <_sk_callback_hsw+0x30d>
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10720,33 +10742,33 @@ _sk_parametric_a_hsw:
   .byte  196,66,101,168,211                  // vfmadd213ps   %ymm11,%ymm3,%ymm10
   .byte  196,226,125,24,24                   // vbroadcastss  (%rax),%ymm3
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,190,33,0,0         // vbroadcastss  0x21be(%rip),%ymm12        # 46f0 <_sk_callback_hsw+0x315>
-  .byte  196,98,125,24,45,185,33,0,0         // vbroadcastss  0x21b9(%rip),%ymm13        # 46f4 <_sk_callback_hsw+0x319>
+  .byte  196,98,125,24,37,186,33,0,0         // vbroadcastss  0x21ba(%rip),%ymm12        # 4700 <_sk_callback_hsw+0x311>
+  .byte  196,98,125,24,45,181,33,0,0         // vbroadcastss  0x21b5(%rip),%ymm13        # 4704 <_sk_callback_hsw+0x315>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,175,33,0,0         // vbroadcastss  0x21af(%rip),%ymm13        # 46f8 <_sk_callback_hsw+0x31d>
+  .byte  196,98,125,24,45,171,33,0,0         // vbroadcastss  0x21ab(%rip),%ymm13        # 4708 <_sk_callback_hsw+0x319>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,165,33,0,0         // vbroadcastss  0x21a5(%rip),%ymm13        # 46fc <_sk_callback_hsw+0x321>
+  .byte  196,98,125,24,45,161,33,0,0         // vbroadcastss  0x21a1(%rip),%ymm13        # 470c <_sk_callback_hsw+0x31d>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,155,33,0,0         // vbroadcastss  0x219b(%rip),%ymm11        # 4700 <_sk_callback_hsw+0x325>
+  .byte  196,98,125,24,29,151,33,0,0         // vbroadcastss  0x2197(%rip),%ymm11        # 4710 <_sk_callback_hsw+0x321>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,145,33,0,0         // vbroadcastss  0x2191(%rip),%ymm12        # 4704 <_sk_callback_hsw+0x329>
+  .byte  196,98,125,24,37,141,33,0,0         // vbroadcastss  0x218d(%rip),%ymm12        # 4714 <_sk_callback_hsw+0x325>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,135,33,0,0         // vbroadcastss  0x2187(%rip),%ymm12        # 4708 <_sk_callback_hsw+0x32d>
+  .byte  196,98,125,24,37,131,33,0,0         // vbroadcastss  0x2183(%rip),%ymm12        # 4718 <_sk_callback_hsw+0x329>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
   .byte  196,99,125,8,211,1                  // vroundps      $0x1,%ymm3,%ymm10
   .byte  196,65,100,92,210                   // vsubps        %ymm10,%ymm3,%ymm10
-  .byte  196,98,125,24,29,104,33,0,0         // vbroadcastss  0x2168(%rip),%ymm11        # 470c <_sk_callback_hsw+0x331>
+  .byte  196,98,125,24,29,100,33,0,0         // vbroadcastss  0x2164(%rip),%ymm11        # 471c <_sk_callback_hsw+0x32d>
   .byte  196,193,100,88,219                  // vaddps        %ymm11,%ymm3,%ymm3
-  .byte  196,98,125,24,29,94,33,0,0          // vbroadcastss  0x215e(%rip),%ymm11        # 4710 <_sk_callback_hsw+0x335>
+  .byte  196,98,125,24,29,90,33,0,0          // vbroadcastss  0x215a(%rip),%ymm11        # 4720 <_sk_callback_hsw+0x331>
   .byte  196,98,45,172,219                   // vfnmadd213ps  %ymm3,%ymm10,%ymm11
-  .byte  196,226,125,24,29,84,33,0,0         // vbroadcastss  0x2154(%rip),%ymm3        # 4714 <_sk_callback_hsw+0x339>
+  .byte  196,226,125,24,29,80,33,0,0         // vbroadcastss  0x2150(%rip),%ymm3        # 4724 <_sk_callback_hsw+0x335>
   .byte  196,193,100,92,218                  // vsubps        %ymm10,%ymm3,%ymm3
-  .byte  196,98,125,24,21,74,33,0,0          // vbroadcastss  0x214a(%rip),%ymm10        # 4718 <_sk_callback_hsw+0x33d>
+  .byte  196,98,125,24,21,70,33,0,0          // vbroadcastss  0x2146(%rip),%ymm10        # 4728 <_sk_callback_hsw+0x339>
   .byte  197,172,94,219                      // vdivps        %ymm3,%ymm10,%ymm3
   .byte  197,164,88,219                      // vaddps        %ymm3,%ymm11,%ymm3
-  .byte  196,98,125,24,21,61,33,0,0          // vbroadcastss  0x213d(%rip),%ymm10        # 471c <_sk_callback_hsw+0x341>
+  .byte  196,98,125,24,21,57,33,0,0          // vbroadcastss  0x2139(%rip),%ymm10        # 472c <_sk_callback_hsw+0x33d>
   .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
   .byte  197,253,91,219                      // vcvtps2dq     %ymm3,%ymm3
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -10754,7 +10776,7 @@ _sk_parametric_a_hsw:
   .byte  196,195,101,74,217,128              // vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,100,95,216                  // vmaxps        %ymm8,%ymm3,%ymm3
-  .byte  196,98,125,24,5,20,33,0,0           // vbroadcastss  0x2114(%rip),%ymm8        # 4720 <_sk_callback_hsw+0x345>
+  .byte  196,98,125,24,5,16,33,0,0           // vbroadcastss  0x2110(%rip),%ymm8        # 4730 <_sk_callback_hsw+0x341>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10763,26 +10785,26 @@ HIDDEN _sk_lab_to_xyz_hsw
 .globl _sk_lab_to_xyz_hsw
 FUNCTION(_sk_lab_to_xyz_hsw)
 _sk_lab_to_xyz_hsw:
-  .byte  196,98,125,24,5,6,33,0,0            // vbroadcastss  0x2106(%rip),%ymm8        # 4724 <_sk_callback_hsw+0x349>
-  .byte  196,98,125,24,13,1,33,0,0           // vbroadcastss  0x2101(%rip),%ymm9        # 4728 <_sk_callback_hsw+0x34d>
-  .byte  196,98,125,24,21,252,32,0,0         // vbroadcastss  0x20fc(%rip),%ymm10        # 472c <_sk_callback_hsw+0x351>
+  .byte  196,98,125,24,5,2,33,0,0            // vbroadcastss  0x2102(%rip),%ymm8        # 4734 <_sk_callback_hsw+0x345>
+  .byte  196,98,125,24,13,253,32,0,0         // vbroadcastss  0x20fd(%rip),%ymm9        # 4738 <_sk_callback_hsw+0x349>
+  .byte  196,98,125,24,21,248,32,0,0         // vbroadcastss  0x20f8(%rip),%ymm10        # 473c <_sk_callback_hsw+0x34d>
   .byte  196,194,53,168,202                  // vfmadd213ps   %ymm10,%ymm9,%ymm1
   .byte  196,194,53,168,210                  // vfmadd213ps   %ymm10,%ymm9,%ymm2
-  .byte  196,98,125,24,13,237,32,0,0         // vbroadcastss  0x20ed(%rip),%ymm9        # 4730 <_sk_callback_hsw+0x355>
+  .byte  196,98,125,24,13,233,32,0,0         // vbroadcastss  0x20e9(%rip),%ymm9        # 4740 <_sk_callback_hsw+0x351>
   .byte  196,66,125,184,200                  // vfmadd231ps   %ymm8,%ymm0,%ymm9
-  .byte  196,226,125,24,5,227,32,0,0         // vbroadcastss  0x20e3(%rip),%ymm0        # 4734 <_sk_callback_hsw+0x359>
+  .byte  196,226,125,24,5,223,32,0,0         // vbroadcastss  0x20df(%rip),%ymm0        # 4744 <_sk_callback_hsw+0x355>
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
-  .byte  196,98,125,24,5,218,32,0,0          // vbroadcastss  0x20da(%rip),%ymm8        # 4738 <_sk_callback_hsw+0x35d>
+  .byte  196,98,125,24,5,214,32,0,0          // vbroadcastss  0x20d6(%rip),%ymm8        # 4748 <_sk_callback_hsw+0x359>
   .byte  196,98,117,168,192                  // vfmadd213ps   %ymm0,%ymm1,%ymm8
-  .byte  196,98,125,24,13,208,32,0,0         // vbroadcastss  0x20d0(%rip),%ymm9        # 473c <_sk_callback_hsw+0x361>
+  .byte  196,98,125,24,13,204,32,0,0         // vbroadcastss  0x20cc(%rip),%ymm9        # 474c <_sk_callback_hsw+0x35d>
   .byte  196,98,109,172,200                  // vfnmadd213ps  %ymm0,%ymm2,%ymm9
   .byte  196,193,60,89,200                   // vmulps        %ymm8,%ymm8,%ymm1
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
-  .byte  196,226,125,24,21,189,32,0,0        // vbroadcastss  0x20bd(%rip),%ymm2        # 4740 <_sk_callback_hsw+0x365>
+  .byte  196,226,125,24,21,185,32,0,0        // vbroadcastss  0x20b9(%rip),%ymm2        # 4750 <_sk_callback_hsw+0x361>
   .byte  197,108,194,209,1                   // vcmpltps      %ymm1,%ymm2,%ymm10
-  .byte  196,98,125,24,29,179,32,0,0         // vbroadcastss  0x20b3(%rip),%ymm11        # 4744 <_sk_callback_hsw+0x369>
+  .byte  196,98,125,24,29,175,32,0,0         // vbroadcastss  0x20af(%rip),%ymm11        # 4754 <_sk_callback_hsw+0x365>
   .byte  196,65,60,88,195                    // vaddps        %ymm11,%ymm8,%ymm8
-  .byte  196,98,125,24,37,169,32,0,0         // vbroadcastss  0x20a9(%rip),%ymm12        # 4748 <_sk_callback_hsw+0x36d>
+  .byte  196,98,125,24,37,165,32,0,0         // vbroadcastss  0x20a5(%rip),%ymm12        # 4758 <_sk_callback_hsw+0x369>
   .byte  196,65,60,89,196                    // vmulps        %ymm12,%ymm8,%ymm8
   .byte  196,99,61,74,193,160                // vblendvps     %ymm10,%ymm1,%ymm8,%ymm8
   .byte  197,252,89,200                      // vmulps        %ymm0,%ymm0,%ymm1
@@ -10797,9 +10819,9 @@ _sk_lab_to_xyz_hsw:
   .byte  196,65,52,88,203                    // vaddps        %ymm11,%ymm9,%ymm9
   .byte  196,65,52,89,204                    // vmulps        %ymm12,%ymm9,%ymm9
   .byte  196,227,53,74,208,32                // vblendvps     %ymm2,%ymm0,%ymm9,%ymm2
-  .byte  196,226,125,24,5,94,32,0,0          // vbroadcastss  0x205e(%rip),%ymm0        # 474c <_sk_callback_hsw+0x371>
+  .byte  196,226,125,24,5,90,32,0,0          // vbroadcastss  0x205a(%rip),%ymm0        # 475c <_sk_callback_hsw+0x36d>
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
-  .byte  196,98,125,24,5,85,32,0,0           // vbroadcastss  0x2055(%rip),%ymm8        # 4750 <_sk_callback_hsw+0x375>
+  .byte  196,98,125,24,5,81,32,0,0           // vbroadcastss  0x2051(%rip),%ymm8        # 4760 <_sk_callback_hsw+0x371>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10813,11 +10835,11 @@ _sk_load_a8_hsw:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,45                              // jne           2741 <_sk_load_a8_hsw+0x3d>
+  .byte  117,45                              // jne           2755 <_sk_load_a8_hsw+0x3d>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,42,32,0,0         // vbroadcastss  0x202a(%rip),%ymm1        # 4754 <_sk_callback_hsw+0x379>
+  .byte  196,226,125,24,13,38,32,0,0         // vbroadcastss  0x2026(%rip),%ymm1        # 4764 <_sk_callback_hsw+0x375>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -10834,9 +10856,9 @@ _sk_load_a8_hsw:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           2749 <_sk_load_a8_hsw+0x45>
+  .byte  117,234                             // jne           275d <_sk_load_a8_hsw+0x45>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,178                             // jmp           2718 <_sk_load_a8_hsw+0x14>
+  .byte  235,178                             // jmp           272c <_sk_load_a8_hsw+0x14>
 
 HIDDEN _sk_gather_a8_hsw
 .globl _sk_gather_a8_hsw
@@ -10882,7 +10904,7 @@ _sk_gather_a8_hsw:
   .byte  196,227,121,32,192,7                // vpinsrb       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,53,31,0,0         // vbroadcastss  0x1f35(%rip),%ymm1        # 4758 <_sk_callback_hsw+0x37d>
+  .byte  196,226,125,24,13,49,31,0,0         // vbroadcastss  0x1f31(%rip),%ymm1        # 4768 <_sk_callback_hsw+0x379>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -10900,14 +10922,14 @@ FUNCTION(_sk_store_a8_hsw)
 _sk_store_a8_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,16,31,0,0           // vbroadcastss  0x1f10(%rip),%ymm8        # 475c <_sk_callback_hsw+0x381>
+  .byte  196,98,125,24,5,12,31,0,0           // vbroadcastss  0x1f0c(%rip),%ymm8        # 476c <_sk_callback_hsw+0x37d>
   .byte  196,65,100,89,192                   // vmulps        %ymm8,%ymm3,%ymm8
   .byte  196,65,125,91,192                   // vcvtps2dq     %ymm8,%ymm8
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  196,65,57,103,192                   // vpackuswb     %xmm8,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           2875 <_sk_store_a8_hsw+0x37>
+  .byte  117,10                              // jne           2889 <_sk_store_a8_hsw+0x37>
   .byte  196,65,123,17,4,58                  // vmovsd        %xmm8,(%r10,%rdi,1)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10915,10 +10937,10 @@ _sk_store_a8_hsw:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            2871 <_sk_store_a8_hsw+0x33>
+  .byte  119,236                             // ja            2885 <_sk_store_a8_hsw+0x33>
   .byte  196,66,121,48,192                   // vpmovzxbw     %xmm8,%xmm8
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 28d8 <_sk_store_a8_hsw+0x9a>
+  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 28ec <_sk_store_a8_hsw+0x9a>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10929,7 +10951,7 @@ _sk_store_a8_hsw:
   .byte  196,67,121,20,68,58,2,4             // vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   .byte  196,67,121,20,68,58,1,2             // vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   .byte  196,67,121,20,4,58,0                // vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  .byte  235,154                             // jmp           2871 <_sk_store_a8_hsw+0x33>
+  .byte  235,154                             // jmp           2885 <_sk_store_a8_hsw+0x33>
   .byte  144                                 // nop
   .byte  246,255                             // idiv          %bh
   .byte  255                                 // (bad)
@@ -10963,14 +10985,14 @@ _sk_load_g8_hsw:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,50                              // jne           2936 <_sk_load_g8_hsw+0x42>
+  .byte  117,50                              // jne           294a <_sk_load_g8_hsw+0x42>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,70,30,0,0         // vbroadcastss  0x1e46(%rip),%ymm1        # 4760 <_sk_callback_hsw+0x385>
+  .byte  196,226,125,24,13,66,30,0,0         // vbroadcastss  0x1e42(%rip),%ymm1        # 4770 <_sk_callback_hsw+0x381>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,59,30,0,0         // vbroadcastss  0x1e3b(%rip),%ymm3        # 4764 <_sk_callback_hsw+0x389>
+  .byte  196,226,125,24,29,55,30,0,0         // vbroadcastss  0x1e37(%rip),%ymm3        # 4774 <_sk_callback_hsw+0x385>
   .byte  76,137,193                          // mov           %r8,%rcx
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
@@ -10984,9 +11006,9 @@ _sk_load_g8_hsw:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           293e <_sk_load_g8_hsw+0x4a>
+  .byte  117,234                             // jne           2952 <_sk_load_g8_hsw+0x4a>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,173                             // jmp           2908 <_sk_load_g8_hsw+0x14>
+  .byte  235,173                             // jmp           291c <_sk_load_g8_hsw+0x14>
 
 HIDDEN _sk_gather_g8_hsw
 .globl _sk_gather_g8_hsw
@@ -11032,10 +11054,10 @@ _sk_gather_g8_hsw:
   .byte  196,227,121,32,192,7                // vpinsrb       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,80,29,0,0         // vbroadcastss  0x1d50(%rip),%ymm1        # 4768 <_sk_callback_hsw+0x38d>
+  .byte  196,226,125,24,13,76,29,0,0         // vbroadcastss  0x1d4c(%rip),%ymm1        # 4778 <_sk_callback_hsw+0x389>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,69,29,0,0         // vbroadcastss  0x1d45(%rip),%ymm3        # 476c <_sk_callback_hsw+0x391>
+  .byte  196,226,125,24,29,65,29,0,0         // vbroadcastss  0x1d41(%rip),%ymm3        # 477c <_sk_callback_hsw+0x38d>
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
   .byte  91                                  // pop           %rbx
@@ -11051,9 +11073,9 @@ _sk_gather_i8_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            2a47 <_sk_gather_i8_hsw+0xf>
+  .byte  116,5                               // je            2a5b <_sk_gather_i8_hsw+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           2a49 <_sk_gather_i8_hsw+0x11>
+  .byte  235,2                               // jmp           2a5d <_sk_gather_i8_hsw+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,87                               // push          %r15
   .byte  65,86                               // push          %r14
@@ -11091,14 +11113,14 @@ _sk_gather_i8_hsw:
   .byte  73,139,64,8                         // mov           0x8(%r8),%rax
   .byte  197,245,118,201                     // vpcmpeqd      %ymm1,%ymm1,%ymm1
   .byte  196,226,117,144,28,128              // vpgatherdd    %ymm1,(%rax,%ymm0,4),%ymm3
-  .byte  197,229,219,5,73,30,0,0             // vpand         0x1e49(%rip),%ymm3,%ymm0        # 4940 <_sk_callback_hsw+0x565>
+  .byte  197,229,219,5,85,30,0,0             // vpand         0x1e55(%rip),%ymm3,%ymm0        # 4960 <_sk_callback_hsw+0x571>
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,108,28,0,0          // vbroadcastss  0x1c6c(%rip),%ymm8        # 4770 <_sk_callback_hsw+0x395>
+  .byte  196,98,125,24,5,104,28,0,0          // vbroadcastss  0x1c68(%rip),%ymm8        # 4780 <_sk_callback_hsw+0x391>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,226,101,0,13,78,30,0,0          // vpshufb       0x1e4e(%rip),%ymm3,%ymm1        # 4960 <_sk_callback_hsw+0x585>
+  .byte  196,226,101,0,13,90,30,0,0          // vpshufb       0x1e5a(%rip),%ymm3,%ymm1        # 4980 <_sk_callback_hsw+0x591>
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,226,101,0,21,92,30,0,0          // vpshufb       0x1e5c(%rip),%ymm3,%ymm2        # 4980 <_sk_callback_hsw+0x5a5>
+  .byte  196,226,101,0,21,104,30,0,0         // vpshufb       0x1e68(%rip),%ymm3,%ymm2        # 49a0 <_sk_callback_hsw+0x5b1>
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
@@ -11119,35 +11141,35 @@ _sk_load_565_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,114                             // jne           2bc4 <_sk_load_565_hsw+0x7c>
+  .byte  117,114                             // jne           2bd8 <_sk_load_565_hsw+0x7c>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  196,226,125,51,208                  // vpmovzxwd     %xmm0,%ymm2
-  .byte  196,226,125,88,5,14,28,0,0          // vpbroadcastd  0x1c0e(%rip),%ymm0        # 4774 <_sk_callback_hsw+0x399>
+  .byte  196,226,125,88,5,10,28,0,0          // vpbroadcastd  0x1c0a(%rip),%ymm0        # 4784 <_sk_callback_hsw+0x395>
   .byte  197,237,219,192                     // vpand         %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,1,28,0,0          // vbroadcastss  0x1c01(%rip),%ymm1        # 4778 <_sk_callback_hsw+0x39d>
+  .byte  196,226,125,24,13,253,27,0,0        // vbroadcastss  0x1bfd(%rip),%ymm1        # 4788 <_sk_callback_hsw+0x399>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,248,27,0,0        // vpbroadcastd  0x1bf8(%rip),%ymm1        # 477c <_sk_callback_hsw+0x3a1>
+  .byte  196,226,125,88,13,244,27,0,0        // vpbroadcastd  0x1bf4(%rip),%ymm1        # 478c <_sk_callback_hsw+0x39d>
   .byte  197,237,219,201                     // vpand         %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,235,27,0,0        // vbroadcastss  0x1beb(%rip),%ymm3        # 4780 <_sk_callback_hsw+0x3a5>
+  .byte  196,226,125,24,29,231,27,0,0        // vbroadcastss  0x1be7(%rip),%ymm3        # 4790 <_sk_callback_hsw+0x3a1>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,88,29,226,27,0,0        // vpbroadcastd  0x1be2(%rip),%ymm3        # 4784 <_sk_callback_hsw+0x3a9>
+  .byte  196,226,125,88,29,222,27,0,0        // vpbroadcastd  0x1bde(%rip),%ymm3        # 4794 <_sk_callback_hsw+0x3a5>
   .byte  197,237,219,211                     // vpand         %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,213,27,0,0        // vbroadcastss  0x1bd5(%rip),%ymm3        # 4788 <_sk_callback_hsw+0x3ad>
+  .byte  196,226,125,24,29,209,27,0,0        // vbroadcastss  0x1bd1(%rip),%ymm3        # 4798 <_sk_callback_hsw+0x3a9>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,202,27,0,0        // vbroadcastss  0x1bca(%rip),%ymm3        # 478c <_sk_callback_hsw+0x3b1>
+  .byte  196,226,125,24,29,198,27,0,0        // vbroadcastss  0x1bc6(%rip),%ymm3        # 479c <_sk_callback_hsw+0x3ad>
   .byte  255,224                             // jmpq          *%rax
   .byte  65,137,200                          // mov           %ecx,%r8d
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,128                             // ja            2b58 <_sk_load_565_hsw+0x10>
+  .byte  119,128                             // ja            2b6c <_sk_load_565_hsw+0x10>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 2c2c <_sk_load_565_hsw+0xe4>
+  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 2c40 <_sk_load_565_hsw+0xe4>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11159,7 +11181,7 @@ _sk_load_565_hsw:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,44,255,255,255                  // jmpq          2b58 <_sk_load_565_hsw+0x10>
+  .byte  233,44,255,255,255                  // jmpq          2b6c <_sk_load_565_hsw+0x10>
   .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -11229,23 +11251,23 @@ _sk_gather_565_hsw:
   .byte  65,15,183,4,88                      // movzwl        (%r8,%rbx,2),%eax
   .byte  197,249,196,192,7                   // vpinsrw       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,51,208                  // vpmovzxwd     %xmm0,%ymm2
-  .byte  196,226,125,88,5,141,26,0,0         // vpbroadcastd  0x1a8d(%rip),%ymm0        # 4790 <_sk_callback_hsw+0x3b5>
+  .byte  196,226,125,88,5,137,26,0,0         // vpbroadcastd  0x1a89(%rip),%ymm0        # 47a0 <_sk_callback_hsw+0x3b1>
   .byte  197,237,219,192                     // vpand         %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,128,26,0,0        // vbroadcastss  0x1a80(%rip),%ymm1        # 4794 <_sk_callback_hsw+0x3b9>
+  .byte  196,226,125,24,13,124,26,0,0        // vbroadcastss  0x1a7c(%rip),%ymm1        # 47a4 <_sk_callback_hsw+0x3b5>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,119,26,0,0        // vpbroadcastd  0x1a77(%rip),%ymm1        # 4798 <_sk_callback_hsw+0x3bd>
+  .byte  196,226,125,88,13,115,26,0,0        // vpbroadcastd  0x1a73(%rip),%ymm1        # 47a8 <_sk_callback_hsw+0x3b9>
   .byte  197,237,219,201                     // vpand         %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,106,26,0,0        // vbroadcastss  0x1a6a(%rip),%ymm3        # 479c <_sk_callback_hsw+0x3c1>
+  .byte  196,226,125,24,29,102,26,0,0        // vbroadcastss  0x1a66(%rip),%ymm3        # 47ac <_sk_callback_hsw+0x3bd>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,88,29,97,26,0,0         // vpbroadcastd  0x1a61(%rip),%ymm3        # 47a0 <_sk_callback_hsw+0x3c5>
+  .byte  196,226,125,88,29,93,26,0,0         // vpbroadcastd  0x1a5d(%rip),%ymm3        # 47b0 <_sk_callback_hsw+0x3c1>
   .byte  197,237,219,211                     // vpand         %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,84,26,0,0         // vbroadcastss  0x1a54(%rip),%ymm3        # 47a4 <_sk_callback_hsw+0x3c9>
+  .byte  196,226,125,24,29,80,26,0,0         // vbroadcastss  0x1a50(%rip),%ymm3        # 47b4 <_sk_callback_hsw+0x3c5>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,73,26,0,0         // vbroadcastss  0x1a49(%rip),%ymm3        # 47a8 <_sk_callback_hsw+0x3cd>
+  .byte  196,226,125,24,29,69,26,0,0         // vbroadcastss  0x1a45(%rip),%ymm3        # 47b8 <_sk_callback_hsw+0x3c9>
   .byte  91                                  // pop           %rbx
   .byte  65,92                               // pop           %r12
   .byte  65,94                               // pop           %r14
@@ -11258,11 +11280,11 @@ FUNCTION(_sk_store_565_hsw)
 _sk_store_565_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,54,26,0,0           // vbroadcastss  0x1a36(%rip),%ymm8        # 47ac <_sk_callback_hsw+0x3d1>
+  .byte  196,98,125,24,5,50,26,0,0           // vbroadcastss  0x1a32(%rip),%ymm8        # 47bc <_sk_callback_hsw+0x3cd>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,53,114,241,11               // vpslld        $0xb,%ymm9,%ymm9
-  .byte  196,98,125,24,21,33,26,0,0          // vbroadcastss  0x1a21(%rip),%ymm10        # 47b0 <_sk_callback_hsw+0x3d5>
+  .byte  196,98,125,24,21,29,26,0,0          // vbroadcastss  0x1a1d(%rip),%ymm10        # 47c0 <_sk_callback_hsw+0x3d1>
   .byte  196,65,116,89,210                   // vmulps        %ymm10,%ymm1,%ymm10
   .byte  196,65,125,91,210                   // vcvtps2dq     %ymm10,%ymm10
   .byte  196,193,45,114,242,5                // vpslld        $0x5,%ymm10,%ymm10
@@ -11273,7 +11295,7 @@ _sk_store_565_hsw:
   .byte  196,67,125,57,193,1                 // vextracti128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           2dcd <_sk_store_565_hsw+0x65>
+  .byte  117,10                              // jne           2de1 <_sk_store_565_hsw+0x65>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11281,9 +11303,9 @@ _sk_store_565_hsw:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            2dc9 <_sk_store_565_hsw+0x61>
+  .byte  119,236                             // ja            2ddd <_sk_store_565_hsw+0x61>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 2e2c <_sk_store_565_hsw+0xc4>
+  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 2e40 <_sk_store_565_hsw+0xc4>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11294,7 +11316,7 @@ _sk_store_565_hsw:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           2dc9 <_sk_store_565_hsw+0x61>
+  .byte  235,159                             // jmp           2ddd <_sk_store_565_hsw+0x61>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -11327,28 +11349,28 @@ _sk_load_4444_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,138,0,0,0                    // jne           2ee0 <_sk_load_4444_hsw+0x98>
+  .byte  15,133,138,0,0,0                    // jne           2ef4 <_sk_load_4444_hsw+0x98>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  196,226,125,51,216                  // vpmovzxwd     %xmm0,%ymm3
-  .byte  196,226,125,88,5,74,25,0,0          // vpbroadcastd  0x194a(%rip),%ymm0        # 47b4 <_sk_callback_hsw+0x3d9>
+  .byte  196,226,125,88,5,70,25,0,0          // vpbroadcastd  0x1946(%rip),%ymm0        # 47c4 <_sk_callback_hsw+0x3d5>
   .byte  197,229,219,192                     // vpand         %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,61,25,0,0         // vbroadcastss  0x193d(%rip),%ymm1        # 47b8 <_sk_callback_hsw+0x3dd>
+  .byte  196,226,125,24,13,57,25,0,0         // vbroadcastss  0x1939(%rip),%ymm1        # 47c8 <_sk_callback_hsw+0x3d9>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,52,25,0,0         // vpbroadcastd  0x1934(%rip),%ymm1        # 47bc <_sk_callback_hsw+0x3e1>
+  .byte  196,226,125,88,13,48,25,0,0         // vpbroadcastd  0x1930(%rip),%ymm1        # 47cc <_sk_callback_hsw+0x3dd>
   .byte  197,229,219,201                     // vpand         %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,39,25,0,0         // vbroadcastss  0x1927(%rip),%ymm2        # 47c0 <_sk_callback_hsw+0x3e5>
+  .byte  196,226,125,24,21,35,25,0,0         // vbroadcastss  0x1923(%rip),%ymm2        # 47d0 <_sk_callback_hsw+0x3e1>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,88,21,30,25,0,0         // vpbroadcastd  0x191e(%rip),%ymm2        # 47c4 <_sk_callback_hsw+0x3e9>
+  .byte  196,226,125,88,21,26,25,0,0         // vpbroadcastd  0x191a(%rip),%ymm2        # 47d4 <_sk_callback_hsw+0x3e5>
   .byte  197,229,219,210                     // vpand         %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,17,25,0,0           // vbroadcastss  0x1911(%rip),%ymm8        # 47c8 <_sk_callback_hsw+0x3ed>
+  .byte  196,98,125,24,5,13,25,0,0           // vbroadcastss  0x190d(%rip),%ymm8        # 47d8 <_sk_callback_hsw+0x3e9>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,88,5,7,25,0,0            // vpbroadcastd  0x1907(%rip),%ymm8        # 47cc <_sk_callback_hsw+0x3f1>
+  .byte  196,98,125,88,5,3,25,0,0            // vpbroadcastd  0x1903(%rip),%ymm8        # 47dc <_sk_callback_hsw+0x3ed>
   .byte  196,193,101,219,216                 // vpand         %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,249,24,0,0          // vbroadcastss  0x18f9(%rip),%ymm8        # 47d0 <_sk_callback_hsw+0x3f5>
+  .byte  196,98,125,24,5,245,24,0,0          // vbroadcastss  0x18f5(%rip),%ymm8        # 47e0 <_sk_callback_hsw+0x3f1>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11357,9 +11379,9 @@ _sk_load_4444_hsw:
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,100,255,255,255              // ja            2e5c <_sk_load_4444_hsw+0x14>
+  .byte  15,135,100,255,255,255              // ja            2e70 <_sk_load_4444_hsw+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 2f4c <_sk_load_4444_hsw+0x104>
+  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 2f60 <_sk_load_4444_hsw+0x104>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11371,7 +11393,7 @@ _sk_load_4444_hsw:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,16,255,255,255                  // jmpq          2e5c <_sk_load_4444_hsw+0x14>
+  .byte  233,16,255,255,255                  // jmpq          2e70 <_sk_load_4444_hsw+0x14>
   .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -11441,25 +11463,25 @@ _sk_gather_4444_hsw:
   .byte  65,15,183,4,88                      // movzwl        (%r8,%rbx,2),%eax
   .byte  197,249,196,192,7                   // vpinsrw       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,51,216                  // vpmovzxwd     %xmm0,%ymm3
-  .byte  196,226,125,88,5,177,23,0,0         // vpbroadcastd  0x17b1(%rip),%ymm0        # 47d4 <_sk_callback_hsw+0x3f9>
+  .byte  196,226,125,88,5,173,23,0,0         // vpbroadcastd  0x17ad(%rip),%ymm0        # 47e4 <_sk_callback_hsw+0x3f5>
   .byte  197,229,219,192                     // vpand         %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,164,23,0,0        // vbroadcastss  0x17a4(%rip),%ymm1        # 47d8 <_sk_callback_hsw+0x3fd>
+  .byte  196,226,125,24,13,160,23,0,0        // vbroadcastss  0x17a0(%rip),%ymm1        # 47e8 <_sk_callback_hsw+0x3f9>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,155,23,0,0        // vpbroadcastd  0x179b(%rip),%ymm1        # 47dc <_sk_callback_hsw+0x401>
+  .byte  196,226,125,88,13,151,23,0,0        // vpbroadcastd  0x1797(%rip),%ymm1        # 47ec <_sk_callback_hsw+0x3fd>
   .byte  197,229,219,201                     // vpand         %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,142,23,0,0        // vbroadcastss  0x178e(%rip),%ymm2        # 47e0 <_sk_callback_hsw+0x405>
+  .byte  196,226,125,24,21,138,23,0,0        // vbroadcastss  0x178a(%rip),%ymm2        # 47f0 <_sk_callback_hsw+0x401>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,88,21,133,23,0,0        // vpbroadcastd  0x1785(%rip),%ymm2        # 47e4 <_sk_callback_hsw+0x409>
+  .byte  196,226,125,88,21,129,23,0,0        // vpbroadcastd  0x1781(%rip),%ymm2        # 47f4 <_sk_callback_hsw+0x405>
   .byte  197,229,219,210                     // vpand         %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,120,23,0,0          // vbroadcastss  0x1778(%rip),%ymm8        # 47e8 <_sk_callback_hsw+0x40d>
+  .byte  196,98,125,24,5,116,23,0,0          // vbroadcastss  0x1774(%rip),%ymm8        # 47f8 <_sk_callback_hsw+0x409>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,88,5,110,23,0,0          // vpbroadcastd  0x176e(%rip),%ymm8        # 47ec <_sk_callback_hsw+0x411>
+  .byte  196,98,125,88,5,106,23,0,0          // vpbroadcastd  0x176a(%rip),%ymm8        # 47fc <_sk_callback_hsw+0x40d>
   .byte  196,193,101,219,216                 // vpand         %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,96,23,0,0           // vbroadcastss  0x1760(%rip),%ymm8        # 47f0 <_sk_callback_hsw+0x415>
+  .byte  196,98,125,24,5,92,23,0,0           // vbroadcastss  0x175c(%rip),%ymm8        # 4800 <_sk_callback_hsw+0x411>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -11474,7 +11496,7 @@ FUNCTION(_sk_store_4444_hsw)
 _sk_store_4444_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,70,23,0,0           // vbroadcastss  0x1746(%rip),%ymm8        # 47f4 <_sk_callback_hsw+0x419>
+  .byte  196,98,125,24,5,66,23,0,0           // vbroadcastss  0x1742(%rip),%ymm8        # 4804 <_sk_callback_hsw+0x415>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,53,114,241,12               // vpslld        $0xc,%ymm9,%ymm9
@@ -11492,7 +11514,7 @@ _sk_store_4444_hsw:
   .byte  196,67,125,57,193,1                 // vextracti128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3111 <_sk_store_4444_hsw+0x71>
+  .byte  117,10                              // jne           3125 <_sk_store_4444_hsw+0x71>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11500,9 +11522,9 @@ _sk_store_4444_hsw:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            310d <_sk_store_4444_hsw+0x6d>
+  .byte  119,236                             // ja            3121 <_sk_store_4444_hsw+0x6d>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 3170 <_sk_store_4444_hsw+0xd0>
+  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 3184 <_sk_store_4444_hsw+0xd0>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11513,7 +11535,7 @@ _sk_store_4444_hsw:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           310d <_sk_store_4444_hsw+0x6d>
+  .byte  235,159                             // jmp           3121 <_sk_store_4444_hsw+0x6d>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -11548,16 +11570,16 @@ _sk_load_8888_hsw:
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  76,3,8                              // add           (%rax),%r9
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,88                              // jne           31f9 <_sk_load_8888_hsw+0x6d>
+  .byte  117,88                              // jne           320d <_sk_load_8888_hsw+0x6d>
   .byte  196,193,126,111,25                  // vmovdqu       (%r9),%ymm3
-  .byte  197,229,219,5,242,23,0,0            // vpand         0x17f2(%rip),%ymm3,%ymm0        # 49a0 <_sk_callback_hsw+0x5c5>
+  .byte  197,229,219,5,254,23,0,0            // vpand         0x17fe(%rip),%ymm3,%ymm0        # 49c0 <_sk_callback_hsw+0x5d1>
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,61,22,0,0           // vbroadcastss  0x163d(%rip),%ymm8        # 47f8 <_sk_callback_hsw+0x41d>
+  .byte  196,98,125,24,5,57,22,0,0           // vbroadcastss  0x1639(%rip),%ymm8        # 4808 <_sk_callback_hsw+0x419>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,226,101,0,13,247,23,0,0         // vpshufb       0x17f7(%rip),%ymm3,%ymm1        # 49c0 <_sk_callback_hsw+0x5e5>
+  .byte  196,226,101,0,13,3,24,0,0           // vpshufb       0x1803(%rip),%ymm3,%ymm1        # 49e0 <_sk_callback_hsw+0x5f1>
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,226,101,0,21,5,24,0,0           // vpshufb       0x1805(%rip),%ymm3,%ymm2        # 49e0 <_sk_callback_hsw+0x605>
+  .byte  196,226,101,0,21,17,24,0,0          // vpshufb       0x1811(%rip),%ymm3,%ymm2        # 4a00 <_sk_callback_hsw+0x611>
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
@@ -11574,7 +11596,7 @@ _sk_load_8888_hsw:
   .byte  196,225,249,110,192                 // vmovq         %rax,%xmm0
   .byte  196,226,125,33,192                  // vpmovsxbd     %xmm0,%ymm0
   .byte  196,194,125,140,25                  // vpmaskmovd    (%r9),%ymm0,%ymm3
-  .byte  235,135                             // jmp           31a6 <_sk_load_8888_hsw+0x1a>
+  .byte  235,135                             // jmp           31ba <_sk_load_8888_hsw+0x1a>
 
 HIDDEN _sk_gather_8888_hsw
 .globl _sk_gather_8888_hsw
@@ -11589,14 +11611,14 @@ _sk_gather_8888_hsw:
   .byte  197,245,254,192                     // vpaddd        %ymm0,%ymm1,%ymm0
   .byte  197,245,118,201                     // vpcmpeqd      %ymm1,%ymm1,%ymm1
   .byte  196,194,117,144,28,128              // vpgatherdd    %ymm1,(%r8,%ymm0,4),%ymm3
-  .byte  197,229,219,5,179,23,0,0            // vpand         0x17b3(%rip),%ymm3,%ymm0        # 4a00 <_sk_callback_hsw+0x625>
+  .byte  197,229,219,5,191,23,0,0            // vpand         0x17bf(%rip),%ymm3,%ymm0        # 4a20 <_sk_callback_hsw+0x631>
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,162,21,0,0          // vbroadcastss  0x15a2(%rip),%ymm8        # 47fc <_sk_callback_hsw+0x421>
+  .byte  196,98,125,24,5,158,21,0,0          // vbroadcastss  0x159e(%rip),%ymm8        # 480c <_sk_callback_hsw+0x41d>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,226,101,0,13,184,23,0,0         // vpshufb       0x17b8(%rip),%ymm3,%ymm1        # 4a20 <_sk_callback_hsw+0x645>
+  .byte  196,226,101,0,13,196,23,0,0         // vpshufb       0x17c4(%rip),%ymm3,%ymm1        # 4a40 <_sk_callback_hsw+0x651>
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,226,101,0,21,198,23,0,0         // vpshufb       0x17c6(%rip),%ymm3,%ymm2        # 4a40 <_sk_callback_hsw+0x665>
+  .byte  196,226,101,0,21,210,23,0,0         // vpshufb       0x17d2(%rip),%ymm3,%ymm2        # 4a60 <_sk_callback_hsw+0x671>
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
@@ -11613,7 +11635,7 @@ _sk_store_8888_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  76,3,8                              // add           (%rax),%r9
-  .byte  196,98,125,24,5,82,21,0,0           // vbroadcastss  0x1552(%rip),%ymm8        # 4800 <_sk_callback_hsw+0x425>
+  .byte  196,98,125,24,5,78,21,0,0           // vbroadcastss  0x154e(%rip),%ymm8        # 4810 <_sk_callback_hsw+0x421>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,65,116,89,208                   // vmulps        %ymm8,%ymm1,%ymm10
@@ -11629,7 +11651,7 @@ _sk_store_8888_hsw:
   .byte  196,65,45,235,192                   // vpor          %ymm8,%ymm10,%ymm8
   .byte  196,65,53,235,192                   // vpor          %ymm8,%ymm9,%ymm8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,12                              // jne           3308 <_sk_store_8888_hsw+0x73>
+  .byte  117,12                              // jne           331c <_sk_store_8888_hsw+0x73>
   .byte  196,65,126,127,1                    // vmovdqu       %ymm8,(%r9)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,137,193                          // mov           %r8,%rcx
@@ -11642,7 +11664,7 @@ _sk_store_8888_hsw:
   .byte  196,97,249,110,200                  // vmovq         %rax,%xmm9
   .byte  196,66,125,33,201                   // vpmovsxbd     %xmm9,%ymm9
   .byte  196,66,53,142,1                     // vpmaskmovd    %ymm8,%ymm9,(%r9)
-  .byte  235,211                             // jmp           3301 <_sk_store_8888_hsw+0x6c>
+  .byte  235,211                             // jmp           3315 <_sk_store_8888_hsw+0x6c>
 
 HIDDEN _sk_load_f16_hsw
 .globl _sk_load_f16_hsw
@@ -11651,7 +11673,7 @@ _sk_load_f16_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,97                              // jne           3399 <_sk_load_f16_hsw+0x6b>
+  .byte  117,97                              // jne           33ad <_sk_load_f16_hsw+0x6b>
   .byte  197,121,16,4,248                    // vmovupd       (%rax,%rdi,8),%xmm8
   .byte  197,249,16,84,248,16                // vmovupd       0x10(%rax,%rdi,8),%xmm2
   .byte  197,249,16,92,248,32                // vmovupd       0x20(%rax,%rdi,8),%xmm3
@@ -11677,29 +11699,29 @@ _sk_load_f16_hsw:
   .byte  197,123,16,4,248                    // vmovsd        (%rax,%rdi,8),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,79                              // je            33f8 <_sk_load_f16_hsw+0xca>
+  .byte  116,79                              // je            340c <_sk_load_f16_hsw+0xca>
   .byte  197,57,22,68,248,8                  // vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,67                              // jb            33f8 <_sk_load_f16_hsw+0xca>
+  .byte  114,67                              // jb            340c <_sk_load_f16_hsw+0xca>
   .byte  197,251,16,84,248,16                // vmovsd        0x10(%rax,%rdi,8),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,68                              // je            3405 <_sk_load_f16_hsw+0xd7>
+  .byte  116,68                              // je            3419 <_sk_load_f16_hsw+0xd7>
   .byte  197,233,22,84,248,24                // vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,56                              // jb            3405 <_sk_load_f16_hsw+0xd7>
+  .byte  114,56                              // jb            3419 <_sk_load_f16_hsw+0xd7>
   .byte  197,251,16,92,248,32                // vmovsd        0x20(%rax,%rdi,8),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,114,255,255,255              // je            334f <_sk_load_f16_hsw+0x21>
+  .byte  15,132,114,255,255,255              // je            3363 <_sk_load_f16_hsw+0x21>
   .byte  197,225,22,92,248,40                // vmovhpd       0x28(%rax,%rdi,8),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,98,255,255,255               // jb            334f <_sk_load_f16_hsw+0x21>
+  .byte  15,130,98,255,255,255               // jb            3363 <_sk_load_f16_hsw+0x21>
   .byte  197,122,126,76,248,48               // vmovq         0x30(%rax,%rdi,8),%xmm9
-  .byte  233,87,255,255,255                  // jmpq          334f <_sk_load_f16_hsw+0x21>
+  .byte  233,87,255,255,255                  // jmpq          3363 <_sk_load_f16_hsw+0x21>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,74,255,255,255                  // jmpq          334f <_sk_load_f16_hsw+0x21>
+  .byte  233,74,255,255,255                  // jmpq          3363 <_sk_load_f16_hsw+0x21>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,65,255,255,255                  // jmpq          334f <_sk_load_f16_hsw+0x21>
+  .byte  233,65,255,255,255                  // jmpq          3363 <_sk_load_f16_hsw+0x21>
 
 HIDDEN _sk_gather_f16_hsw
 .globl _sk_gather_f16_hsw
@@ -11757,7 +11779,7 @@ _sk_store_f16_hsw:
   .byte  196,65,57,98,205                    // vpunpckldq    %xmm13,%xmm8,%xmm9
   .byte  196,65,57,106,197                   // vpunpckhdq    %xmm13,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,27                              // jne           34fd <_sk_store_f16_hsw+0x65>
+  .byte  117,27                              // jne           3511 <_sk_store_f16_hsw+0x65>
   .byte  197,120,17,28,248                   // vmovups       %xmm11,(%rax,%rdi,8)
   .byte  197,120,17,84,248,16                // vmovups       %xmm10,0x10(%rax,%rdi,8)
   .byte  197,120,17,76,248,32                // vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -11766,22 +11788,22 @@ _sk_store_f16_hsw:
   .byte  255,224                             // jmpq          *%rax
   .byte  197,121,214,28,248                  // vmovq         %xmm11,(%rax,%rdi,8)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,241                             // je            34f9 <_sk_store_f16_hsw+0x61>
+  .byte  116,241                             // je            350d <_sk_store_f16_hsw+0x61>
   .byte  197,121,23,92,248,8                 // vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,229                             // jb            34f9 <_sk_store_f16_hsw+0x61>
+  .byte  114,229                             // jb            350d <_sk_store_f16_hsw+0x61>
   .byte  197,121,214,84,248,16               // vmovq         %xmm10,0x10(%rax,%rdi,8)
-  .byte  116,221                             // je            34f9 <_sk_store_f16_hsw+0x61>
+  .byte  116,221                             // je            350d <_sk_store_f16_hsw+0x61>
   .byte  197,121,23,84,248,24                // vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,209                             // jb            34f9 <_sk_store_f16_hsw+0x61>
+  .byte  114,209                             // jb            350d <_sk_store_f16_hsw+0x61>
   .byte  197,121,214,76,248,32               // vmovq         %xmm9,0x20(%rax,%rdi,8)
-  .byte  116,201                             // je            34f9 <_sk_store_f16_hsw+0x61>
+  .byte  116,201                             // je            350d <_sk_store_f16_hsw+0x61>
   .byte  197,121,23,76,248,40                // vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,189                             // jb            34f9 <_sk_store_f16_hsw+0x61>
+  .byte  114,189                             // jb            350d <_sk_store_f16_hsw+0x61>
   .byte  197,121,214,68,248,48               // vmovq         %xmm8,0x30(%rax,%rdi,8)
-  .byte  235,181                             // jmp           34f9 <_sk_store_f16_hsw+0x61>
+  .byte  235,181                             // jmp           350d <_sk_store_f16_hsw+0x61>
 
 HIDDEN _sk_load_u16_be_hsw
 .globl _sk_load_u16_be_hsw
@@ -11791,7 +11813,7 @@ _sk_load_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,204,0,0,0                    // jne           3626 <_sk_load_u16_be_hsw+0xe2>
+  .byte  15,133,204,0,0,0                    // jne           363a <_sk_load_u16_be_hsw+0xe2>
   .byte  196,65,121,16,4,64                  // vmovupd       (%r8,%rax,2),%xmm8
   .byte  196,193,121,16,84,64,16             // vmovupd       0x10(%r8,%rax,2),%xmm2
   .byte  196,193,121,16,92,64,32             // vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -11810,7 +11832,7 @@ _sk_load_u16_be_hsw:
   .byte  197,241,235,192                     // vpor          %xmm0,%xmm1,%xmm0
   .byte  196,226,125,51,192                  // vpmovzxwd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,21,73,18,0,0          // vbroadcastss  0x1249(%rip),%ymm10        # 4804 <_sk_callback_hsw+0x429>
+  .byte  196,98,125,24,21,69,18,0,0          // vbroadcastss  0x1245(%rip),%ymm10        # 4814 <_sk_callback_hsw+0x425>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,185,109,202                     // vpunpckhqdq   %xmm2,%xmm8,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -11838,29 +11860,29 @@ _sk_load_u16_be_hsw:
   .byte  196,65,123,16,4,64                  // vmovsd        (%r8,%rax,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            368c <_sk_load_u16_be_hsw+0x148>
+  .byte  116,85                              // je            36a0 <_sk_load_u16_be_hsw+0x148>
   .byte  196,65,57,22,68,64,8                // vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            368c <_sk_load_u16_be_hsw+0x148>
+  .byte  114,72                              // jb            36a0 <_sk_load_u16_be_hsw+0x148>
   .byte  196,193,123,16,84,64,16             // vmovsd        0x10(%r8,%rax,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            3699 <_sk_load_u16_be_hsw+0x155>
+  .byte  116,72                              // je            36ad <_sk_load_u16_be_hsw+0x155>
   .byte  196,193,105,22,84,64,24             // vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            3699 <_sk_load_u16_be_hsw+0x155>
+  .byte  114,59                              // jb            36ad <_sk_load_u16_be_hsw+0x155>
   .byte  196,193,123,16,92,64,32             // vmovsd        0x20(%r8,%rax,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,6,255,255,255                // je            3575 <_sk_load_u16_be_hsw+0x31>
+  .byte  15,132,6,255,255,255                // je            3589 <_sk_load_u16_be_hsw+0x31>
   .byte  196,193,97,22,92,64,40              // vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,245,254,255,255              // jb            3575 <_sk_load_u16_be_hsw+0x31>
+  .byte  15,130,245,254,255,255              // jb            3589 <_sk_load_u16_be_hsw+0x31>
   .byte  196,65,122,126,76,64,48             // vmovq         0x30(%r8,%rax,2),%xmm9
-  .byte  233,233,254,255,255                 // jmpq          3575 <_sk_load_u16_be_hsw+0x31>
+  .byte  233,233,254,255,255                 // jmpq          3589 <_sk_load_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,220,254,255,255                 // jmpq          3575 <_sk_load_u16_be_hsw+0x31>
+  .byte  233,220,254,255,255                 // jmpq          3589 <_sk_load_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,211,254,255,255                 // jmpq          3575 <_sk_load_u16_be_hsw+0x31>
+  .byte  233,211,254,255,255                 // jmpq          3589 <_sk_load_u16_be_hsw+0x31>
 
 HIDDEN _sk_load_rgb_u16_be_hsw
 .globl _sk_load_rgb_u16_be_hsw
@@ -11870,7 +11892,7 @@ _sk_load_rgb_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,127                        // lea           (%rdi,%rdi,2),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,204,0,0,0                    // jne           3780 <_sk_load_rgb_u16_be_hsw+0xde>
+  .byte  15,133,204,0,0,0                    // jne           3794 <_sk_load_rgb_u16_be_hsw+0xde>
   .byte  196,193,122,111,4,64                // vmovdqu       (%r8,%rax,2),%xmm0
   .byte  196,193,122,111,84,64,12            // vmovdqu       0xc(%r8,%rax,2),%xmm2
   .byte  196,193,122,111,76,64,24            // vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -11894,7 +11916,7 @@ _sk_load_rgb_u16_be_hsw:
   .byte  197,241,235,192                     // vpor          %xmm0,%xmm1,%xmm0
   .byte  196,226,125,51,192                  // vpmovzxwd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,21,218,16,0,0         // vbroadcastss  0x10da(%rip),%ymm10        # 4808 <_sk_callback_hsw+0x42d>
+  .byte  196,98,125,24,21,214,16,0,0         // vbroadcastss  0x10d6(%rip),%ymm10        # 4818 <_sk_callback_hsw+0x429>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,185,109,202                     // vpunpckhqdq   %xmm2,%xmm8,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -11911,41 +11933,41 @@ _sk_load_rgb_u16_be_hsw:
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,142,16,0,0        // vbroadcastss  0x108e(%rip),%ymm3        # 480c <_sk_callback_hsw+0x431>
+  .byte  196,226,125,24,29,138,16,0,0        // vbroadcastss  0x108a(%rip),%ymm3        # 481c <_sk_callback_hsw+0x42d>
   .byte  255,224                             // jmpq          *%rax
   .byte  196,193,121,110,4,64                // vmovd         (%r8,%rax,2),%xmm0
   .byte  196,193,121,196,68,64,4,2           // vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           3799 <_sk_load_rgb_u16_be_hsw+0xf7>
-  .byte  233,79,255,255,255                  // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  117,5                               // jne           37ad <_sk_load_rgb_u16_be_hsw+0xf7>
+  .byte  233,79,255,255,255                  // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
   .byte  196,193,121,110,76,64,6             // vmovd         0x6(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,68,64,10,2           // vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            37c8 <_sk_load_rgb_u16_be_hsw+0x126>
+  .byte  114,26                              // jb            37dc <_sk_load_rgb_u16_be_hsw+0x126>
   .byte  196,193,121,110,76,64,12            // vmovd         0xc(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,84,64,16,2          // vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           37cd <_sk_load_rgb_u16_be_hsw+0x12b>
-  .byte  233,32,255,255,255                  // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
-  .byte  233,27,255,255,255                  // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           37e1 <_sk_load_rgb_u16_be_hsw+0x12b>
+  .byte  233,32,255,255,255                  // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,27,255,255,255                  // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
   .byte  196,193,121,110,76,64,18            // vmovd         0x12(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,76,64,22,2           // vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            37fc <_sk_load_rgb_u16_be_hsw+0x15a>
+  .byte  114,26                              // jb            3810 <_sk_load_rgb_u16_be_hsw+0x15a>
   .byte  196,193,121,110,76,64,24            // vmovd         0x18(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,76,64,28,2          // vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           3801 <_sk_load_rgb_u16_be_hsw+0x15f>
-  .byte  233,236,254,255,255                 // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
-  .byte  233,231,254,255,255                 // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           3815 <_sk_load_rgb_u16_be_hsw+0x15f>
+  .byte  233,236,254,255,255                 // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,231,254,255,255                 // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
   .byte  196,193,121,110,92,64,30            // vmovd         0x1e(%r8,%rax,2),%xmm3
   .byte  196,65,97,196,92,64,34,2            // vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            382a <_sk_load_rgb_u16_be_hsw+0x188>
+  .byte  114,20                              // jb            383e <_sk_load_rgb_u16_be_hsw+0x188>
   .byte  196,193,121,110,92,64,36            // vmovd         0x24(%r8,%rax,2),%xmm3
   .byte  196,193,97,196,92,64,40,2           // vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  .byte  233,190,254,255,255                 // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
-  .byte  233,185,254,255,255                 // jmpq          36e8 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,190,254,255,255                 // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,185,254,255,255                 // jmpq          36fc <_sk_load_rgb_u16_be_hsw+0x46>
 
 HIDDEN _sk_store_u16_be_hsw
 .globl _sk_store_u16_be_hsw
@@ -11954,7 +11976,7 @@ _sk_store_u16_be_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
-  .byte  196,98,125,24,5,203,15,0,0          // vbroadcastss  0xfcb(%rip),%ymm8        # 4810 <_sk_callback_hsw+0x435>
+  .byte  196,98,125,24,5,199,15,0,0          // vbroadcastss  0xfc7(%rip),%ymm8        # 4820 <_sk_callback_hsw+0x431>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,67,125,25,202,1                 // vextractf128  $0x1,%ymm9,%xmm10
@@ -11992,7 +12014,7 @@ _sk_store_u16_be_hsw:
   .byte  196,65,17,98,200                    // vpunpckldq    %xmm8,%xmm13,%xmm9
   .byte  196,65,17,106,192                   // vpunpckhdq    %xmm8,%xmm13,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,31                              // jne           3929 <_sk_store_u16_be_hsw+0xfa>
+  .byte  117,31                              // jne           393d <_sk_store_u16_be_hsw+0xfa>
   .byte  196,65,120,17,28,64                 // vmovups       %xmm11,(%r8,%rax,2)
   .byte  196,65,120,17,84,64,16              // vmovups       %xmm10,0x10(%r8,%rax,2)
   .byte  196,65,120,17,76,64,32              // vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -12001,22 +12023,22 @@ _sk_store_u16_be_hsw:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,214,28,64                // vmovq         %xmm11,(%r8,%rax,2)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  116,240                             // je            3939 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,23,92,64,8               // vmovhpd       %xmm11,0x8(%r8,%rax,2)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  114,227                             // jb            3939 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,214,84,64,16             // vmovq         %xmm10,0x10(%r8,%rax,2)
-  .byte  116,218                             // je            3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  116,218                             // je            3939 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,23,84,64,24              // vmovhpd       %xmm10,0x18(%r8,%rax,2)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  114,205                             // jb            3939 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,214,76,64,32             // vmovq         %xmm9,0x20(%r8,%rax,2)
-  .byte  116,196                             // je            3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  116,196                             // je            3939 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,23,76,64,40              // vmovhpd       %xmm9,0x28(%r8,%rax,2)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,183                             // jb            3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  114,183                             // jb            3939 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,214,68,64,48             // vmovq         %xmm8,0x30(%r8,%rax,2)
-  .byte  235,174                             // jmp           3925 <_sk_store_u16_be_hsw+0xf6>
+  .byte  235,174                             // jmp           3939 <_sk_store_u16_be_hsw+0xf6>
 
 HIDDEN _sk_load_f32_hsw
 .globl _sk_load_f32_hsw
@@ -12024,10 +12046,10 @@ FUNCTION(_sk_load_f32_hsw)
 _sk_load_f32_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  119,110                             // ja            39ed <_sk_load_f32_hsw+0x76>
+  .byte  119,110                             // ja            3a01 <_sk_load_f32_hsw+0x76>
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
-  .byte  76,141,21,135,0,0,0                 // lea           0x87(%rip),%r10        # 3a18 <_sk_load_f32_hsw+0xa1>
+  .byte  76,141,21,135,0,0,0                 // lea           0x87(%rip),%r10        # 3a2c <_sk_load_f32_hsw+0xa1>
   .byte  73,99,4,138                         // movslq        (%r10,%rcx,4),%rax
   .byte  76,1,208                            // add           %r10,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -12088,7 +12110,7 @@ _sk_store_f32_hsw:
   .byte  196,65,37,20,196                    // vunpcklpd     %ymm12,%ymm11,%ymm8
   .byte  196,65,37,21,220                    // vunpckhpd     %ymm12,%ymm11,%ymm11
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,55                              // jne           3aa5 <_sk_store_f32_hsw+0x6d>
+  .byte  117,55                              // jne           3ab9 <_sk_store_f32_hsw+0x6d>
   .byte  196,67,45,24,225,1                  // vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   .byte  196,67,61,24,235,1                  // vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   .byte  196,67,45,6,201,49                  // vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -12101,22 +12123,22 @@ _sk_store_f32_hsw:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,17,20,128                // vmovupd       %xmm10,(%r8,%rax,4)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  116,240                             // je            3ab5 <_sk_store_f32_hsw+0x69>
   .byte  196,65,121,17,76,128,16             // vmovupd       %xmm9,0x10(%r8,%rax,4)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  114,227                             // jb            3ab5 <_sk_store_f32_hsw+0x69>
   .byte  196,65,121,17,68,128,32             // vmovupd       %xmm8,0x20(%r8,%rax,4)
-  .byte  116,218                             // je            3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  116,218                             // je            3ab5 <_sk_store_f32_hsw+0x69>
   .byte  196,65,121,17,92,128,48             // vmovupd       %xmm11,0x30(%r8,%rax,4)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  114,205                             // jb            3ab5 <_sk_store_f32_hsw+0x69>
   .byte  196,67,125,25,84,128,64,1           // vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  .byte  116,195                             // je            3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  116,195                             // je            3ab5 <_sk_store_f32_hsw+0x69>
   .byte  196,67,125,25,76,128,80,1           // vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,181                             // jb            3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  114,181                             // jb            3ab5 <_sk_store_f32_hsw+0x69>
   .byte  196,67,125,25,68,128,96,1           // vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  .byte  235,171                             // jmp           3aa1 <_sk_store_f32_hsw+0x69>
+  .byte  235,171                             // jmp           3ab5 <_sk_store_f32_hsw+0x69>
 
 HIDDEN _sk_clamp_x_hsw
 .globl _sk_clamp_x_hsw
@@ -12226,11 +12248,11 @@ HIDDEN _sk_luminance_to_alpha_hsw
 .globl _sk_luminance_to_alpha_hsw
 FUNCTION(_sk_luminance_to_alpha_hsw)
 _sk_luminance_to_alpha_hsw:
-  .byte  196,226,125,24,29,229,11,0,0        // vbroadcastss  0xbe5(%rip),%ymm3        # 4814 <_sk_callback_hsw+0x439>
-  .byte  196,98,125,24,5,224,11,0,0          // vbroadcastss  0xbe0(%rip),%ymm8        # 4818 <_sk_callback_hsw+0x43d>
+  .byte  196,226,125,24,29,225,11,0,0        // vbroadcastss  0xbe1(%rip),%ymm3        # 4824 <_sk_callback_hsw+0x435>
+  .byte  196,98,125,24,5,220,11,0,0          // vbroadcastss  0xbdc(%rip),%ymm8        # 4828 <_sk_callback_hsw+0x439>
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  196,226,125,184,203                 // vfmadd231ps   %ymm3,%ymm0,%ymm1
-  .byte  196,226,125,24,29,209,11,0,0        // vbroadcastss  0xbd1(%rip),%ymm3        # 481c <_sk_callback_hsw+0x441>
+  .byte  196,226,125,24,29,205,11,0,0        // vbroadcastss  0xbcd(%rip),%ymm3        # 482c <_sk_callback_hsw+0x43d>
   .byte  196,226,109,168,217                 // vfmadd213ps   %ymm1,%ymm2,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -12375,7 +12397,7 @@ _sk_linear_gradient_hsw:
   .byte  196,98,125,24,72,28                 // vbroadcastss  0x1c(%rax),%ymm9
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  15,132,143,0,0,0                    // je            3f23 <_sk_linear_gradient_hsw+0xb5>
+  .byte  15,132,143,0,0,0                    // je            3f37 <_sk_linear_gradient_hsw+0xb5>
   .byte  72,139,64,8                         // mov           0x8(%rax),%rax
   .byte  72,131,192,32                       // add           $0x20,%rax
   .byte  196,65,28,87,228                    // vxorps        %ymm12,%ymm12,%ymm12
@@ -12402,8 +12424,8 @@ _sk_linear_gradient_hsw:
   .byte  196,67,13,74,201,208                // vblendvps     %ymm13,%ymm9,%ymm14,%ymm9
   .byte  72,131,192,36                       // add           $0x24,%rax
   .byte  73,255,200                          // dec           %r8
-  .byte  117,140                             // jne           3ead <_sk_linear_gradient_hsw+0x3f>
-  .byte  235,17                              // jmp           3f34 <_sk_linear_gradient_hsw+0xc6>
+  .byte  117,140                             // jne           3ec1 <_sk_linear_gradient_hsw+0x3f>
+  .byte  235,17                              // jmp           3f48 <_sk_linear_gradient_hsw+0xc6>
   .byte  197,244,87,201                      // vxorps        %ymm1,%ymm1,%ymm1
   .byte  197,236,87,210                      // vxorps        %ymm2,%ymm2,%ymm2
   .byte  197,228,87,219                      // vxorps        %ymm3,%ymm3,%ymm3
@@ -12450,24 +12472,24 @@ _sk_xy_to_polar_unit_hsw:
   .byte  196,65,52,95,226                    // vmaxps        %ymm10,%ymm9,%ymm12
   .byte  196,65,36,94,220                    // vdivps        %ymm12,%ymm11,%ymm11
   .byte  196,65,36,89,227                    // vmulps        %ymm11,%ymm11,%ymm12
-  .byte  196,98,125,24,45,81,8,0,0           // vbroadcastss  0x851(%rip),%ymm13        # 4820 <_sk_callback_hsw+0x445>
-  .byte  196,98,125,24,53,76,8,0,0           // vbroadcastss  0x84c(%rip),%ymm14        # 4824 <_sk_callback_hsw+0x449>
+  .byte  196,98,125,24,45,77,8,0,0           // vbroadcastss  0x84d(%rip),%ymm13        # 4830 <_sk_callback_hsw+0x441>
+  .byte  196,98,125,24,53,72,8,0,0           // vbroadcastss  0x848(%rip),%ymm14        # 4834 <_sk_callback_hsw+0x445>
   .byte  196,66,29,184,245                   // vfmadd231ps   %ymm13,%ymm12,%ymm14
-  .byte  196,98,125,24,45,66,8,0,0           // vbroadcastss  0x842(%rip),%ymm13        # 4828 <_sk_callback_hsw+0x44d>
+  .byte  196,98,125,24,45,62,8,0,0           // vbroadcastss  0x83e(%rip),%ymm13        # 4838 <_sk_callback_hsw+0x449>
   .byte  196,66,29,184,238                   // vfmadd231ps   %ymm14,%ymm12,%ymm13
-  .byte  196,98,125,24,53,56,8,0,0           // vbroadcastss  0x838(%rip),%ymm14        # 482c <_sk_callback_hsw+0x451>
+  .byte  196,98,125,24,53,52,8,0,0           // vbroadcastss  0x834(%rip),%ymm14        # 483c <_sk_callback_hsw+0x44d>
   .byte  196,66,29,184,245                   // vfmadd231ps   %ymm13,%ymm12,%ymm14
   .byte  196,65,36,89,222                    // vmulps        %ymm14,%ymm11,%ymm11
   .byte  196,65,52,194,202,1                 // vcmpltps      %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,21,35,8,0,0           // vbroadcastss  0x823(%rip),%ymm10        # 4830 <_sk_callback_hsw+0x455>
+  .byte  196,98,125,24,21,31,8,0,0           // vbroadcastss  0x81f(%rip),%ymm10        # 4840 <_sk_callback_hsw+0x451>
   .byte  196,65,44,92,211                    // vsubps        %ymm11,%ymm10,%ymm10
   .byte  196,67,37,74,202,144                // vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   .byte  196,193,124,194,192,1               // vcmpltps      %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,21,13,8,0,0           // vbroadcastss  0x80d(%rip),%ymm10        # 4834 <_sk_callback_hsw+0x459>
+  .byte  196,98,125,24,21,9,8,0,0            // vbroadcastss  0x809(%rip),%ymm10        # 4844 <_sk_callback_hsw+0x455>
   .byte  196,65,44,92,209                    // vsubps        %ymm9,%ymm10,%ymm10
   .byte  196,195,53,74,194,0                 // vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   .byte  196,65,116,194,200,1                // vcmpltps      %ymm8,%ymm1,%ymm9
-  .byte  196,98,125,24,21,247,7,0,0          // vbroadcastss  0x7f7(%rip),%ymm10        # 4838 <_sk_callback_hsw+0x45d>
+  .byte  196,98,125,24,21,243,7,0,0          // vbroadcastss  0x7f3(%rip),%ymm10        # 4848 <_sk_callback_hsw+0x459>
   .byte  197,44,92,208                       // vsubps        %ymm0,%ymm10,%ymm10
   .byte  196,195,125,74,194,144              // vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   .byte  196,65,124,194,200,3                // vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -12491,7 +12513,7 @@ HIDDEN _sk_save_xy_hsw
 FUNCTION(_sk_save_xy_hsw)
 _sk_save_xy_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,192,7,0,0           // vbroadcastss  0x7c0(%rip),%ymm8        # 483c <_sk_callback_hsw+0x461>
+  .byte  196,98,125,24,5,188,7,0,0           // vbroadcastss  0x7bc(%rip),%ymm8        # 484c <_sk_callback_hsw+0x45d>
   .byte  196,65,124,88,200                   // vaddps        %ymm8,%ymm0,%ymm9
   .byte  196,67,125,8,209,1                  // vroundps      $0x1,%ymm9,%ymm10
   .byte  196,65,52,92,202                    // vsubps        %ymm10,%ymm9,%ymm9
@@ -12525,9 +12547,9 @@ HIDDEN _sk_bilinear_nx_hsw
 FUNCTION(_sk_bilinear_nx_hsw)
 _sk_bilinear_nx_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,84,7,0,0           // vbroadcastss  0x754(%rip),%ymm0        # 4840 <_sk_callback_hsw+0x465>
+  .byte  196,226,125,24,5,80,7,0,0           // vbroadcastss  0x750(%rip),%ymm0        # 4850 <_sk_callback_hsw+0x461>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,75,7,0,0            // vbroadcastss  0x74b(%rip),%ymm8        # 4844 <_sk_callback_hsw+0x469>
+  .byte  196,98,125,24,5,71,7,0,0            // vbroadcastss  0x747(%rip),%ymm8        # 4854 <_sk_callback_hsw+0x465>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12538,7 +12560,7 @@ HIDDEN _sk_bilinear_px_hsw
 FUNCTION(_sk_bilinear_px_hsw)
 _sk_bilinear_px_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,51,7,0,0           // vbroadcastss  0x733(%rip),%ymm0        # 4848 <_sk_callback_hsw+0x46d>
+  .byte  196,226,125,24,5,47,7,0,0           // vbroadcastss  0x72f(%rip),%ymm0        # 4858 <_sk_callback_hsw+0x469>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -12550,9 +12572,9 @@ HIDDEN _sk_bilinear_ny_hsw
 FUNCTION(_sk_bilinear_ny_hsw)
 _sk_bilinear_ny_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,23,7,0,0          // vbroadcastss  0x717(%rip),%ymm1        # 484c <_sk_callback_hsw+0x471>
+  .byte  196,226,125,24,13,19,7,0,0          // vbroadcastss  0x713(%rip),%ymm1        # 485c <_sk_callback_hsw+0x46d>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,13,7,0,0            // vbroadcastss  0x70d(%rip),%ymm8        # 4850 <_sk_callback_hsw+0x475>
+  .byte  196,98,125,24,5,9,7,0,0             // vbroadcastss  0x709(%rip),%ymm8        # 4860 <_sk_callback_hsw+0x471>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12563,7 +12585,7 @@ HIDDEN _sk_bilinear_py_hsw
 FUNCTION(_sk_bilinear_py_hsw)
 _sk_bilinear_py_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,245,6,0,0         // vbroadcastss  0x6f5(%rip),%ymm1        # 4854 <_sk_callback_hsw+0x479>
+  .byte  196,226,125,24,13,241,6,0,0         // vbroadcastss  0x6f1(%rip),%ymm1        # 4864 <_sk_callback_hsw+0x475>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -12575,13 +12597,13 @@ HIDDEN _sk_bicubic_n3x_hsw
 FUNCTION(_sk_bicubic_n3x_hsw)
 _sk_bicubic_n3x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,216,6,0,0          // vbroadcastss  0x6d8(%rip),%ymm0        # 4858 <_sk_callback_hsw+0x47d>
+  .byte  196,226,125,24,5,212,6,0,0          // vbroadcastss  0x6d4(%rip),%ymm0        # 4868 <_sk_callback_hsw+0x479>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,207,6,0,0           // vbroadcastss  0x6cf(%rip),%ymm8        # 485c <_sk_callback_hsw+0x481>
+  .byte  196,98,125,24,5,203,6,0,0           // vbroadcastss  0x6cb(%rip),%ymm8        # 486c <_sk_callback_hsw+0x47d>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,192,6,0,0          // vbroadcastss  0x6c0(%rip),%ymm10        # 4860 <_sk_callback_hsw+0x485>
-  .byte  196,98,125,24,29,187,6,0,0          // vbroadcastss  0x6bb(%rip),%ymm11        # 4864 <_sk_callback_hsw+0x489>
+  .byte  196,98,125,24,21,188,6,0,0          // vbroadcastss  0x6bc(%rip),%ymm10        # 4870 <_sk_callback_hsw+0x481>
+  .byte  196,98,125,24,29,183,6,0,0          // vbroadcastss  0x6b7(%rip),%ymm11        # 4874 <_sk_callback_hsw+0x485>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,36,89,193                    // vmulps        %ymm9,%ymm11,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -12593,16 +12615,16 @@ HIDDEN _sk_bicubic_n1x_hsw
 FUNCTION(_sk_bicubic_n1x_hsw)
 _sk_bicubic_n1x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,158,6,0,0          // vbroadcastss  0x69e(%rip),%ymm0        # 4868 <_sk_callback_hsw+0x48d>
+  .byte  196,226,125,24,5,154,6,0,0          // vbroadcastss  0x69a(%rip),%ymm0        # 4878 <_sk_callback_hsw+0x489>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,149,6,0,0           // vbroadcastss  0x695(%rip),%ymm8        # 486c <_sk_callback_hsw+0x491>
+  .byte  196,98,125,24,5,145,6,0,0           // vbroadcastss  0x691(%rip),%ymm8        # 487c <_sk_callback_hsw+0x48d>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,139,6,0,0          // vbroadcastss  0x68b(%rip),%ymm9        # 4870 <_sk_callback_hsw+0x495>
-  .byte  196,98,125,24,21,134,6,0,0          // vbroadcastss  0x686(%rip),%ymm10        # 4874 <_sk_callback_hsw+0x499>
+  .byte  196,98,125,24,13,135,6,0,0          // vbroadcastss  0x687(%rip),%ymm9        # 4880 <_sk_callback_hsw+0x491>
+  .byte  196,98,125,24,21,130,6,0,0          // vbroadcastss  0x682(%rip),%ymm10        # 4884 <_sk_callback_hsw+0x495>
   .byte  196,66,61,168,209                   // vfmadd213ps   %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,13,124,6,0,0          // vbroadcastss  0x67c(%rip),%ymm9        # 4878 <_sk_callback_hsw+0x49d>
+  .byte  196,98,125,24,13,120,6,0,0          // vbroadcastss  0x678(%rip),%ymm9        # 4888 <_sk_callback_hsw+0x499>
   .byte  196,66,61,184,202                   // vfmadd231ps   %ymm10,%ymm8,%ymm9
-  .byte  196,98,125,24,21,114,6,0,0          // vbroadcastss  0x672(%rip),%ymm10        # 487c <_sk_callback_hsw+0x4a1>
+  .byte  196,98,125,24,21,110,6,0,0          // vbroadcastss  0x66e(%rip),%ymm10        # 488c <_sk_callback_hsw+0x49d>
   .byte  196,66,61,184,209                   // vfmadd231ps   %ymm9,%ymm8,%ymm10
   .byte  197,124,17,144,128,0,0,0            // vmovups       %ymm10,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12613,14 +12635,14 @@ HIDDEN _sk_bicubic_p1x_hsw
 FUNCTION(_sk_bicubic_p1x_hsw)
 _sk_bicubic_p1x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,90,6,0,0            // vbroadcastss  0x65a(%rip),%ymm8        # 4880 <_sk_callback_hsw+0x4a5>
+  .byte  196,98,125,24,5,86,6,0,0            // vbroadcastss  0x656(%rip),%ymm8        # 4890 <_sk_callback_hsw+0x4a1>
   .byte  197,188,88,0                        // vaddps        (%rax),%ymm8,%ymm0
   .byte  197,124,16,72,64                    // vmovups       0x40(%rax),%ymm9
-  .byte  196,98,125,24,21,76,6,0,0           // vbroadcastss  0x64c(%rip),%ymm10        # 4884 <_sk_callback_hsw+0x4a9>
-  .byte  196,98,125,24,29,71,6,0,0           // vbroadcastss  0x647(%rip),%ymm11        # 4888 <_sk_callback_hsw+0x4ad>
+  .byte  196,98,125,24,21,72,6,0,0           // vbroadcastss  0x648(%rip),%ymm10        # 4894 <_sk_callback_hsw+0x4a5>
+  .byte  196,98,125,24,29,67,6,0,0           // vbroadcastss  0x643(%rip),%ymm11        # 4898 <_sk_callback_hsw+0x4a9>
   .byte  196,66,53,168,218                   // vfmadd213ps   %ymm10,%ymm9,%ymm11
   .byte  196,66,53,168,216                   // vfmadd213ps   %ymm8,%ymm9,%ymm11
-  .byte  196,98,125,24,5,56,6,0,0            // vbroadcastss  0x638(%rip),%ymm8        # 488c <_sk_callback_hsw+0x4b1>
+  .byte  196,98,125,24,5,52,6,0,0            // vbroadcastss  0x634(%rip),%ymm8        # 489c <_sk_callback_hsw+0x4ad>
   .byte  196,66,53,184,195                   // vfmadd231ps   %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12631,12 +12653,12 @@ HIDDEN _sk_bicubic_p3x_hsw
 FUNCTION(_sk_bicubic_p3x_hsw)
 _sk_bicubic_p3x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,32,6,0,0           // vbroadcastss  0x620(%rip),%ymm0        # 4890 <_sk_callback_hsw+0x4b5>
+  .byte  196,226,125,24,5,28,6,0,0           // vbroadcastss  0x61c(%rip),%ymm0        # 48a0 <_sk_callback_hsw+0x4b1>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,13,6,0,0           // vbroadcastss  0x60d(%rip),%ymm10        # 4894 <_sk_callback_hsw+0x4b9>
-  .byte  196,98,125,24,29,8,6,0,0            // vbroadcastss  0x608(%rip),%ymm11        # 4898 <_sk_callback_hsw+0x4bd>
+  .byte  196,98,125,24,21,9,6,0,0            // vbroadcastss  0x609(%rip),%ymm10        # 48a4 <_sk_callback_hsw+0x4b5>
+  .byte  196,98,125,24,29,4,6,0,0            // vbroadcastss  0x604(%rip),%ymm11        # 48a8 <_sk_callback_hsw+0x4b9>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,52,89,195                    // vmulps        %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -12648,13 +12670,13 @@ HIDDEN _sk_bicubic_n3y_hsw
 FUNCTION(_sk_bicubic_n3y_hsw)
 _sk_bicubic_n3y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,235,5,0,0         // vbroadcastss  0x5eb(%rip),%ymm1        # 489c <_sk_callback_hsw+0x4c1>
+  .byte  196,226,125,24,13,231,5,0,0         // vbroadcastss  0x5e7(%rip),%ymm1        # 48ac <_sk_callback_hsw+0x4bd>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,225,5,0,0           // vbroadcastss  0x5e1(%rip),%ymm8        # 48a0 <_sk_callback_hsw+0x4c5>
+  .byte  196,98,125,24,5,221,5,0,0           // vbroadcastss  0x5dd(%rip),%ymm8        # 48b0 <_sk_callback_hsw+0x4c1>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,210,5,0,0          // vbroadcastss  0x5d2(%rip),%ymm10        # 48a4 <_sk_callback_hsw+0x4c9>
-  .byte  196,98,125,24,29,205,5,0,0          // vbroadcastss  0x5cd(%rip),%ymm11        # 48a8 <_sk_callback_hsw+0x4cd>
+  .byte  196,98,125,24,21,206,5,0,0          // vbroadcastss  0x5ce(%rip),%ymm10        # 48b4 <_sk_callback_hsw+0x4c5>
+  .byte  196,98,125,24,29,201,5,0,0          // vbroadcastss  0x5c9(%rip),%ymm11        # 48b8 <_sk_callback_hsw+0x4c9>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,36,89,193                    // vmulps        %ymm9,%ymm11,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -12666,16 +12688,16 @@ HIDDEN _sk_bicubic_n1y_hsw
 FUNCTION(_sk_bicubic_n1y_hsw)
 _sk_bicubic_n1y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,176,5,0,0         // vbroadcastss  0x5b0(%rip),%ymm1        # 48ac <_sk_callback_hsw+0x4d1>
+  .byte  196,226,125,24,13,172,5,0,0         // vbroadcastss  0x5ac(%rip),%ymm1        # 48bc <_sk_callback_hsw+0x4cd>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,166,5,0,0           // vbroadcastss  0x5a6(%rip),%ymm8        # 48b0 <_sk_callback_hsw+0x4d5>
+  .byte  196,98,125,24,5,162,5,0,0           // vbroadcastss  0x5a2(%rip),%ymm8        # 48c0 <_sk_callback_hsw+0x4d1>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,156,5,0,0          // vbroadcastss  0x59c(%rip),%ymm9        # 48b4 <_sk_callback_hsw+0x4d9>
-  .byte  196,98,125,24,21,151,5,0,0          // vbroadcastss  0x597(%rip),%ymm10        # 48b8 <_sk_callback_hsw+0x4dd>
+  .byte  196,98,125,24,13,152,5,0,0          // vbroadcastss  0x598(%rip),%ymm9        # 48c4 <_sk_callback_hsw+0x4d5>
+  .byte  196,98,125,24,21,147,5,0,0          // vbroadcastss  0x593(%rip),%ymm10        # 48c8 <_sk_callback_hsw+0x4d9>
   .byte  196,66,61,168,209                   // vfmadd213ps   %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,13,141,5,0,0          // vbroadcastss  0x58d(%rip),%ymm9        # 48bc <_sk_callback_hsw+0x4e1>
+  .byte  196,98,125,24,13,137,5,0,0          // vbroadcastss  0x589(%rip),%ymm9        # 48cc <_sk_callback_hsw+0x4dd>
   .byte  196,66,61,184,202                   // vfmadd231ps   %ymm10,%ymm8,%ymm9
-  .byte  196,98,125,24,21,131,5,0,0          // vbroadcastss  0x583(%rip),%ymm10        # 48c0 <_sk_callback_hsw+0x4e5>
+  .byte  196,98,125,24,21,127,5,0,0          // vbroadcastss  0x57f(%rip),%ymm10        # 48d0 <_sk_callback_hsw+0x4e1>
   .byte  196,66,61,184,209                   // vfmadd231ps   %ymm9,%ymm8,%ymm10
   .byte  197,124,17,144,160,0,0,0            // vmovups       %ymm10,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12686,14 +12708,14 @@ HIDDEN _sk_bicubic_p1y_hsw
 FUNCTION(_sk_bicubic_p1y_hsw)
 _sk_bicubic_p1y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,107,5,0,0           // vbroadcastss  0x56b(%rip),%ymm8        # 48c4 <_sk_callback_hsw+0x4e9>
+  .byte  196,98,125,24,5,103,5,0,0           // vbroadcastss  0x567(%rip),%ymm8        # 48d4 <_sk_callback_hsw+0x4e5>
   .byte  197,188,88,72,32                    // vaddps        0x20(%rax),%ymm8,%ymm1
   .byte  197,124,16,72,96                    // vmovups       0x60(%rax),%ymm9
-  .byte  196,98,125,24,21,92,5,0,0           // vbroadcastss  0x55c(%rip),%ymm10        # 48c8 <_sk_callback_hsw+0x4ed>
-  .byte  196,98,125,24,29,87,5,0,0           // vbroadcastss  0x557(%rip),%ymm11        # 48cc <_sk_callback_hsw+0x4f1>
+  .byte  196,98,125,24,21,88,5,0,0           // vbroadcastss  0x558(%rip),%ymm10        # 48d8 <_sk_callback_hsw+0x4e9>
+  .byte  196,98,125,24,29,83,5,0,0           // vbroadcastss  0x553(%rip),%ymm11        # 48dc <_sk_callback_hsw+0x4ed>
   .byte  196,66,53,168,218                   // vfmadd213ps   %ymm10,%ymm9,%ymm11
   .byte  196,66,53,168,216                   // vfmadd213ps   %ymm8,%ymm9,%ymm11
-  .byte  196,98,125,24,5,72,5,0,0            // vbroadcastss  0x548(%rip),%ymm8        # 48d0 <_sk_callback_hsw+0x4f5>
+  .byte  196,98,125,24,5,68,5,0,0            // vbroadcastss  0x544(%rip),%ymm8        # 48e0 <_sk_callback_hsw+0x4f1>
   .byte  196,66,53,184,195                   // vfmadd231ps   %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12704,12 +12726,12 @@ HIDDEN _sk_bicubic_p3y_hsw
 FUNCTION(_sk_bicubic_p3y_hsw)
 _sk_bicubic_p3y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,48,5,0,0          // vbroadcastss  0x530(%rip),%ymm1        # 48d4 <_sk_callback_hsw+0x4f9>
+  .byte  196,226,125,24,13,44,5,0,0          // vbroadcastss  0x52c(%rip),%ymm1        # 48e4 <_sk_callback_hsw+0x4f5>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,28,5,0,0           // vbroadcastss  0x51c(%rip),%ymm10        # 48d8 <_sk_callback_hsw+0x4fd>
-  .byte  196,98,125,24,29,23,5,0,0           // vbroadcastss  0x517(%rip),%ymm11        # 48dc <_sk_callback_hsw+0x501>
+  .byte  196,98,125,24,21,24,5,0,0           // vbroadcastss  0x518(%rip),%ymm10        # 48e8 <_sk_callback_hsw+0x4f9>
+  .byte  196,98,125,24,29,19,5,0,0           // vbroadcastss  0x513(%rip),%ymm11        # 48ec <_sk_callback_hsw+0x4fd>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,52,89,195                    // vmulps        %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -12834,25 +12856,25 @@ BALIGN4
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 45b5 <.literal4+0xb5>
+  .byte  71,225,61                           // rex.RXB       loope 45c9 <.literal4+0xb5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 45c5 <.literal4+0xc5>
+  .byte  71,225,61                           // rex.RXB       loope 45d9 <.literal4+0xc5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 45d5 <.literal4+0xd5>
+  .byte  71,225,61                           // rex.RXB       loope 45e9 <.literal4+0xd5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 45e5 <.literal4+0xe5>
+  .byte  71,225,61                           // rex.RXB       loope 45f9 <.literal4+0xe5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -12901,24 +12923,26 @@ BALIGN4
   .byte  190,129,128,128,59                  // mov           $0x3b808081,%esi
   .byte  129,128,128,59,0,248,0,0,8,33       // addl          $0x21080000,-0x7ffc480(%rax)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4631 <.literal4+0x131>
+  .byte  224,7                               // loopne        4645 <.literal4+0x131>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
   .byte  31                                  // (bad)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,8                                 // add           %cl,(%rax)
-  .byte  33,4,61,0,0,128,63                  // and           %eax,0x3f800000(,%rdi,1)
-  .byte  129,128,128,59,128,0,128,55,0,0     // addl          $0x3780,0x803b80(%rax)
+  .byte  33,4,61,129,128,128,59              // and           %eax,0x3b808081(,%rdi,1)
+  .byte  128,0,128                           // addb          $0x80,(%rax)
+  .byte  55                                  // (bad)
+  .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
   .byte  0,52,255                            // add           %dh,(%rdi,%rdi,8)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            465c <.literal4+0x15c>
+  .byte  127,0                               // jg            466c <.literal4+0x158>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            46d5 <.literal4+0x1d5>
+  .byte  119,115                             // ja            46e5 <.literal4+0x1d1>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -12932,10 +12956,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4690 <.literal4+0x190>
+  .byte  127,0                               // jg            46a0 <.literal4+0x18c>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4709 <.literal4+0x209>
+  .byte  119,115                             // ja            4719 <.literal4+0x205>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -12949,10 +12973,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            46c4 <.literal4+0x1c4>
+  .byte  127,0                               // jg            46d4 <.literal4+0x1c0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            473d <.literal4+0x23d>
+  .byte  119,115                             // ja            474d <.literal4+0x239>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -12966,10 +12990,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            46f8 <.literal4+0x1f8>
+  .byte  127,0                               // jg            4708 <.literal4+0x1f4>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4771 <.literal4+0x271>
+  .byte  119,115                             // ja            4781 <.literal4+0x26d>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -12982,7 +13006,7 @@ BALIGN4
   .byte  0,75,0                              // add           %cl,0x0(%rbx)
   .byte  0,128,63,0,0,200                    // add           %al,-0x37ffffc1(%rax)
   .byte  66,0,0                              // rex.X         add %al,(%rax)
-  .byte  127,67                              // jg            476f <.literal4+0x26f>
+  .byte  127,67                              // jg            477f <.literal4+0x26b>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -12994,10 +13018,10 @@ BALIGN4
   .byte  190,80,128,3,62                     // mov           $0x3e038050,%esi
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           478f <.literal4+0x28f>
+  .byte  118,63                              // jbe           479f <.literal4+0x28b>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            47a3 <.literal4+0x2a3>
+  .byte  127,67                              // jg            47b3 <.literal4+0x29f>
   .byte  129,128,128,59,0,0,128,63,129,128   // addl          $0x80813f80,0x3b80(%rax)
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,128,63,129,128,128                // add           %al,-0x7f7f7ec1(%rax)
@@ -13006,7 +13030,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4785 <.literal4+0x285>
+  .byte  224,7                               // loopne        4795 <.literal4+0x281>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -13018,7 +13042,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        47a1 <.literal4+0x2a1>
+  .byte  224,7                               // loopne        47b1 <.literal4+0x29d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -13029,7 +13053,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            47f6 <.literal4+0x2f6>
+  .byte  124,66                              // jl            4806 <.literal4+0x2f2>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,55,0,15                 // mov           %ecx,0xf003788(%rax)
@@ -13047,9 +13071,9 @@ BALIGN4
   .byte  137,136,136,59,15,0                 // mov           %ecx,0xf3b88(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,61,0,0                  // mov           %ecx,0x3d88(%rax)
-  .byte  112,65                              // jo            4839 <.literal4+0x339>
+  .byte  112,65                              // jo            4849 <.literal4+0x335>
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            4847 <.literal4+0x347>
+  .byte  127,67                              // jg            4857 <.literal4+0x343>
   .byte  128,0,128                           // addb          $0x80,(%rax)
   .byte  55                                  // (bad)
   .byte  128,0,128                           // addb          $0x80,(%rax)
@@ -13057,7 +13081,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  255                                 // (bad)
-  .byte  127,71                              // jg            485b <.literal4+0x35b>
+  .byte  127,71                              // jg            486b <.literal4+0x357>
   .byte  208                                 // (bad)
   .byte  179,89                              // mov           $0x59,%bl
   .byte  62,89                               // ds            pop %rcx
@@ -13154,16 +13178,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004908 <_sk_callback_hsw+0xa00052d>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004928 <_sk_callback_hsw+0xa000539>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004910 <_sk_callback_hsw+0x12000535>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004930 <_sk_callback_hsw+0x12000541>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004918 <_sk_callback_hsw+0x1a00053d>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004938 <_sk_callback_hsw+0x1a000549>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004920 <_sk_callback_hsw+0x3000545>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004940 <_sk_callback_hsw+0x3000551>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13206,16 +13230,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004968 <_sk_callback_hsw+0xa00058d>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004988 <_sk_callback_hsw+0xa000599>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004970 <_sk_callback_hsw+0x12000595>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004990 <_sk_callback_hsw+0x120005a1>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004978 <_sk_callback_hsw+0x1a00059d>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004998 <_sk_callback_hsw+0x1a0005a9>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004980 <_sk_callback_hsw+0x30005a5>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 30049a0 <_sk_callback_hsw+0x30005b1>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13258,16 +13282,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a0049c8 <_sk_callback_hsw+0xa0005ed>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a0049e8 <_sk_callback_hsw+0xa0005f9>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 120049d0 <_sk_callback_hsw+0x120005f5>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 120049f0 <_sk_callback_hsw+0x12000601>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a0049d8 <_sk_callback_hsw+0x1a0005fd>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a0049f8 <_sk_callback_hsw+0x1a000609>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 30049e0 <_sk_callback_hsw+0x3000605>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004a00 <_sk_callback_hsw+0x3000611>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13310,16 +13334,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004a28 <_sk_callback_hsw+0xa00064d>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004a48 <_sk_callback_hsw+0xa000659>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004a30 <_sk_callback_hsw+0x12000655>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004a50 <_sk_callback_hsw+0x12000661>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004a38 <_sk_callback_hsw+0x1a00065d>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004a58 <_sk_callback_hsw+0x1a000669>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004a40 <_sk_callback_hsw+0x3000665>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004a60 <_sk_callback_hsw+0x3000671>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13440,14 +13464,14 @@ _sk_seed_shader_avx:
   .byte  197,249,112,192,0                   // vpshufd       $0x0,%xmm0,%xmm0
   .byte  196,227,125,24,192,1                // vinsertf128   $0x1,%xmm0,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,219,91,0,0        // vbroadcastss  0x5bdb(%rip),%ymm1        # 5ca4 <_sk_callback_avx+0x128>
+  .byte  196,226,125,24,13,251,91,0,0        // vbroadcastss  0x5bfb(%rip),%ymm1        # 5cc4 <_sk_callback_avx+0x128>
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
   .byte  197,252,88,2                        // vaddps        (%rdx),%ymm0,%ymm0
   .byte  196,226,125,24,16                   // vbroadcastss  (%rax),%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  197,236,88,201                      // vaddps        %ymm1,%ymm2,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,21,191,91,0,0        // vbroadcastss  0x5bbf(%rip),%ymm2        # 5ca8 <_sk_callback_avx+0x12c>
+  .byte  196,226,125,24,21,223,91,0,0        // vbroadcastss  0x5bdf(%rip),%ymm2        # 5cc8 <_sk_callback_avx+0x12c>
   .byte  197,228,87,219                      // vxorps        %ymm3,%ymm3,%ymm3
   .byte  197,220,87,228                      // vxorps        %ymm4,%ymm4,%ymm4
   .byte  197,212,87,237                      // vxorps        %ymm5,%ymm5,%ymm5
@@ -13469,7 +13493,7 @@ _sk_dither_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  196,66,125,24,8                     // vbroadcastss  (%r8),%ymm9
   .byte  196,65,60,87,209                    // vxorps        %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,29,119,91,0,0         // vbroadcastss  0x5b77(%rip),%ymm11        # 5cac <_sk_callback_avx+0x130>
+  .byte  196,98,125,24,29,151,91,0,0         // vbroadcastss  0x5b97(%rip),%ymm11        # 5ccc <_sk_callback_avx+0x130>
   .byte  196,65,44,84,203                    // vandps        %ymm11,%ymm10,%ymm9
   .byte  196,193,25,114,241,5                // vpslld        $0x5,%xmm9,%xmm12
   .byte  196,67,125,25,201,1                 // vextractf128  $0x1,%ymm9,%xmm9
@@ -13480,8 +13504,8 @@ _sk_dither_avx:
   .byte  196,67,125,25,219,1                 // vextractf128  $0x1,%ymm11,%xmm11
   .byte  196,193,33,114,243,4                // vpslld        $0x4,%xmm11,%xmm11
   .byte  196,67,29,24,219,1                  // vinsertf128   $0x1,%xmm11,%ymm12,%ymm11
-  .byte  196,98,125,24,37,56,91,0,0          // vbroadcastss  0x5b38(%rip),%ymm12        # 5cb0 <_sk_callback_avx+0x134>
-  .byte  196,98,125,24,45,51,91,0,0          // vbroadcastss  0x5b33(%rip),%ymm13        # 5cb4 <_sk_callback_avx+0x138>
+  .byte  196,98,125,24,37,88,91,0,0          // vbroadcastss  0x5b58(%rip),%ymm12        # 5cd0 <_sk_callback_avx+0x134>
+  .byte  196,98,125,24,45,83,91,0,0          // vbroadcastss  0x5b53(%rip),%ymm13        # 5cd4 <_sk_callback_avx+0x138>
   .byte  196,65,44,84,245                    // vandps        %ymm13,%ymm10,%ymm14
   .byte  196,193,1,114,246,2                 // vpslld        $0x2,%xmm14,%xmm15
   .byte  196,67,125,25,246,1                 // vextractf128  $0x1,%ymm14,%xmm14
@@ -13508,9 +13532,9 @@ _sk_dither_avx:
   .byte  196,65,60,86,193                    // vorps         %ymm9,%ymm8,%ymm8
   .byte  196,65,60,86,194                    // vorps         %ymm10,%ymm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,158,90,0,0         // vbroadcastss  0x5a9e(%rip),%ymm9        # 5cb8 <_sk_callback_avx+0x13c>
+  .byte  196,98,125,24,13,190,90,0,0         // vbroadcastss  0x5abe(%rip),%ymm9        # 5cd8 <_sk_callback_avx+0x13c>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,148,90,0,0         // vbroadcastss  0x5a94(%rip),%ymm9        # 5cbc <_sk_callback_avx+0x140>
+  .byte  196,98,125,24,13,180,90,0,0         // vbroadcastss  0x5ab4(%rip),%ymm9        # 5cdc <_sk_callback_avx+0x140>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  196,98,125,24,72,8                  // vbroadcastss  0x8(%rax),%ymm9
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
@@ -13548,7 +13572,7 @@ HIDDEN _sk_srcatop_avx
 FUNCTION(_sk_srcatop_avx)
 _sk_srcatop_avx:
   .byte  197,252,89,199                      // vmulps        %ymm7,%ymm0,%ymm0
-  .byte  196,98,125,24,5,58,90,0,0           // vbroadcastss  0x5a3a(%rip),%ymm8        # 5cc0 <_sk_callback_avx+0x144>
+  .byte  196,98,125,24,5,90,90,0,0           // vbroadcastss  0x5a5a(%rip),%ymm8        # 5ce0 <_sk_callback_avx+0x144>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,204                       // vmulps        %ymm4,%ymm8,%ymm9
   .byte  197,180,88,192                      // vaddps        %ymm0,%ymm9,%ymm0
@@ -13569,7 +13593,7 @@ HIDDEN _sk_dstatop_avx
 FUNCTION(_sk_dstatop_avx)
 _sk_dstatop_avx:
   .byte  197,100,89,196                      // vmulps        %ymm4,%ymm3,%ymm8
-  .byte  196,98,125,24,13,252,89,0,0         // vbroadcastss  0x59fc(%rip),%ymm9        # 5cc4 <_sk_callback_avx+0x148>
+  .byte  196,98,125,24,13,28,90,0,0          // vbroadcastss  0x5a1c(%rip),%ymm9        # 5ce4 <_sk_callback_avx+0x148>
   .byte  197,52,92,207                       // vsubps        %ymm7,%ymm9,%ymm9
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
   .byte  197,188,88,192                      // vaddps        %ymm0,%ymm8,%ymm0
@@ -13611,7 +13635,7 @@ HIDDEN _sk_srcout_avx
 .globl _sk_srcout_avx
 FUNCTION(_sk_srcout_avx)
 _sk_srcout_avx:
-  .byte  196,98,125,24,5,155,89,0,0          // vbroadcastss  0x599b(%rip),%ymm8        # 5cc8 <_sk_callback_avx+0x14c>
+  .byte  196,98,125,24,5,187,89,0,0          // vbroadcastss  0x59bb(%rip),%ymm8        # 5ce8 <_sk_callback_avx+0x14c>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -13624,7 +13648,7 @@ HIDDEN _sk_dstout_avx
 .globl _sk_dstout_avx
 FUNCTION(_sk_dstout_avx)
 _sk_dstout_avx:
-  .byte  196,226,125,24,5,126,89,0,0         // vbroadcastss  0x597e(%rip),%ymm0        # 5ccc <_sk_callback_avx+0x150>
+  .byte  196,226,125,24,5,158,89,0,0         // vbroadcastss  0x599e(%rip),%ymm0        # 5cec <_sk_callback_avx+0x150>
   .byte  197,252,92,219                      // vsubps        %ymm3,%ymm0,%ymm3
   .byte  197,228,89,196                      // vmulps        %ymm4,%ymm3,%ymm0
   .byte  197,228,89,205                      // vmulps        %ymm5,%ymm3,%ymm1
@@ -13637,7 +13661,7 @@ HIDDEN _sk_srcover_avx
 .globl _sk_srcover_avx
 FUNCTION(_sk_srcover_avx)
 _sk_srcover_avx:
-  .byte  196,98,125,24,5,97,89,0,0           // vbroadcastss  0x5961(%rip),%ymm8        # 5cd0 <_sk_callback_avx+0x154>
+  .byte  196,98,125,24,5,129,89,0,0          // vbroadcastss  0x5981(%rip),%ymm8        # 5cf0 <_sk_callback_avx+0x154>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,204                       // vmulps        %ymm4,%ymm8,%ymm9
   .byte  197,180,88,192                      // vaddps        %ymm0,%ymm9,%ymm0
@@ -13654,7 +13678,7 @@ HIDDEN _sk_dstover_avx
 .globl _sk_dstover_avx
 FUNCTION(_sk_dstover_avx)
 _sk_dstover_avx:
-  .byte  196,98,125,24,5,52,89,0,0           // vbroadcastss  0x5934(%rip),%ymm8        # 5cd4 <_sk_callback_avx+0x158>
+  .byte  196,98,125,24,5,84,89,0,0           // vbroadcastss  0x5954(%rip),%ymm8        # 5cf4 <_sk_callback_avx+0x158>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,252,88,196                      // vaddps        %ymm4,%ymm0,%ymm0
@@ -13682,7 +13706,7 @@ HIDDEN _sk_multiply_avx
 .globl _sk_multiply_avx
 FUNCTION(_sk_multiply_avx)
 _sk_multiply_avx:
-  .byte  196,98,125,24,5,243,88,0,0          // vbroadcastss  0x58f3(%rip),%ymm8        # 5cd8 <_sk_callback_avx+0x15c>
+  .byte  196,98,125,24,5,19,89,0,0           // vbroadcastss  0x5913(%rip),%ymm8        # 5cf8 <_sk_callback_avx+0x15c>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,208                       // vmulps        %ymm0,%ymm9,%ymm10
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -13742,7 +13766,7 @@ HIDDEN _sk_xor__avx
 .globl _sk_xor__avx
 FUNCTION(_sk_xor__avx)
 _sk_xor__avx:
-  .byte  196,98,125,24,5,66,88,0,0           // vbroadcastss  0x5842(%rip),%ymm8        # 5cdc <_sk_callback_avx+0x160>
+  .byte  196,98,125,24,5,98,88,0,0           // vbroadcastss  0x5862(%rip),%ymm8        # 5cfc <_sk_callback_avx+0x160>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -13779,7 +13803,7 @@ _sk_darken_avx:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,95,209                  // vmaxps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,194,87,0,0          // vbroadcastss  0x57c2(%rip),%ymm8        # 5ce0 <_sk_callback_avx+0x164>
+  .byte  196,98,125,24,5,226,87,0,0          // vbroadcastss  0x57e2(%rip),%ymm8        # 5d00 <_sk_callback_avx+0x164>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -13805,7 +13829,7 @@ _sk_lighten_avx:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,110,87,0,0          // vbroadcastss  0x576e(%rip),%ymm8        # 5ce4 <_sk_callback_avx+0x168>
+  .byte  196,98,125,24,5,142,87,0,0          // vbroadcastss  0x578e(%rip),%ymm8        # 5d04 <_sk_callback_avx+0x168>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -13834,7 +13858,7 @@ _sk_difference_avx:
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,14,87,0,0           // vbroadcastss  0x570e(%rip),%ymm8        # 5ce8 <_sk_callback_avx+0x16c>
+  .byte  196,98,125,24,5,46,87,0,0           // vbroadcastss  0x572e(%rip),%ymm8        # 5d08 <_sk_callback_avx+0x16c>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -13857,7 +13881,7 @@ _sk_exclusion_avx:
   .byte  197,236,89,214                      // vmulps        %ymm6,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,201,86,0,0          // vbroadcastss  0x56c9(%rip),%ymm8        # 5cec <_sk_callback_avx+0x170>
+  .byte  196,98,125,24,5,233,86,0,0          // vbroadcastss  0x56e9(%rip),%ymm8        # 5d0c <_sk_callback_avx+0x170>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -13868,7 +13892,7 @@ HIDDEN _sk_colorburn_avx
 .globl _sk_colorburn_avx
 FUNCTION(_sk_colorburn_avx)
 _sk_colorburn_avx:
-  .byte  196,98,125,24,5,180,86,0,0          // vbroadcastss  0x56b4(%rip),%ymm8        # 5cf0 <_sk_callback_avx+0x174>
+  .byte  196,98,125,24,5,212,86,0,0          // vbroadcastss  0x56d4(%rip),%ymm8        # 5d10 <_sk_callback_avx+0x174>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,216                       // vmulps        %ymm0,%ymm9,%ymm11
   .byte  196,65,44,87,210                    // vxorps        %ymm10,%ymm10,%ymm10
@@ -13930,7 +13954,7 @@ HIDDEN _sk_colordodge_avx
 FUNCTION(_sk_colordodge_avx)
 _sk_colordodge_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
-  .byte  196,98,125,24,13,176,85,0,0         // vbroadcastss  0x55b0(%rip),%ymm9        # 5cf4 <_sk_callback_avx+0x178>
+  .byte  196,98,125,24,13,208,85,0,0         // vbroadcastss  0x55d0(%rip),%ymm9        # 5d14 <_sk_callback_avx+0x178>
   .byte  197,52,92,215                       // vsubps        %ymm7,%ymm9,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,52,92,203                       // vsubps        %ymm3,%ymm9,%ymm9
@@ -13987,7 +14011,7 @@ HIDDEN _sk_hardlight_avx
 .globl _sk_hardlight_avx
 FUNCTION(_sk_hardlight_avx)
 _sk_hardlight_avx:
-  .byte  196,98,125,24,5,194,84,0,0          // vbroadcastss  0x54c2(%rip),%ymm8        # 5cf8 <_sk_callback_avx+0x17c>
+  .byte  196,98,125,24,5,226,84,0,0          // vbroadcastss  0x54e2(%rip),%ymm8        # 5d18 <_sk_callback_avx+0x17c>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,200                       // vmulps        %ymm0,%ymm10,%ymm9
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14042,7 +14066,7 @@ HIDDEN _sk_overlay_avx
 .globl _sk_overlay_avx
 FUNCTION(_sk_overlay_avx)
 _sk_overlay_avx:
-  .byte  196,98,125,24,5,235,83,0,0          // vbroadcastss  0x53eb(%rip),%ymm8        # 5cfc <_sk_callback_avx+0x180>
+  .byte  196,98,125,24,5,11,84,0,0           // vbroadcastss  0x540b(%rip),%ymm8        # 5d1c <_sk_callback_avx+0x180>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,200                       // vmulps        %ymm0,%ymm10,%ymm9
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14108,10 +14132,10 @@ _sk_softlight_avx:
   .byte  196,65,60,88,192                    // vaddps        %ymm8,%ymm8,%ymm8
   .byte  196,65,60,89,216                    // vmulps        %ymm8,%ymm8,%ymm11
   .byte  196,65,60,88,195                    // vaddps        %ymm11,%ymm8,%ymm8
-  .byte  196,98,125,24,29,226,82,0,0         // vbroadcastss  0x52e2(%rip),%ymm11        # 5d04 <_sk_callback_avx+0x188>
+  .byte  196,98,125,24,29,2,83,0,0           // vbroadcastss  0x5302(%rip),%ymm11        # 5d24 <_sk_callback_avx+0x188>
   .byte  196,65,28,88,235                    // vaddps        %ymm11,%ymm12,%ymm13
   .byte  196,65,20,89,192                    // vmulps        %ymm8,%ymm13,%ymm8
-  .byte  196,98,125,24,45,211,82,0,0         // vbroadcastss  0x52d3(%rip),%ymm13        # 5d08 <_sk_callback_avx+0x18c>
+  .byte  196,98,125,24,45,243,82,0,0         // vbroadcastss  0x52f3(%rip),%ymm13        # 5d28 <_sk_callback_avx+0x18c>
   .byte  196,65,28,89,245                    // vmulps        %ymm13,%ymm12,%ymm14
   .byte  196,65,12,88,192                    // vaddps        %ymm8,%ymm14,%ymm8
   .byte  196,65,124,82,244                   // vrsqrtps      %ymm12,%ymm14
@@ -14122,7 +14146,7 @@ _sk_softlight_avx:
   .byte  197,4,194,255,2                     // vcmpleps      %ymm7,%ymm15,%ymm15
   .byte  196,67,13,74,240,240                // vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   .byte  197,116,88,249                      // vaddps        %ymm1,%ymm1,%ymm15
-  .byte  196,98,125,24,5,145,82,0,0          // vbroadcastss  0x5291(%rip),%ymm8        # 5d00 <_sk_callback_avx+0x184>
+  .byte  196,98,125,24,5,177,82,0,0          // vbroadcastss  0x52b1(%rip),%ymm8        # 5d20 <_sk_callback_avx+0x184>
   .byte  196,65,60,92,228                    // vsubps        %ymm12,%ymm8,%ymm12
   .byte  197,132,92,195                      // vsubps        %ymm3,%ymm15,%ymm0
   .byte  196,65,124,89,228                   // vmulps        %ymm12,%ymm0,%ymm12
@@ -14219,7 +14243,7 @@ FUNCTION(_sk_hue_avx)
 _sk_hue_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,208,0                // vcmpeqps      %ymm8,%ymm3,%ymm10
-  .byte  196,98,125,24,13,243,80,0,0         // vbroadcastss  0x50f3(%rip),%ymm9        # 5d0c <_sk_callback_avx+0x190>
+  .byte  196,98,125,24,13,19,81,0,0          // vbroadcastss  0x5113(%rip),%ymm9        # 5d2c <_sk_callback_avx+0x190>
   .byte  197,52,94,219                       // vdivps        %ymm3,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,172,89,192                      // vmulps        %ymm0,%ymm10,%ymm0
@@ -14248,12 +14272,12 @@ _sk_hue_avx:
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  196,193,108,94,212                  // vdivps        %ymm12,%ymm2,%ymm2
   .byte  196,195,109,74,208,208              // vblendvps     %ymm13,%ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,21,104,80,0,0         // vbroadcastss  0x5068(%rip),%ymm10        # 5d10 <_sk_callback_avx+0x194>
+  .byte  196,98,125,24,21,136,80,0,0         // vbroadcastss  0x5088(%rip),%ymm10        # 5d30 <_sk_callback_avx+0x194>
   .byte  196,65,92,89,218                    // vmulps        %ymm10,%ymm4,%ymm11
-  .byte  196,98,125,24,37,94,80,0,0          // vbroadcastss  0x505e(%rip),%ymm12        # 5d14 <_sk_callback_avx+0x198>
+  .byte  196,98,125,24,37,126,80,0,0         // vbroadcastss  0x507e(%rip),%ymm12        # 5d34 <_sk_callback_avx+0x198>
   .byte  196,65,84,89,236                    // vmulps        %ymm12,%ymm5,%ymm13
   .byte  196,65,36,88,221                    // vaddps        %ymm13,%ymm11,%ymm11
-  .byte  196,98,125,24,45,79,80,0,0          // vbroadcastss  0x504f(%rip),%ymm13        # 5d18 <_sk_callback_avx+0x19c>
+  .byte  196,98,125,24,45,111,80,0,0         // vbroadcastss  0x506f(%rip),%ymm13        # 5d38 <_sk_callback_avx+0x19c>
   .byte  196,65,76,89,245                    // vmulps        %ymm13,%ymm6,%ymm14
   .byte  196,65,36,88,222                    // vaddps        %ymm14,%ymm11,%ymm11
   .byte  196,65,124,89,242                   // vmulps        %ymm10,%ymm0,%ymm14
@@ -14327,7 +14351,7 @@ FUNCTION(_sk_saturation_avx)
 _sk_saturation_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,68,194,208,0                 // vcmpeqps      %ymm8,%ymm7,%ymm10
-  .byte  196,98,125,24,13,12,79,0,0          // vbroadcastss  0x4f0c(%rip),%ymm9        # 5d1c <_sk_callback_avx+0x1a0>
+  .byte  196,98,125,24,13,44,79,0,0          // vbroadcastss  0x4f2c(%rip),%ymm9        # 5d3c <_sk_callback_avx+0x1a0>
   .byte  197,52,94,223                       // vdivps        %ymm7,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,44,89,220                       // vmulps        %ymm4,%ymm10,%ymm11
@@ -14356,12 +14380,12 @@ _sk_saturation_avx:
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  197,252,94,194                      // vdivps        %ymm2,%ymm0,%ymm0
   .byte  196,195,125,74,192,208              // vblendvps     %ymm13,%ymm8,%ymm0,%ymm0
-  .byte  196,226,125,24,13,136,78,0,0        // vbroadcastss  0x4e88(%rip),%ymm1        # 5d20 <_sk_callback_avx+0x1a4>
+  .byte  196,226,125,24,13,168,78,0,0        // vbroadcastss  0x4ea8(%rip),%ymm1        # 5d40 <_sk_callback_avx+0x1a4>
   .byte  197,220,89,209                      // vmulps        %ymm1,%ymm4,%ymm2
-  .byte  196,98,125,24,21,127,78,0,0         // vbroadcastss  0x4e7f(%rip),%ymm10        # 5d24 <_sk_callback_avx+0x1a8>
+  .byte  196,98,125,24,21,159,78,0,0         // vbroadcastss  0x4e9f(%rip),%ymm10        # 5d44 <_sk_callback_avx+0x1a8>
   .byte  196,65,84,89,234                    // vmulps        %ymm10,%ymm5,%ymm13
   .byte  196,193,108,88,213                  // vaddps        %ymm13,%ymm2,%ymm2
-  .byte  196,98,125,24,45,112,78,0,0         // vbroadcastss  0x4e70(%rip),%ymm13        # 5d28 <_sk_callback_avx+0x1ac>
+  .byte  196,98,125,24,45,144,78,0,0         // vbroadcastss  0x4e90(%rip),%ymm13        # 5d48 <_sk_callback_avx+0x1ac>
   .byte  196,65,76,89,245                    // vmulps        %ymm13,%ymm6,%ymm14
   .byte  196,193,108,88,214                  // vaddps        %ymm14,%ymm2,%ymm2
   .byte  197,36,89,241                       // vmulps        %ymm1,%ymm11,%ymm14
@@ -14435,18 +14459,18 @@ FUNCTION(_sk_color_avx)
 _sk_color_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,208,0                // vcmpeqps      %ymm8,%ymm3,%ymm10
-  .byte  196,98,125,24,13,49,77,0,0          // vbroadcastss  0x4d31(%rip),%ymm9        # 5d2c <_sk_callback_avx+0x1b0>
+  .byte  196,98,125,24,13,81,77,0,0          // vbroadcastss  0x4d51(%rip),%ymm9        # 5d4c <_sk_callback_avx+0x1b0>
   .byte  197,52,94,219                       // vdivps        %ymm3,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,172,89,192                      // vmulps        %ymm0,%ymm10,%ymm0
   .byte  197,172,89,201                      // vmulps        %ymm1,%ymm10,%ymm1
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
-  .byte  196,98,125,24,21,22,77,0,0          // vbroadcastss  0x4d16(%rip),%ymm10        # 5d30 <_sk_callback_avx+0x1b4>
+  .byte  196,98,125,24,21,54,77,0,0          // vbroadcastss  0x4d36(%rip),%ymm10        # 5d50 <_sk_callback_avx+0x1b4>
   .byte  196,65,92,89,218                    // vmulps        %ymm10,%ymm4,%ymm11
-  .byte  196,98,125,24,37,12,77,0,0          // vbroadcastss  0x4d0c(%rip),%ymm12        # 5d34 <_sk_callback_avx+0x1b8>
+  .byte  196,98,125,24,37,44,77,0,0          // vbroadcastss  0x4d2c(%rip),%ymm12        # 5d54 <_sk_callback_avx+0x1b8>
   .byte  196,65,84,89,236                    // vmulps        %ymm12,%ymm5,%ymm13
   .byte  196,65,36,88,221                    // vaddps        %ymm13,%ymm11,%ymm11
-  .byte  196,98,125,24,45,253,76,0,0         // vbroadcastss  0x4cfd(%rip),%ymm13        # 5d38 <_sk_callback_avx+0x1bc>
+  .byte  196,98,125,24,45,29,77,0,0          // vbroadcastss  0x4d1d(%rip),%ymm13        # 5d58 <_sk_callback_avx+0x1bc>
   .byte  196,65,76,89,245                    // vmulps        %ymm13,%ymm6,%ymm14
   .byte  196,65,36,88,222                    // vaddps        %ymm14,%ymm11,%ymm11
   .byte  196,65,124,89,242                   // vmulps        %ymm10,%ymm0,%ymm14
@@ -14520,18 +14544,18 @@ FUNCTION(_sk_luminosity_avx)
 _sk_luminosity_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,68,194,208,0                 // vcmpeqps      %ymm8,%ymm7,%ymm10
-  .byte  196,98,125,24,13,186,75,0,0         // vbroadcastss  0x4bba(%rip),%ymm9        # 5d3c <_sk_callback_avx+0x1c0>
+  .byte  196,98,125,24,13,218,75,0,0         // vbroadcastss  0x4bda(%rip),%ymm9        # 5d5c <_sk_callback_avx+0x1c0>
   .byte  197,52,94,223                       // vdivps        %ymm7,%ymm9,%ymm11
   .byte  196,67,37,74,208,160                // vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   .byte  197,44,89,220                       // vmulps        %ymm4,%ymm10,%ymm11
   .byte  197,44,89,229                       // vmulps        %ymm5,%ymm10,%ymm12
   .byte  197,44,89,214                       // vmulps        %ymm6,%ymm10,%ymm10
-  .byte  196,98,125,24,45,159,75,0,0         // vbroadcastss  0x4b9f(%rip),%ymm13        # 5d40 <_sk_callback_avx+0x1c4>
+  .byte  196,98,125,24,45,191,75,0,0         // vbroadcastss  0x4bbf(%rip),%ymm13        # 5d60 <_sk_callback_avx+0x1c4>
   .byte  196,193,124,89,197                  // vmulps        %ymm13,%ymm0,%ymm0
-  .byte  196,98,125,24,53,149,75,0,0         // vbroadcastss  0x4b95(%rip),%ymm14        # 5d44 <_sk_callback_avx+0x1c8>
+  .byte  196,98,125,24,53,181,75,0,0         // vbroadcastss  0x4bb5(%rip),%ymm14        # 5d64 <_sk_callback_avx+0x1c8>
   .byte  196,193,116,89,206                  // vmulps        %ymm14,%ymm1,%ymm1
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,135,75,0,0        // vbroadcastss  0x4b87(%rip),%ymm1        # 5d48 <_sk_callback_avx+0x1cc>
+  .byte  196,226,125,24,13,167,75,0,0        // vbroadcastss  0x4ba7(%rip),%ymm1        # 5d68 <_sk_callback_avx+0x1cc>
   .byte  197,236,89,209                      // vmulps        %ymm1,%ymm2,%ymm2
   .byte  197,252,88,194                      // vaddps        %ymm2,%ymm0,%ymm0
   .byte  196,193,36,89,213                   // vmulps        %ymm13,%ymm11,%ymm2
@@ -14615,7 +14639,7 @@ HIDDEN _sk_clamp_1_avx
 .globl _sk_clamp_1_avx
 FUNCTION(_sk_clamp_1_avx)
 _sk_clamp_1_avx:
-  .byte  196,98,125,24,5,48,74,0,0           // vbroadcastss  0x4a30(%rip),%ymm8        # 5d4c <_sk_callback_avx+0x1d0>
+  .byte  196,98,125,24,5,80,74,0,0           // vbroadcastss  0x4a50(%rip),%ymm8        # 5d6c <_sk_callback_avx+0x1d0>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
@@ -14627,7 +14651,7 @@ HIDDEN _sk_clamp_a_avx
 .globl _sk_clamp_a_avx
 FUNCTION(_sk_clamp_a_avx)
 _sk_clamp_a_avx:
-  .byte  196,98,125,24,5,19,74,0,0           // vbroadcastss  0x4a13(%rip),%ymm8        # 5d50 <_sk_callback_avx+0x1d4>
+  .byte  196,98,125,24,5,51,74,0,0           // vbroadcastss  0x4a33(%rip),%ymm8        # 5d70 <_sk_callback_avx+0x1d4>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  197,252,93,195                      // vminps        %ymm3,%ymm0,%ymm0
   .byte  197,244,93,203                      // vminps        %ymm3,%ymm1,%ymm1
@@ -14713,7 +14737,7 @@ FUNCTION(_sk_unpremul_avx)
 _sk_unpremul_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,200,0                // vcmpeqps      %ymm8,%ymm3,%ymm9
-  .byte  196,98,125,24,21,91,73,0,0          // vbroadcastss  0x495b(%rip),%ymm10        # 5d54 <_sk_callback_avx+0x1d8>
+  .byte  196,98,125,24,21,123,73,0,0         // vbroadcastss  0x497b(%rip),%ymm10        # 5d74 <_sk_callback_avx+0x1d8>
   .byte  197,44,94,211                       // vdivps        %ymm3,%ymm10,%ymm10
   .byte  196,67,45,74,192,144                // vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
@@ -14726,17 +14750,17 @@ HIDDEN _sk_from_srgb_avx
 .globl _sk_from_srgb_avx
 FUNCTION(_sk_from_srgb_avx)
 _sk_from_srgb_avx:
-  .byte  196,98,125,24,5,60,73,0,0           // vbroadcastss  0x493c(%rip),%ymm8        # 5d58 <_sk_callback_avx+0x1dc>
+  .byte  196,98,125,24,5,92,73,0,0           // vbroadcastss  0x495c(%rip),%ymm8        # 5d78 <_sk_callback_avx+0x1dc>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  197,124,89,208                      // vmulps        %ymm0,%ymm0,%ymm10
-  .byte  196,98,125,24,29,46,73,0,0          // vbroadcastss  0x492e(%rip),%ymm11        # 5d5c <_sk_callback_avx+0x1e0>
+  .byte  196,98,125,24,29,78,73,0,0          // vbroadcastss  0x494e(%rip),%ymm11        # 5d7c <_sk_callback_avx+0x1e0>
   .byte  196,65,124,89,227                   // vmulps        %ymm11,%ymm0,%ymm12
-  .byte  196,98,125,24,45,36,73,0,0          // vbroadcastss  0x4924(%rip),%ymm13        # 5d60 <_sk_callback_avx+0x1e4>
+  .byte  196,98,125,24,45,68,73,0,0          // vbroadcastss  0x4944(%rip),%ymm13        # 5d80 <_sk_callback_avx+0x1e4>
   .byte  196,65,28,88,229                    // vaddps        %ymm13,%ymm12,%ymm12
   .byte  196,65,44,89,212                    // vmulps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,21,73,0,0          // vbroadcastss  0x4915(%rip),%ymm12        # 5d64 <_sk_callback_avx+0x1e8>
+  .byte  196,98,125,24,37,53,73,0,0          // vbroadcastss  0x4935(%rip),%ymm12        # 5d84 <_sk_callback_avx+0x1e8>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,53,11,73,0,0          // vbroadcastss  0x490b(%rip),%ymm14        # 5d68 <_sk_callback_avx+0x1ec>
+  .byte  196,98,125,24,53,43,73,0,0          // vbroadcastss  0x492b(%rip),%ymm14        # 5d88 <_sk_callback_avx+0x1ec>
   .byte  196,193,124,194,198,1               // vcmpltps      %ymm14,%ymm0,%ymm0
   .byte  196,195,45,74,193,0                 // vblendvps     %ymm0,%ymm9,%ymm10,%ymm0
   .byte  196,65,116,89,200                   // vmulps        %ymm8,%ymm1,%ymm9
@@ -14765,18 +14789,18 @@ _sk_to_srgb_avx:
   .byte  197,124,82,192                      // vrsqrtps      %ymm0,%ymm8
   .byte  196,65,124,83,200                   // vrcpps        %ymm8,%ymm9
   .byte  196,65,124,82,208                   // vrsqrtps      %ymm8,%ymm10
-  .byte  196,98,125,24,5,150,72,0,0          // vbroadcastss  0x4896(%rip),%ymm8        # 5d6c <_sk_callback_avx+0x1f0>
+  .byte  196,98,125,24,5,182,72,0,0          // vbroadcastss  0x48b6(%rip),%ymm8        # 5d8c <_sk_callback_avx+0x1f0>
   .byte  196,65,124,89,216                   // vmulps        %ymm8,%ymm0,%ymm11
-  .byte  196,98,125,24,37,140,72,0,0         // vbroadcastss  0x488c(%rip),%ymm12        # 5d70 <_sk_callback_avx+0x1f4>
+  .byte  196,98,125,24,37,172,72,0,0         // vbroadcastss  0x48ac(%rip),%ymm12        # 5d90 <_sk_callback_avx+0x1f4>
   .byte  196,65,52,89,204                    // vmulps        %ymm12,%ymm9,%ymm9
-  .byte  196,98,125,24,45,130,72,0,0         // vbroadcastss  0x4882(%rip),%ymm13        # 5d74 <_sk_callback_avx+0x1f8>
+  .byte  196,98,125,24,45,162,72,0,0         // vbroadcastss  0x48a2(%rip),%ymm13        # 5d94 <_sk_callback_avx+0x1f8>
   .byte  196,65,52,88,205                    // vaddps        %ymm13,%ymm9,%ymm9
-  .byte  196,98,125,24,53,120,72,0,0         // vbroadcastss  0x4878(%rip),%ymm14        # 5d78 <_sk_callback_avx+0x1fc>
+  .byte  196,98,125,24,53,152,72,0,0         // vbroadcastss  0x4898(%rip),%ymm14        # 5d98 <_sk_callback_avx+0x1fc>
   .byte  196,65,44,89,214                    // vmulps        %ymm14,%ymm10,%ymm10
   .byte  196,65,44,88,201                    // vaddps        %ymm9,%ymm10,%ymm9
-  .byte  196,98,125,24,21,105,72,0,0         // vbroadcastss  0x4869(%rip),%ymm10        # 5d7c <_sk_callback_avx+0x200>
+  .byte  196,98,125,24,21,137,72,0,0         // vbroadcastss  0x4889(%rip),%ymm10        # 5d9c <_sk_callback_avx+0x200>
   .byte  196,65,44,93,201                    // vminps        %ymm9,%ymm10,%ymm9
-  .byte  196,98,125,24,61,95,72,0,0          // vbroadcastss  0x485f(%rip),%ymm15        # 5d80 <_sk_callback_avx+0x204>
+  .byte  196,98,125,24,61,127,72,0,0         // vbroadcastss  0x487f(%rip),%ymm15        # 5da0 <_sk_callback_avx+0x204>
   .byte  196,193,124,194,199,1               // vcmpltps      %ymm15,%ymm0,%ymm0
   .byte  196,195,53,74,195,0                 // vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   .byte  197,124,82,201                      // vrsqrtps      %ymm1,%ymm9
@@ -14813,7 +14837,7 @@ _sk_rgb_to_hsl_avx:
   .byte  197,124,93,201                      // vminps        %ymm1,%ymm0,%ymm9
   .byte  197,52,93,202                       // vminps        %ymm2,%ymm9,%ymm9
   .byte  196,65,60,92,209                    // vsubps        %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,29,197,71,0,0         // vbroadcastss  0x47c5(%rip),%ymm11        # 5d84 <_sk_callback_avx+0x208>
+  .byte  196,98,125,24,29,229,71,0,0         // vbroadcastss  0x47e5(%rip),%ymm11        # 5da4 <_sk_callback_avx+0x208>
   .byte  196,65,36,94,218                    // vdivps        %ymm10,%ymm11,%ymm11
   .byte  197,116,92,226                      // vsubps        %ymm2,%ymm1,%ymm12
   .byte  196,65,28,89,227                    // vmulps        %ymm11,%ymm12,%ymm12
@@ -14823,19 +14847,19 @@ _sk_rgb_to_hsl_avx:
   .byte  196,193,108,89,211                  // vmulps        %ymm11,%ymm2,%ymm2
   .byte  197,252,92,201                      // vsubps        %ymm1,%ymm0,%ymm1
   .byte  196,193,116,89,203                  // vmulps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,158,71,0,0         // vbroadcastss  0x479e(%rip),%ymm11        # 5d90 <_sk_callback_avx+0x214>
+  .byte  196,98,125,24,29,190,71,0,0         // vbroadcastss  0x47be(%rip),%ymm11        # 5db0 <_sk_callback_avx+0x214>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,140,71,0,0         // vbroadcastss  0x478c(%rip),%ymm11        # 5d8c <_sk_callback_avx+0x210>
+  .byte  196,98,125,24,29,172,71,0,0         // vbroadcastss  0x47ac(%rip),%ymm11        # 5dac <_sk_callback_avx+0x210>
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
   .byte  196,227,117,74,202,224              // vblendvps     %ymm14,%ymm2,%ymm1,%ymm1
-  .byte  196,226,125,24,21,116,71,0,0        // vbroadcastss  0x4774(%rip),%ymm2        # 5d88 <_sk_callback_avx+0x20c>
+  .byte  196,226,125,24,21,148,71,0,0        // vbroadcastss  0x4794(%rip),%ymm2        # 5da8 <_sk_callback_avx+0x20c>
   .byte  196,65,12,87,246                    // vxorps        %ymm14,%ymm14,%ymm14
   .byte  196,227,13,74,210,208               // vblendvps     %ymm13,%ymm2,%ymm14,%ymm2
   .byte  197,188,194,192,0                   // vcmpeqps      %ymm0,%ymm8,%ymm0
   .byte  196,193,108,88,212                  // vaddps        %ymm12,%ymm2,%ymm2
   .byte  196,227,117,74,194,0                // vblendvps     %ymm0,%ymm2,%ymm1,%ymm0
   .byte  196,193,60,88,201                   // vaddps        %ymm9,%ymm8,%ymm1
-  .byte  196,98,125,24,37,91,71,0,0          // vbroadcastss  0x475b(%rip),%ymm12        # 5d98 <_sk_callback_avx+0x21c>
+  .byte  196,98,125,24,37,123,71,0,0         // vbroadcastss  0x477b(%rip),%ymm12        # 5db8 <_sk_callback_avx+0x21c>
   .byte  196,193,116,89,212                  // vmulps        %ymm12,%ymm1,%ymm2
   .byte  197,28,194,226,1                    // vcmpltps      %ymm2,%ymm12,%ymm12
   .byte  196,65,36,92,216                    // vsubps        %ymm8,%ymm11,%ymm11
@@ -14845,7 +14869,7 @@ _sk_rgb_to_hsl_avx:
   .byte  197,172,94,201                      // vdivps        %ymm1,%ymm10,%ymm1
   .byte  196,195,125,74,198,128              // vblendvps     %ymm8,%ymm14,%ymm0,%ymm0
   .byte  196,195,117,74,206,128              // vblendvps     %ymm8,%ymm14,%ymm1,%ymm1
-  .byte  196,98,125,24,5,30,71,0,0           // vbroadcastss  0x471e(%rip),%ymm8        # 5d94 <_sk_callback_avx+0x218>
+  .byte  196,98,125,24,5,62,71,0,0           // vbroadcastss  0x473e(%rip),%ymm8        # 5db4 <_sk_callback_avx+0x218>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -14862,7 +14886,7 @@ _sk_hsl_to_rgb_avx:
   .byte  197,252,17,92,36,128                // vmovups       %ymm3,-0x80(%rsp)
   .byte  197,252,40,225                      // vmovaps       %ymm1,%ymm4
   .byte  197,252,40,216                      // vmovaps       %ymm0,%ymm3
-  .byte  196,98,125,24,5,235,70,0,0          // vbroadcastss  0x46eb(%rip),%ymm8        # 5d9c <_sk_callback_avx+0x220>
+  .byte  196,98,125,24,5,11,71,0,0           // vbroadcastss  0x470b(%rip),%ymm8        # 5dbc <_sk_callback_avx+0x220>
   .byte  197,60,194,202,2                    // vcmpleps      %ymm2,%ymm8,%ymm9
   .byte  197,92,89,210                       // vmulps        %ymm2,%ymm4,%ymm10
   .byte  196,65,92,92,218                    // vsubps        %ymm10,%ymm4,%ymm11
@@ -14870,23 +14894,23 @@ _sk_hsl_to_rgb_avx:
   .byte  197,52,88,210                       // vaddps        %ymm2,%ymm9,%ymm10
   .byte  197,108,88,202                      // vaddps        %ymm2,%ymm2,%ymm9
   .byte  196,65,52,92,202                    // vsubps        %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,29,197,70,0,0         // vbroadcastss  0x46c5(%rip),%ymm11        # 5da0 <_sk_callback_avx+0x224>
+  .byte  196,98,125,24,29,229,70,0,0         // vbroadcastss  0x46e5(%rip),%ymm11        # 5dc0 <_sk_callback_avx+0x224>
   .byte  196,65,100,88,219                   // vaddps        %ymm11,%ymm3,%ymm11
   .byte  196,67,125,8,227,1                  // vroundps      $0x1,%ymm11,%ymm12
   .byte  196,65,36,92,252                    // vsubps        %ymm12,%ymm11,%ymm15
   .byte  196,65,44,92,217                    // vsubps        %ymm9,%ymm10,%ymm11
-  .byte  196,98,125,24,37,175,70,0,0         // vbroadcastss  0x46af(%rip),%ymm12        # 5da8 <_sk_callback_avx+0x22c>
+  .byte  196,98,125,24,37,207,70,0,0         // vbroadcastss  0x46cf(%rip),%ymm12        # 5dc8 <_sk_callback_avx+0x22c>
   .byte  196,193,4,89,196                    // vmulps        %ymm12,%ymm15,%ymm0
-  .byte  196,98,125,24,45,165,70,0,0         // vbroadcastss  0x46a5(%rip),%ymm13        # 5dac <_sk_callback_avx+0x230>
+  .byte  196,98,125,24,45,197,70,0,0         // vbroadcastss  0x46c5(%rip),%ymm13        # 5dcc <_sk_callback_avx+0x230>
   .byte  197,20,92,240                       // vsubps        %ymm0,%ymm13,%ymm14
   .byte  196,65,36,89,246                    // vmulps        %ymm14,%ymm11,%ymm14
   .byte  196,65,52,88,246                    // vaddps        %ymm14,%ymm9,%ymm14
-  .byte  196,226,125,24,13,134,70,0,0        // vbroadcastss  0x4686(%rip),%ymm1        # 5da4 <_sk_callback_avx+0x228>
+  .byte  196,226,125,24,13,166,70,0,0        // vbroadcastss  0x46a6(%rip),%ymm1        # 5dc4 <_sk_callback_avx+0x228>
   .byte  196,193,116,194,255,2               // vcmpleps      %ymm15,%ymm1,%ymm7
   .byte  196,195,13,74,249,112               // vblendvps     %ymm7,%ymm9,%ymm14,%ymm7
   .byte  196,65,60,194,247,2                 // vcmpleps      %ymm15,%ymm8,%ymm14
   .byte  196,227,45,74,255,224               // vblendvps     %ymm14,%ymm7,%ymm10,%ymm7
-  .byte  196,98,125,24,53,113,70,0,0         // vbroadcastss  0x4671(%rip),%ymm14        # 5db0 <_sk_callback_avx+0x234>
+  .byte  196,98,125,24,53,145,70,0,0         // vbroadcastss  0x4691(%rip),%ymm14        # 5dd0 <_sk_callback_avx+0x234>
   .byte  196,65,12,194,255,2                 // vcmpleps      %ymm15,%ymm14,%ymm15
   .byte  196,193,124,89,195                  // vmulps        %ymm11,%ymm0,%ymm0
   .byte  197,180,88,192                      // vaddps        %ymm0,%ymm9,%ymm0
@@ -14905,7 +14929,7 @@ _sk_hsl_to_rgb_avx:
   .byte  197,164,89,247                      // vmulps        %ymm7,%ymm11,%ymm6
   .byte  197,180,88,246                      // vaddps        %ymm6,%ymm9,%ymm6
   .byte  196,227,77,74,237,0                 // vblendvps     %ymm0,%ymm5,%ymm6,%ymm5
-  .byte  196,226,125,24,5,19,70,0,0          // vbroadcastss  0x4613(%rip),%ymm0        # 5db4 <_sk_callback_avx+0x238>
+  .byte  196,226,125,24,5,51,70,0,0          // vbroadcastss  0x4633(%rip),%ymm0        # 5dd4 <_sk_callback_avx+0x238>
   .byte  197,228,88,192                      // vaddps        %ymm0,%ymm3,%ymm0
   .byte  196,227,125,8,216,1                 // vroundps      $0x1,%ymm0,%ymm3
   .byte  197,252,92,195                      // vsubps        %ymm3,%ymm0,%ymm0
@@ -14964,7 +14988,7 @@ _sk_scale_u8_avx:
   .byte  196,66,121,49,192                   // vpmovzxbd     %xmm8,%xmm8
   .byte  196,67,53,24,192,1                  // vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,60,69,0,0          // vbroadcastss  0x453c(%rip),%ymm9        # 5db8 <_sk_callback_avx+0x23c>
+  .byte  196,98,125,24,13,92,69,0,0          // vbroadcastss  0x455c(%rip),%ymm9        # 5dd8 <_sk_callback_avx+0x23c>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -15023,7 +15047,7 @@ _sk_lerp_u8_avx:
   .byte  196,66,121,49,192                   // vpmovzxbd     %xmm8,%xmm8
   .byte  196,67,53,24,192,1                  // vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,136,68,0,0         // vbroadcastss  0x4488(%rip),%ymm9        # 5dbc <_sk_callback_avx+0x240>
+  .byte  196,98,125,24,13,168,68,0,0         // vbroadcastss  0x44a8(%rip),%ymm9        # 5ddc <_sk_callback_avx+0x240>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
@@ -15060,78 +15084,88 @@ _sk_lerp_565_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,174,0,0,0                    // jne           1a58 <_sk_lerp_565_avx+0xbc>
+  .byte  15,133,208,0,0,0                    // jne           1a7a <_sk_lerp_565_avx+0xde>
   .byte  196,65,122,111,4,122                // vmovdqu       (%r10,%rdi,2),%xmm8
-  .byte  197,225,239,219                     // vpxor         %xmm3,%xmm3,%xmm3
-  .byte  197,185,105,219                     // vpunpckhwd    %xmm3,%xmm8,%xmm3
+  .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
+  .byte  196,65,57,105,201                   // vpunpckhwd    %xmm9,%xmm8,%xmm9
   .byte  196,66,121,51,192                   // vpmovzxwd     %xmm8,%xmm8
-  .byte  196,227,61,24,219,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm3
-  .byte  196,98,125,24,5,244,67,0,0          // vbroadcastss  0x43f4(%rip),%ymm8        # 5dc0 <_sk_callback_avx+0x244>
-  .byte  196,65,100,84,192                   // vandps        %ymm8,%ymm3,%ymm8
-  .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,229,67,0,0         // vbroadcastss  0x43e5(%rip),%ymm9        # 5dc4 <_sk_callback_avx+0x248>
-  .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,219,67,0,0         // vbroadcastss  0x43db(%rip),%ymm9        # 5dc8 <_sk_callback_avx+0x24c>
-  .byte  196,65,100,84,201                   // vandps        %ymm9,%ymm3,%ymm9
+  .byte  196,67,61,24,193,1                  // vinsertf128   $0x1,%xmm9,%ymm8,%ymm8
+  .byte  196,98,125,24,13,18,68,0,0          // vbroadcastss  0x4412(%rip),%ymm9        # 5de0 <_sk_callback_avx+0x244>
+  .byte  196,65,60,84,201                    // vandps        %ymm9,%ymm8,%ymm9
   .byte  196,65,124,91,201                   // vcvtdq2ps     %ymm9,%ymm9
-  .byte  196,98,125,24,21,204,67,0,0         // vbroadcastss  0x43cc(%rip),%ymm10        # 5dcc <_sk_callback_avx+0x250>
+  .byte  196,98,125,24,21,3,68,0,0           // vbroadcastss  0x4403(%rip),%ymm10        # 5de4 <_sk_callback_avx+0x248>
   .byte  196,65,52,89,202                    // vmulps        %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,21,194,67,0,0         // vbroadcastss  0x43c2(%rip),%ymm10        # 5dd0 <_sk_callback_avx+0x254>
-  .byte  196,193,100,84,218                  // vandps        %ymm10,%ymm3,%ymm3
-  .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,21,180,67,0,0         // vbroadcastss  0x43b4(%rip),%ymm10        # 5dd4 <_sk_callback_avx+0x258>
-  .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
+  .byte  196,98,125,24,21,249,67,0,0         // vbroadcastss  0x43f9(%rip),%ymm10        # 5de8 <_sk_callback_avx+0x24c>
+  .byte  196,65,60,84,210                    // vandps        %ymm10,%ymm8,%ymm10
+  .byte  196,65,124,91,210                   // vcvtdq2ps     %ymm10,%ymm10
+  .byte  196,98,125,24,29,234,67,0,0         // vbroadcastss  0x43ea(%rip),%ymm11        # 5dec <_sk_callback_avx+0x250>
+  .byte  196,65,44,89,211                    // vmulps        %ymm11,%ymm10,%ymm10
+  .byte  196,98,125,24,29,224,67,0,0         // vbroadcastss  0x43e0(%rip),%ymm11        # 5df0 <_sk_callback_avx+0x254>
+  .byte  196,65,60,84,195                    // vandps        %ymm11,%ymm8,%ymm8
+  .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
+  .byte  196,98,125,24,29,209,67,0,0         // vbroadcastss  0x43d1(%rip),%ymm11        # 5df4 <_sk_callback_avx+0x258>
+  .byte  196,65,60,89,195                    // vmulps        %ymm11,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
-  .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
+  .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  197,252,88,196                      // vaddps        %ymm4,%ymm0,%ymm0
   .byte  197,244,92,205                      // vsubps        %ymm5,%ymm1,%ymm1
-  .byte  196,193,116,89,201                  // vmulps        %ymm9,%ymm1,%ymm1
+  .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  197,244,88,205                      // vaddps        %ymm5,%ymm1,%ymm1
   .byte  197,236,92,214                      // vsubps        %ymm6,%ymm2,%ymm2
-  .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
+  .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,236,88,214                      // vaddps        %ymm6,%ymm2,%ymm2
+  .byte  197,228,92,223                      // vsubps        %ymm7,%ymm3,%ymm3
+  .byte  196,65,100,89,201                   // vmulps        %ymm9,%ymm3,%ymm9
+  .byte  197,52,88,207                       // vaddps        %ymm7,%ymm9,%ymm9
+  .byte  196,65,100,89,210                   // vmulps        %ymm10,%ymm3,%ymm10
+  .byte  197,44,88,215                       // vaddps        %ymm7,%ymm10,%ymm10
+  .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
+  .byte  197,228,88,223                      // vaddps        %ymm7,%ymm3,%ymm3
+  .byte  197,172,95,219                      // vmaxps        %ymm3,%ymm10,%ymm3
+  .byte  197,180,95,219                      // vmaxps        %ymm3,%ymm9,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,130,67,0,0        // vbroadcastss  0x4382(%rip),%ymm3        # 5dd8 <_sk_callback_avx+0x25c>
   .byte  255,224                             // jmpq          *%rax
   .byte  65,137,200                          // mov           %ecx,%r8d
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  196,65,57,239,192                   // vpxor         %xmm8,%xmm8,%xmm8
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,63,255,255,255               // ja            19b0 <_sk_lerp_565_avx+0x14>
+  .byte  15,135,29,255,255,255               // ja            19b0 <_sk_lerp_565_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,76,0,0,0                  // lea           0x4c(%rip),%r9        # 1ac8 <_sk_lerp_565_avx+0x12c>
+  .byte  76,141,13,74,0,0,0                  // lea           0x4a(%rip),%r9        # 1ae8 <_sk_lerp_565_avx+0x14c>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
-  .byte  197,225,239,219                     // vpxor         %xmm3,%xmm3,%xmm3
-  .byte  196,65,97,196,68,122,12,6           // vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm3,%xmm8
+  .byte  196,65,57,239,192                   // vpxor         %xmm8,%xmm8,%xmm8
+  .byte  196,65,57,196,68,122,12,6           // vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,10,5           // vpinsrw       $0x5,0xa(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,8,4            // vpinsrw       $0x4,0x8(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,6,3            // vpinsrw       $0x3,0x6(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,4,2            // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,2,1            // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,4,122,0               // vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
-  .byte  233,235,254,255,255                 // jmpq          19b0 <_sk_lerp_565_avx+0x14>
-  .byte  15,31,0                             // nopl          (%rax)
-  .byte  241                                 // icebp
+  .byte  233,200,254,255,255                 // jmpq          19b0 <_sk_lerp_565_avx+0x14>
+  .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  233,255,255,255,225                 // jmpq          ffffffffe2001ad0 <_sk_callback_avx+0xffffffffe1ffbf54>
+  .byte  236                                 // in            (%dx),%al
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
+  .byte  255,228                             // jmpq          *%rsp
   .byte  255                                 // (bad)
-  .byte  217,255                             // fcos
   .byte  255                                 // (bad)
-  .byte  255,209                             // callq         *%rcx
   .byte  255                                 // (bad)
+  .byte  220,255                             // fdivr         %st,%st(7)
   .byte  255                                 // (bad)
-  .byte  255,201                             // dec           %ecx
+  .byte  255,212                             // callq         *%rsp
+  .byte  255                                 // (bad)
+  .byte  255                                 // (bad)
+  .byte  255,204                             // dec           %esp
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  189                                 // .byte         0xbd
+  .byte  191                                 // .byte         0xbf
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
@@ -15143,7 +15177,7 @@ _sk_load_tables_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,26,2,0,0                     // jne           1d0c <_sk_load_tables_avx+0x228>
+  .byte  15,133,26,2,0,0                     // jne           1d2c <_sk_load_tables_avx+0x228>
   .byte  196,65,124,16,4,184                 // vmovups       (%r8,%rdi,4),%ymm8
   .byte  85                                  // push          %rbp
   .byte  65,87                               // push          %r15
@@ -15151,7 +15185,7 @@ _sk_load_tables_avx:
   .byte  65,85                               // push          %r13
   .byte  65,84                               // push          %r12
   .byte  83                                  // push          %rbx
-  .byte  197,124,40,13,182,69,0,0            // vmovaps       0x45b6(%rip),%ymm9        # 60c0 <_sk_callback_avx+0x544>
+  .byte  197,124,40,13,182,69,0,0            // vmovaps       0x45b6(%rip),%ymm9        # 60e0 <_sk_callback_avx+0x544>
   .byte  196,193,60,84,193                   // vandps        %ymm9,%ymm8,%ymm0
   .byte  196,193,249,126,193                 // vmovq         %xmm0,%r9
   .byte  69,137,203                          // mov           %r9d,%r11d
@@ -15243,7 +15277,7 @@ _sk_load_tables_avx:
   .byte  196,193,97,114,210,24               // vpsrld        $0x18,%xmm10,%xmm3
   .byte  196,227,61,24,219,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,227,64,0,0          // vbroadcastss  0x40e3(%rip),%ymm8        # 5ddc <_sk_callback_avx+0x260>
+  .byte  196,98,125,24,5,223,64,0,0          // vbroadcastss  0x40df(%rip),%ymm8        # 5df8 <_sk_callback_avx+0x25c>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -15258,9 +15292,9 @@ _sk_load_tables_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  65,254,201                          // dec           %r9b
   .byte  65,128,249,6                        // cmp           $0x6,%r9b
-  .byte  15,135,211,253,255,255              // ja            1af8 <_sk_load_tables_avx+0x14>
+  .byte  15,135,211,253,255,255              // ja            1b18 <_sk_load_tables_avx+0x14>
   .byte  69,15,182,201                       // movzbl        %r9b,%r9d
-  .byte  76,141,21,140,0,0,0                 // lea           0x8c(%rip),%r10        # 1dbc <_sk_load_tables_avx+0x2d8>
+  .byte  76,141,21,140,0,0,0                 // lea           0x8c(%rip),%r10        # 1ddc <_sk_load_tables_avx+0x2d8>
   .byte  79,99,12,138                        // movslq        (%r10,%r9,4),%r9
   .byte  77,1,209                            // add           %r10,%r9
   .byte  65,255,225                          // jmpq          *%r9
@@ -15283,7 +15317,7 @@ _sk_load_tables_avx:
   .byte  196,99,61,12,192,15                 // vblendps      $0xf,%ymm0,%ymm8,%ymm8
   .byte  196,195,57,34,4,184,0               // vpinsrd       $0x0,(%r8,%rdi,4),%xmm8,%xmm0
   .byte  196,99,61,12,192,15                 // vblendps      $0xf,%ymm0,%ymm8,%ymm8
-  .byte  233,62,253,255,255                  // jmpq          1af8 <_sk_load_tables_avx+0x14>
+  .byte  233,62,253,255,255                  // jmpq          1b18 <_sk_load_tables_avx+0x14>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  236                                 // in            (%dx),%al
   .byte  255                                 // (bad)
@@ -15301,7 +15335,7 @@ _sk_load_tables_avx:
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  126,255                             // jle           1dd5 <_sk_load_tables_avx+0x2f1>
+  .byte  126,255                             // jle           1df5 <_sk_load_tables_avx+0x2f1>
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
 
@@ -15313,7 +15347,7 @@ _sk_load_tables_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,113,2,0,0                    // jne           205f <_sk_load_tables_u16_be_avx+0x287>
+  .byte  15,133,113,2,0,0                    // jne           207f <_sk_load_tables_u16_be_avx+0x287>
   .byte  196,1,121,16,4,72                   // vmovupd       (%r8,%r9,2),%xmm8
   .byte  196,129,121,16,84,72,16             // vmovupd       0x10(%r8,%r9,2),%xmm2
   .byte  196,129,121,16,92,72,32             // vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -15335,7 +15369,7 @@ _sk_load_tables_u16_be_avx:
   .byte  197,177,108,208                     // vpunpcklqdq   %xmm0,%xmm9,%xmm2
   .byte  197,177,109,200                     // vpunpckhqdq   %xmm0,%xmm9,%xmm1
   .byte  196,65,57,108,212                   // vpunpcklqdq   %xmm12,%xmm8,%xmm10
-  .byte  197,121,111,29,246,66,0,0           // vmovdqa       0x42f6(%rip),%xmm11        # 6140 <_sk_callback_avx+0x5c4>
+  .byte  197,121,111,29,246,66,0,0           // vmovdqa       0x42f6(%rip),%xmm11        # 6160 <_sk_callback_avx+0x5c4>
   .byte  196,193,105,219,195                 // vpand         %xmm11,%xmm2,%xmm0
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  196,193,121,105,209                 // vpunpckhwd    %xmm9,%xmm0,%xmm2
@@ -15434,7 +15468,7 @@ _sk_load_tables_u16_be_avx:
   .byte  196,226,121,51,219                  // vpmovzxwd     %xmm3,%xmm3
   .byte  196,195,101,24,216,1                // vinsertf128   $0x1,%xmm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,148,61,0,0          // vbroadcastss  0x3d94(%rip),%ymm8        # 5de0 <_sk_callback_avx+0x264>
+  .byte  196,98,125,24,5,144,61,0,0          // vbroadcastss  0x3d90(%rip),%ymm8        # 5dfc <_sk_callback_avx+0x260>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -15447,29 +15481,29 @@ _sk_load_tables_u16_be_avx:
   .byte  196,1,123,16,4,72                   // vmovsd        (%r8,%r9,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            20c5 <_sk_load_tables_u16_be_avx+0x2ed>
+  .byte  116,85                              // je            20e5 <_sk_load_tables_u16_be_avx+0x2ed>
   .byte  196,1,57,22,68,72,8                 // vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            20c5 <_sk_load_tables_u16_be_avx+0x2ed>
+  .byte  114,72                              // jb            20e5 <_sk_load_tables_u16_be_avx+0x2ed>
   .byte  196,129,123,16,84,72,16             // vmovsd        0x10(%r8,%r9,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            20d2 <_sk_load_tables_u16_be_avx+0x2fa>
+  .byte  116,72                              // je            20f2 <_sk_load_tables_u16_be_avx+0x2fa>
   .byte  196,129,105,22,84,72,24             // vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            20d2 <_sk_load_tables_u16_be_avx+0x2fa>
+  .byte  114,59                              // jb            20f2 <_sk_load_tables_u16_be_avx+0x2fa>
   .byte  196,129,123,16,92,72,32             // vmovsd        0x20(%r8,%r9,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,97,253,255,255               // je            1e09 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  15,132,97,253,255,255               // je            1e29 <_sk_load_tables_u16_be_avx+0x31>
   .byte  196,129,97,22,92,72,40              // vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,80,253,255,255               // jb            1e09 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  15,130,80,253,255,255               // jb            1e29 <_sk_load_tables_u16_be_avx+0x31>
   .byte  196,1,122,126,76,72,48              // vmovq         0x30(%r8,%r9,2),%xmm9
-  .byte  233,68,253,255,255                  // jmpq          1e09 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  233,68,253,255,255                  // jmpq          1e29 <_sk_load_tables_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,55,253,255,255                  // jmpq          1e09 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  233,55,253,255,255                  // jmpq          1e29 <_sk_load_tables_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,46,253,255,255                  // jmpq          1e09 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  233,46,253,255,255                  // jmpq          1e29 <_sk_load_tables_u16_be_avx+0x31>
 
 HIDDEN _sk_load_tables_rgb_u16_be_avx
 .globl _sk_load_tables_rgb_u16_be_avx
@@ -15479,7 +15513,7 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,127                       // lea           (%rdi,%rdi,2),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,93,2,0,0                     // jne           234a <_sk_load_tables_rgb_u16_be_avx+0x26f>
+  .byte  15,133,93,2,0,0                     // jne           236a <_sk_load_tables_rgb_u16_be_avx+0x26f>
   .byte  196,129,122,111,4,72                // vmovdqu       (%r8,%r9,2),%xmm0
   .byte  196,129,122,111,84,72,12            // vmovdqu       0xc(%r8,%r9,2),%xmm2
   .byte  196,129,122,111,76,72,24            // vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -15506,7 +15540,7 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  197,185,108,202                     // vpunpcklqdq   %xmm2,%xmm8,%xmm1
   .byte  197,185,109,210                     // vpunpckhqdq   %xmm2,%xmm8,%xmm2
   .byte  197,121,108,195                     // vpunpcklqdq   %xmm3,%xmm0,%xmm8
-  .byte  197,121,111,13,239,63,0,0           // vmovdqa       0x3fef(%rip),%xmm9        # 6150 <_sk_callback_avx+0x5d4>
+  .byte  197,121,111,13,239,63,0,0           // vmovdqa       0x3fef(%rip),%xmm9        # 6170 <_sk_callback_avx+0x5d4>
   .byte  196,193,113,219,193                 // vpand         %xmm9,%xmm1,%xmm0
   .byte  196,65,41,239,210                   // vpxor         %xmm10,%xmm10,%xmm10
   .byte  196,193,121,105,202                 // vpunpckhwd    %xmm10,%xmm0,%xmm1
@@ -15598,7 +15632,7 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  196,227,105,33,211,48               // vinsertps     $0x30,%xmm3,%xmm2,%xmm2
   .byte  196,195,109,24,208,1                // vinsertf128   $0x1,%xmm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,166,58,0,0        // vbroadcastss  0x3aa6(%rip),%ymm3        # 5de4 <_sk_callback_avx+0x268>
+  .byte  196,226,125,24,29,162,58,0,0        // vbroadcastss  0x3aa2(%rip),%ymm3        # 5e00 <_sk_callback_avx+0x264>
   .byte  91                                  // pop           %rbx
   .byte  65,92                               // pop           %r12
   .byte  65,93                               // pop           %r13
@@ -15609,36 +15643,36 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  196,129,121,110,4,72                // vmovd         (%r8,%r9,2),%xmm0
   .byte  196,129,121,196,68,72,4,2           // vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           2363 <_sk_load_tables_rgb_u16_be_avx+0x288>
-  .byte  233,190,253,255,255                 // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  117,5                               // jne           2383 <_sk_load_tables_rgb_u16_be_avx+0x288>
+  .byte  233,190,253,255,255                 // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
   .byte  196,129,121,110,76,72,6             // vmovd         0x6(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,68,72,10,2            // vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            2392 <_sk_load_tables_rgb_u16_be_avx+0x2b7>
+  .byte  114,26                              // jb            23b2 <_sk_load_tables_rgb_u16_be_avx+0x2b7>
   .byte  196,129,121,110,76,72,12            // vmovd         0xc(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,84,72,16,2          // vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           2397 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
-  .byte  233,143,253,255,255                 // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  .byte  233,138,253,255,255                 // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           23b7 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
+  .byte  233,143,253,255,255                 // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,138,253,255,255                 // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
   .byte  196,129,121,110,76,72,18            // vmovd         0x12(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,76,72,22,2            // vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            23c6 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
+  .byte  114,26                              // jb            23e6 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
   .byte  196,129,121,110,76,72,24            // vmovd         0x18(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,76,72,28,2          // vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           23cb <_sk_load_tables_rgb_u16_be_avx+0x2f0>
-  .byte  233,91,253,255,255                  // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  .byte  233,86,253,255,255                  // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           23eb <_sk_load_tables_rgb_u16_be_avx+0x2f0>
+  .byte  233,91,253,255,255                  // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,86,253,255,255                  // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
   .byte  196,129,121,110,92,72,30            // vmovd         0x1e(%r8,%r9,2),%xmm3
   .byte  196,1,97,196,92,72,34,2             // vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            23f4 <_sk_load_tables_rgb_u16_be_avx+0x319>
+  .byte  114,20                              // jb            2414 <_sk_load_tables_rgb_u16_be_avx+0x319>
   .byte  196,129,121,110,92,72,36            // vmovd         0x24(%r8,%r9,2),%xmm3
   .byte  196,129,97,196,92,72,40,2           // vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  .byte  233,45,253,255,255                  // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  .byte  233,40,253,255,255                  // jmpq          2121 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,45,253,255,255                  // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,40,253,255,255                  // jmpq          2141 <_sk_load_tables_rgb_u16_be_avx+0x46>
 
 HIDDEN _sk_byte_tables_avx
 .globl _sk_byte_tables_avx
@@ -15651,7 +15685,7 @@ _sk_byte_tables_avx:
   .byte  65,84                               // push          %r12
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,218,57,0,0          // vbroadcastss  0x39da(%rip),%ymm8        # 5de8 <_sk_callback_avx+0x26c>
+  .byte  196,98,125,24,5,214,57,0,0          // vbroadcastss  0x39d6(%rip),%ymm8        # 5e04 <_sk_callback_avx+0x268>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,195,249,22,192,1                // vpextrq       $0x1,%xmm0,%r8
@@ -15688,7 +15722,7 @@ _sk_byte_tables_avx:
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,53,24,192,1                 // vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,40,57,0,0          // vbroadcastss  0x3928(%rip),%ymm9        # 5dec <_sk_callback_avx+0x270>
+  .byte  196,98,125,24,13,36,57,0,0          // vbroadcastss  0x3924(%rip),%ymm9        # 5e08 <_sk_callback_avx+0x26c>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -15850,7 +15884,7 @@ _sk_byte_tables_rgb_avx:
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,53,24,192,1                 // vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,78,54,0,0          // vbroadcastss  0x364e(%rip),%ymm9        # 5df0 <_sk_callback_avx+0x274>
+  .byte  196,98,125,24,13,74,54,0,0          // vbroadcastss  0x364a(%rip),%ymm9        # 5e0c <_sk_callback_avx+0x270>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -16147,36 +16181,36 @@ _sk_parametric_r_avx:
   .byte  196,193,124,88,195                  // vaddps        %ymm11,%ymm0,%ymm0
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,216                      // vcvtdq2ps     %ymm0,%ymm11
-  .byte  196,98,125,24,37,172,49,0,0         // vbroadcastss  0x31ac(%rip),%ymm12        # 5df4 <_sk_callback_avx+0x278>
+  .byte  196,98,125,24,37,168,49,0,0         // vbroadcastss  0x31a8(%rip),%ymm12        # 5e10 <_sk_callback_avx+0x274>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,162,49,0,0         // vbroadcastss  0x31a2(%rip),%ymm12        # 5df8 <_sk_callback_avx+0x27c>
+  .byte  196,98,125,24,37,158,49,0,0         // vbroadcastss  0x319e(%rip),%ymm12        # 5e14 <_sk_callback_avx+0x278>
   .byte  196,193,124,84,196                  // vandps        %ymm12,%ymm0,%ymm0
-  .byte  196,98,125,24,37,152,49,0,0         // vbroadcastss  0x3198(%rip),%ymm12        # 5dfc <_sk_callback_avx+0x280>
+  .byte  196,98,125,24,37,148,49,0,0         // vbroadcastss  0x3194(%rip),%ymm12        # 5e18 <_sk_callback_avx+0x27c>
   .byte  196,193,124,86,196                  // vorps         %ymm12,%ymm0,%ymm0
-  .byte  196,98,125,24,37,142,49,0,0         // vbroadcastss  0x318e(%rip),%ymm12        # 5e00 <_sk_callback_avx+0x284>
+  .byte  196,98,125,24,37,138,49,0,0         // vbroadcastss  0x318a(%rip),%ymm12        # 5e1c <_sk_callback_avx+0x280>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,132,49,0,0         // vbroadcastss  0x3184(%rip),%ymm12        # 5e04 <_sk_callback_avx+0x288>
+  .byte  196,98,125,24,37,128,49,0,0         // vbroadcastss  0x3180(%rip),%ymm12        # 5e20 <_sk_callback_avx+0x284>
   .byte  196,65,124,89,228                   // vmulps        %ymm12,%ymm0,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,117,49,0,0         // vbroadcastss  0x3175(%rip),%ymm12        # 5e08 <_sk_callback_avx+0x28c>
+  .byte  196,98,125,24,37,113,49,0,0         // vbroadcastss  0x3171(%rip),%ymm12        # 5e24 <_sk_callback_avx+0x288>
   .byte  196,193,124,88,196                  // vaddps        %ymm12,%ymm0,%ymm0
-  .byte  196,98,125,24,37,107,49,0,0         // vbroadcastss  0x316b(%rip),%ymm12        # 5e0c <_sk_callback_avx+0x290>
+  .byte  196,98,125,24,37,103,49,0,0         // vbroadcastss  0x3167(%rip),%ymm12        # 5e28 <_sk_callback_avx+0x28c>
   .byte  197,156,94,192                      // vdivps        %ymm0,%ymm12,%ymm0
   .byte  197,164,92,192                      // vsubps        %ymm0,%ymm11,%ymm0
   .byte  197,172,89,192                      // vmulps        %ymm0,%ymm10,%ymm0
   .byte  196,99,125,8,208,1                  // vroundps      $0x1,%ymm0,%ymm10
   .byte  196,65,124,92,210                   // vsubps        %ymm10,%ymm0,%ymm10
-  .byte  196,98,125,24,29,79,49,0,0          // vbroadcastss  0x314f(%rip),%ymm11        # 5e10 <_sk_callback_avx+0x294>
+  .byte  196,98,125,24,29,75,49,0,0          // vbroadcastss  0x314b(%rip),%ymm11        # 5e2c <_sk_callback_avx+0x290>
   .byte  196,193,124,88,195                  // vaddps        %ymm11,%ymm0,%ymm0
-  .byte  196,98,125,24,29,69,49,0,0          // vbroadcastss  0x3145(%rip),%ymm11        # 5e14 <_sk_callback_avx+0x298>
+  .byte  196,98,125,24,29,65,49,0,0          // vbroadcastss  0x3141(%rip),%ymm11        # 5e30 <_sk_callback_avx+0x294>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,124,92,195                  // vsubps        %ymm11,%ymm0,%ymm0
-  .byte  196,98,125,24,29,54,49,0,0          // vbroadcastss  0x3136(%rip),%ymm11        # 5e18 <_sk_callback_avx+0x29c>
+  .byte  196,98,125,24,29,50,49,0,0          // vbroadcastss  0x3132(%rip),%ymm11        # 5e34 <_sk_callback_avx+0x298>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,44,49,0,0          // vbroadcastss  0x312c(%rip),%ymm11        # 5e1c <_sk_callback_avx+0x2a0>
+  .byte  196,98,125,24,29,40,49,0,0          // vbroadcastss  0x3128(%rip),%ymm11        # 5e38 <_sk_callback_avx+0x29c>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,124,88,194                  // vaddps        %ymm10,%ymm0,%ymm0
-  .byte  196,98,125,24,21,29,49,0,0          // vbroadcastss  0x311d(%rip),%ymm10        # 5e20 <_sk_callback_avx+0x2a4>
+  .byte  196,98,125,24,21,25,49,0,0          // vbroadcastss  0x3119(%rip),%ymm10        # 5e3c <_sk_callback_avx+0x2a0>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16184,7 +16218,7 @@ _sk_parametric_r_avx:
   .byte  196,195,125,74,193,128              // vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,124,95,192                  // vmaxps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,244,48,0,0          // vbroadcastss  0x30f4(%rip),%ymm8        # 5e24 <_sk_callback_avx+0x2a8>
+  .byte  196,98,125,24,5,240,48,0,0          // vbroadcastss  0x30f0(%rip),%ymm8        # 5e40 <_sk_callback_avx+0x2a4>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16206,36 +16240,36 @@ _sk_parametric_g_avx:
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,217                      // vcvtdq2ps     %ymm1,%ymm11
-  .byte  196,98,125,24,37,165,48,0,0         // vbroadcastss  0x30a5(%rip),%ymm12        # 5e28 <_sk_callback_avx+0x2ac>
+  .byte  196,98,125,24,37,161,48,0,0         // vbroadcastss  0x30a1(%rip),%ymm12        # 5e44 <_sk_callback_avx+0x2a8>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,155,48,0,0         // vbroadcastss  0x309b(%rip),%ymm12        # 5e2c <_sk_callback_avx+0x2b0>
+  .byte  196,98,125,24,37,151,48,0,0         // vbroadcastss  0x3097(%rip),%ymm12        # 5e48 <_sk_callback_avx+0x2ac>
   .byte  196,193,116,84,204                  // vandps        %ymm12,%ymm1,%ymm1
-  .byte  196,98,125,24,37,145,48,0,0         // vbroadcastss  0x3091(%rip),%ymm12        # 5e30 <_sk_callback_avx+0x2b4>
+  .byte  196,98,125,24,37,141,48,0,0         // vbroadcastss  0x308d(%rip),%ymm12        # 5e4c <_sk_callback_avx+0x2b0>
   .byte  196,193,116,86,204                  // vorps         %ymm12,%ymm1,%ymm1
-  .byte  196,98,125,24,37,135,48,0,0         // vbroadcastss  0x3087(%rip),%ymm12        # 5e34 <_sk_callback_avx+0x2b8>
+  .byte  196,98,125,24,37,131,48,0,0         // vbroadcastss  0x3083(%rip),%ymm12        # 5e50 <_sk_callback_avx+0x2b4>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,125,48,0,0         // vbroadcastss  0x307d(%rip),%ymm12        # 5e38 <_sk_callback_avx+0x2bc>
+  .byte  196,98,125,24,37,121,48,0,0         // vbroadcastss  0x3079(%rip),%ymm12        # 5e54 <_sk_callback_avx+0x2b8>
   .byte  196,65,116,89,228                   // vmulps        %ymm12,%ymm1,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,110,48,0,0         // vbroadcastss  0x306e(%rip),%ymm12        # 5e3c <_sk_callback_avx+0x2c0>
+  .byte  196,98,125,24,37,106,48,0,0         // vbroadcastss  0x306a(%rip),%ymm12        # 5e58 <_sk_callback_avx+0x2bc>
   .byte  196,193,116,88,204                  // vaddps        %ymm12,%ymm1,%ymm1
-  .byte  196,98,125,24,37,100,48,0,0         // vbroadcastss  0x3064(%rip),%ymm12        # 5e40 <_sk_callback_avx+0x2c4>
+  .byte  196,98,125,24,37,96,48,0,0          // vbroadcastss  0x3060(%rip),%ymm12        # 5e5c <_sk_callback_avx+0x2c0>
   .byte  197,156,94,201                      // vdivps        %ymm1,%ymm12,%ymm1
   .byte  197,164,92,201                      // vsubps        %ymm1,%ymm11,%ymm1
   .byte  197,172,89,201                      // vmulps        %ymm1,%ymm10,%ymm1
   .byte  196,99,125,8,209,1                  // vroundps      $0x1,%ymm1,%ymm10
   .byte  196,65,116,92,210                   // vsubps        %ymm10,%ymm1,%ymm10
-  .byte  196,98,125,24,29,72,48,0,0          // vbroadcastss  0x3048(%rip),%ymm11        # 5e44 <_sk_callback_avx+0x2c8>
+  .byte  196,98,125,24,29,68,48,0,0          // vbroadcastss  0x3044(%rip),%ymm11        # 5e60 <_sk_callback_avx+0x2c4>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,62,48,0,0          // vbroadcastss  0x303e(%rip),%ymm11        # 5e48 <_sk_callback_avx+0x2cc>
+  .byte  196,98,125,24,29,58,48,0,0          // vbroadcastss  0x303a(%rip),%ymm11        # 5e64 <_sk_callback_avx+0x2c8>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,116,92,203                  // vsubps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,47,48,0,0          // vbroadcastss  0x302f(%rip),%ymm11        # 5e4c <_sk_callback_avx+0x2d0>
+  .byte  196,98,125,24,29,43,48,0,0          // vbroadcastss  0x302b(%rip),%ymm11        # 5e68 <_sk_callback_avx+0x2cc>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,37,48,0,0          // vbroadcastss  0x3025(%rip),%ymm11        # 5e50 <_sk_callback_avx+0x2d4>
+  .byte  196,98,125,24,29,33,48,0,0          // vbroadcastss  0x3021(%rip),%ymm11        # 5e6c <_sk_callback_avx+0x2d0>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,116,88,202                  // vaddps        %ymm10,%ymm1,%ymm1
-  .byte  196,98,125,24,21,22,48,0,0          // vbroadcastss  0x3016(%rip),%ymm10        # 5e54 <_sk_callback_avx+0x2d8>
+  .byte  196,98,125,24,21,18,48,0,0          // vbroadcastss  0x3012(%rip),%ymm10        # 5e70 <_sk_callback_avx+0x2d4>
   .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16243,7 +16277,7 @@ _sk_parametric_g_avx:
   .byte  196,195,117,74,201,128              // vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,116,95,200                  // vmaxps        %ymm8,%ymm1,%ymm1
-  .byte  196,98,125,24,5,237,47,0,0          // vbroadcastss  0x2fed(%rip),%ymm8        # 5e58 <_sk_callback_avx+0x2dc>
+  .byte  196,98,125,24,5,233,47,0,0          // vbroadcastss  0x2fe9(%rip),%ymm8        # 5e74 <_sk_callback_avx+0x2d8>
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16265,36 +16299,36 @@ _sk_parametric_b_avx:
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,218                      // vcvtdq2ps     %ymm2,%ymm11
-  .byte  196,98,125,24,37,158,47,0,0         // vbroadcastss  0x2f9e(%rip),%ymm12        # 5e5c <_sk_callback_avx+0x2e0>
+  .byte  196,98,125,24,37,154,47,0,0         // vbroadcastss  0x2f9a(%rip),%ymm12        # 5e78 <_sk_callback_avx+0x2dc>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,148,47,0,0         // vbroadcastss  0x2f94(%rip),%ymm12        # 5e60 <_sk_callback_avx+0x2e4>
+  .byte  196,98,125,24,37,144,47,0,0         // vbroadcastss  0x2f90(%rip),%ymm12        # 5e7c <_sk_callback_avx+0x2e0>
   .byte  196,193,108,84,212                  // vandps        %ymm12,%ymm2,%ymm2
-  .byte  196,98,125,24,37,138,47,0,0         // vbroadcastss  0x2f8a(%rip),%ymm12        # 5e64 <_sk_callback_avx+0x2e8>
+  .byte  196,98,125,24,37,134,47,0,0         // vbroadcastss  0x2f86(%rip),%ymm12        # 5e80 <_sk_callback_avx+0x2e4>
   .byte  196,193,108,86,212                  // vorps         %ymm12,%ymm2,%ymm2
-  .byte  196,98,125,24,37,128,47,0,0         // vbroadcastss  0x2f80(%rip),%ymm12        # 5e68 <_sk_callback_avx+0x2ec>
+  .byte  196,98,125,24,37,124,47,0,0         // vbroadcastss  0x2f7c(%rip),%ymm12        # 5e84 <_sk_callback_avx+0x2e8>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,118,47,0,0         // vbroadcastss  0x2f76(%rip),%ymm12        # 5e6c <_sk_callback_avx+0x2f0>
+  .byte  196,98,125,24,37,114,47,0,0         // vbroadcastss  0x2f72(%rip),%ymm12        # 5e88 <_sk_callback_avx+0x2ec>
   .byte  196,65,108,89,228                   // vmulps        %ymm12,%ymm2,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,103,47,0,0         // vbroadcastss  0x2f67(%rip),%ymm12        # 5e70 <_sk_callback_avx+0x2f4>
+  .byte  196,98,125,24,37,99,47,0,0          // vbroadcastss  0x2f63(%rip),%ymm12        # 5e8c <_sk_callback_avx+0x2f0>
   .byte  196,193,108,88,212                  // vaddps        %ymm12,%ymm2,%ymm2
-  .byte  196,98,125,24,37,93,47,0,0          // vbroadcastss  0x2f5d(%rip),%ymm12        # 5e74 <_sk_callback_avx+0x2f8>
+  .byte  196,98,125,24,37,89,47,0,0          // vbroadcastss  0x2f59(%rip),%ymm12        # 5e90 <_sk_callback_avx+0x2f4>
   .byte  197,156,94,210                      // vdivps        %ymm2,%ymm12,%ymm2
   .byte  197,164,92,210                      // vsubps        %ymm2,%ymm11,%ymm2
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  196,99,125,8,210,1                  // vroundps      $0x1,%ymm2,%ymm10
   .byte  196,65,108,92,210                   // vsubps        %ymm10,%ymm2,%ymm10
-  .byte  196,98,125,24,29,65,47,0,0          // vbroadcastss  0x2f41(%rip),%ymm11        # 5e78 <_sk_callback_avx+0x2fc>
+  .byte  196,98,125,24,29,61,47,0,0          // vbroadcastss  0x2f3d(%rip),%ymm11        # 5e94 <_sk_callback_avx+0x2f8>
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
-  .byte  196,98,125,24,29,55,47,0,0          // vbroadcastss  0x2f37(%rip),%ymm11        # 5e7c <_sk_callback_avx+0x300>
+  .byte  196,98,125,24,29,51,47,0,0          // vbroadcastss  0x2f33(%rip),%ymm11        # 5e98 <_sk_callback_avx+0x2fc>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,108,92,211                  // vsubps        %ymm11,%ymm2,%ymm2
-  .byte  196,98,125,24,29,40,47,0,0          // vbroadcastss  0x2f28(%rip),%ymm11        # 5e80 <_sk_callback_avx+0x304>
+  .byte  196,98,125,24,29,36,47,0,0          // vbroadcastss  0x2f24(%rip),%ymm11        # 5e9c <_sk_callback_avx+0x300>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,30,47,0,0          // vbroadcastss  0x2f1e(%rip),%ymm11        # 5e84 <_sk_callback_avx+0x308>
+  .byte  196,98,125,24,29,26,47,0,0          // vbroadcastss  0x2f1a(%rip),%ymm11        # 5ea0 <_sk_callback_avx+0x304>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,108,88,210                  // vaddps        %ymm10,%ymm2,%ymm2
-  .byte  196,98,125,24,21,15,47,0,0          // vbroadcastss  0x2f0f(%rip),%ymm10        # 5e88 <_sk_callback_avx+0x30c>
+  .byte  196,98,125,24,21,11,47,0,0          // vbroadcastss  0x2f0b(%rip),%ymm10        # 5ea4 <_sk_callback_avx+0x308>
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  197,253,91,210                      // vcvtps2dq     %ymm2,%ymm2
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16302,7 +16336,7 @@ _sk_parametric_b_avx:
   .byte  196,195,109,74,209,128              // vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,108,95,208                  // vmaxps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,230,46,0,0          // vbroadcastss  0x2ee6(%rip),%ymm8        # 5e8c <_sk_callback_avx+0x310>
+  .byte  196,98,125,24,5,226,46,0,0          // vbroadcastss  0x2ee2(%rip),%ymm8        # 5ea8 <_sk_callback_avx+0x30c>
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16324,36 +16358,36 @@ _sk_parametric_a_avx:
   .byte  196,193,100,88,219                  // vaddps        %ymm11,%ymm3,%ymm3
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,219                      // vcvtdq2ps     %ymm3,%ymm11
-  .byte  196,98,125,24,37,151,46,0,0         // vbroadcastss  0x2e97(%rip),%ymm12        # 5e90 <_sk_callback_avx+0x314>
+  .byte  196,98,125,24,37,147,46,0,0         // vbroadcastss  0x2e93(%rip),%ymm12        # 5eac <_sk_callback_avx+0x310>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,141,46,0,0         // vbroadcastss  0x2e8d(%rip),%ymm12        # 5e94 <_sk_callback_avx+0x318>
+  .byte  196,98,125,24,37,137,46,0,0         // vbroadcastss  0x2e89(%rip),%ymm12        # 5eb0 <_sk_callback_avx+0x314>
   .byte  196,193,100,84,220                  // vandps        %ymm12,%ymm3,%ymm3
-  .byte  196,98,125,24,37,131,46,0,0         // vbroadcastss  0x2e83(%rip),%ymm12        # 5e98 <_sk_callback_avx+0x31c>
+  .byte  196,98,125,24,37,127,46,0,0         // vbroadcastss  0x2e7f(%rip),%ymm12        # 5eb4 <_sk_callback_avx+0x318>
   .byte  196,193,100,86,220                  // vorps         %ymm12,%ymm3,%ymm3
-  .byte  196,98,125,24,37,121,46,0,0         // vbroadcastss  0x2e79(%rip),%ymm12        # 5e9c <_sk_callback_avx+0x320>
+  .byte  196,98,125,24,37,117,46,0,0         // vbroadcastss  0x2e75(%rip),%ymm12        # 5eb8 <_sk_callback_avx+0x31c>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,111,46,0,0         // vbroadcastss  0x2e6f(%rip),%ymm12        # 5ea0 <_sk_callback_avx+0x324>
+  .byte  196,98,125,24,37,107,46,0,0         // vbroadcastss  0x2e6b(%rip),%ymm12        # 5ebc <_sk_callback_avx+0x320>
   .byte  196,65,100,89,228                   // vmulps        %ymm12,%ymm3,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,96,46,0,0          // vbroadcastss  0x2e60(%rip),%ymm12        # 5ea4 <_sk_callback_avx+0x328>
+  .byte  196,98,125,24,37,92,46,0,0          // vbroadcastss  0x2e5c(%rip),%ymm12        # 5ec0 <_sk_callback_avx+0x324>
   .byte  196,193,100,88,220                  // vaddps        %ymm12,%ymm3,%ymm3
-  .byte  196,98,125,24,37,86,46,0,0          // vbroadcastss  0x2e56(%rip),%ymm12        # 5ea8 <_sk_callback_avx+0x32c>
+  .byte  196,98,125,24,37,82,46,0,0          // vbroadcastss  0x2e52(%rip),%ymm12        # 5ec4 <_sk_callback_avx+0x328>
   .byte  197,156,94,219                      // vdivps        %ymm3,%ymm12,%ymm3
   .byte  197,164,92,219                      // vsubps        %ymm3,%ymm11,%ymm3
   .byte  197,172,89,219                      // vmulps        %ymm3,%ymm10,%ymm3
   .byte  196,99,125,8,211,1                  // vroundps      $0x1,%ymm3,%ymm10
   .byte  196,65,100,92,210                   // vsubps        %ymm10,%ymm3,%ymm10
-  .byte  196,98,125,24,29,58,46,0,0          // vbroadcastss  0x2e3a(%rip),%ymm11        # 5eac <_sk_callback_avx+0x330>
+  .byte  196,98,125,24,29,54,46,0,0          // vbroadcastss  0x2e36(%rip),%ymm11        # 5ec8 <_sk_callback_avx+0x32c>
   .byte  196,193,100,88,219                  // vaddps        %ymm11,%ymm3,%ymm3
-  .byte  196,98,125,24,29,48,46,0,0          // vbroadcastss  0x2e30(%rip),%ymm11        # 5eb0 <_sk_callback_avx+0x334>
+  .byte  196,98,125,24,29,44,46,0,0          // vbroadcastss  0x2e2c(%rip),%ymm11        # 5ecc <_sk_callback_avx+0x330>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,100,92,219                  // vsubps        %ymm11,%ymm3,%ymm3
-  .byte  196,98,125,24,29,33,46,0,0          // vbroadcastss  0x2e21(%rip),%ymm11        # 5eb4 <_sk_callback_avx+0x338>
+  .byte  196,98,125,24,29,29,46,0,0          // vbroadcastss  0x2e1d(%rip),%ymm11        # 5ed0 <_sk_callback_avx+0x334>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,23,46,0,0          // vbroadcastss  0x2e17(%rip),%ymm11        # 5eb8 <_sk_callback_avx+0x33c>
+  .byte  196,98,125,24,29,19,46,0,0          // vbroadcastss  0x2e13(%rip),%ymm11        # 5ed4 <_sk_callback_avx+0x338>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,100,88,218                  // vaddps        %ymm10,%ymm3,%ymm3
-  .byte  196,98,125,24,21,8,46,0,0           // vbroadcastss  0x2e08(%rip),%ymm10        # 5ebc <_sk_callback_avx+0x340>
+  .byte  196,98,125,24,21,4,46,0,0           // vbroadcastss  0x2e04(%rip),%ymm10        # 5ed8 <_sk_callback_avx+0x33c>
   .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
   .byte  197,253,91,219                      // vcvtps2dq     %ymm3,%ymm3
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16361,7 +16395,7 @@ _sk_parametric_a_avx:
   .byte  196,195,101,74,217,128              // vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,100,95,216                  // vmaxps        %ymm8,%ymm3,%ymm3
-  .byte  196,98,125,24,5,223,45,0,0          // vbroadcastss  0x2ddf(%rip),%ymm8        # 5ec0 <_sk_callback_avx+0x344>
+  .byte  196,98,125,24,5,219,45,0,0          // vbroadcastss  0x2ddb(%rip),%ymm8        # 5edc <_sk_callback_avx+0x340>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16370,31 +16404,31 @@ HIDDEN _sk_lab_to_xyz_avx
 .globl _sk_lab_to_xyz_avx
 FUNCTION(_sk_lab_to_xyz_avx)
 _sk_lab_to_xyz_avx:
-  .byte  196,98,125,24,5,209,45,0,0          // vbroadcastss  0x2dd1(%rip),%ymm8        # 5ec4 <_sk_callback_avx+0x348>
+  .byte  196,98,125,24,5,205,45,0,0          // vbroadcastss  0x2dcd(%rip),%ymm8        # 5ee0 <_sk_callback_avx+0x344>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,199,45,0,0          // vbroadcastss  0x2dc7(%rip),%ymm8        # 5ec8 <_sk_callback_avx+0x34c>
+  .byte  196,98,125,24,5,195,45,0,0          // vbroadcastss  0x2dc3(%rip),%ymm8        # 5ee4 <_sk_callback_avx+0x348>
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,98,125,24,13,189,45,0,0         // vbroadcastss  0x2dbd(%rip),%ymm9        # 5ecc <_sk_callback_avx+0x350>
+  .byte  196,98,125,24,13,185,45,0,0         // vbroadcastss  0x2db9(%rip),%ymm9        # 5ee8 <_sk_callback_avx+0x34c>
   .byte  196,193,116,88,201                  // vaddps        %ymm9,%ymm1,%ymm1
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  196,193,108,88,209                  // vaddps        %ymm9,%ymm2,%ymm2
-  .byte  196,98,125,24,5,169,45,0,0          // vbroadcastss  0x2da9(%rip),%ymm8        # 5ed0 <_sk_callback_avx+0x354>
+  .byte  196,98,125,24,5,165,45,0,0          // vbroadcastss  0x2da5(%rip),%ymm8        # 5eec <_sk_callback_avx+0x350>
   .byte  196,193,124,88,192                  // vaddps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,159,45,0,0          // vbroadcastss  0x2d9f(%rip),%ymm8        # 5ed4 <_sk_callback_avx+0x358>
+  .byte  196,98,125,24,5,155,45,0,0          // vbroadcastss  0x2d9b(%rip),%ymm8        # 5ef0 <_sk_callback_avx+0x354>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,149,45,0,0          // vbroadcastss  0x2d95(%rip),%ymm8        # 5ed8 <_sk_callback_avx+0x35c>
+  .byte  196,98,125,24,5,145,45,0,0          // vbroadcastss  0x2d91(%rip),%ymm8        # 5ef4 <_sk_callback_avx+0x358>
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  197,252,88,201                      // vaddps        %ymm1,%ymm0,%ymm1
-  .byte  196,98,125,24,5,135,45,0,0          // vbroadcastss  0x2d87(%rip),%ymm8        # 5edc <_sk_callback_avx+0x360>
+  .byte  196,98,125,24,5,131,45,0,0          // vbroadcastss  0x2d83(%rip),%ymm8        # 5ef8 <_sk_callback_avx+0x35c>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,252,92,210                      // vsubps        %ymm2,%ymm0,%ymm2
   .byte  197,116,89,193                      // vmulps        %ymm1,%ymm1,%ymm8
   .byte  196,65,116,89,192                   // vmulps        %ymm8,%ymm1,%ymm8
-  .byte  196,98,125,24,13,112,45,0,0         // vbroadcastss  0x2d70(%rip),%ymm9        # 5ee0 <_sk_callback_avx+0x364>
+  .byte  196,98,125,24,13,108,45,0,0         // vbroadcastss  0x2d6c(%rip),%ymm9        # 5efc <_sk_callback_avx+0x360>
   .byte  196,65,52,194,208,1                 // vcmpltps      %ymm8,%ymm9,%ymm10
-  .byte  196,98,125,24,29,101,45,0,0         // vbroadcastss  0x2d65(%rip),%ymm11        # 5ee4 <_sk_callback_avx+0x368>
+  .byte  196,98,125,24,29,97,45,0,0          // vbroadcastss  0x2d61(%rip),%ymm11        # 5f00 <_sk_callback_avx+0x364>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,37,91,45,0,0          // vbroadcastss  0x2d5b(%rip),%ymm12        # 5ee8 <_sk_callback_avx+0x36c>
+  .byte  196,98,125,24,37,87,45,0,0          // vbroadcastss  0x2d57(%rip),%ymm12        # 5f04 <_sk_callback_avx+0x368>
   .byte  196,193,116,89,204                  // vmulps        %ymm12,%ymm1,%ymm1
   .byte  196,67,117,74,192,160               // vblendvps     %ymm10,%ymm8,%ymm1,%ymm8
   .byte  197,252,89,200                      // vmulps        %ymm0,%ymm0,%ymm1
@@ -16409,9 +16443,9 @@ _sk_lab_to_xyz_avx:
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
   .byte  196,193,108,89,212                  // vmulps        %ymm12,%ymm2,%ymm2
   .byte  196,227,109,74,208,144              // vblendvps     %ymm9,%ymm0,%ymm2,%ymm2
-  .byte  196,226,125,24,5,17,45,0,0          // vbroadcastss  0x2d11(%rip),%ymm0        # 5eec <_sk_callback_avx+0x370>
+  .byte  196,226,125,24,5,13,45,0,0          // vbroadcastss  0x2d0d(%rip),%ymm0        # 5f08 <_sk_callback_avx+0x36c>
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
-  .byte  196,98,125,24,5,8,45,0,0            // vbroadcastss  0x2d08(%rip),%ymm8        # 5ef0 <_sk_callback_avx+0x374>
+  .byte  196,98,125,24,5,4,45,0,0            // vbroadcastss  0x2d04(%rip),%ymm8        # 5f0c <_sk_callback_avx+0x370>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16425,14 +16459,14 @@ _sk_load_a8_avx:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,62                              // jne           323f <_sk_load_a8_avx+0x4e>
+  .byte  117,62                              // jne           325f <_sk_load_a8_avx+0x4e>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,121,49,200                  // vpmovzxbd     %xmm0,%xmm1
   .byte  196,227,121,4,192,229               // vpermilps     $0xe5,%xmm0,%xmm0
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,117,24,192,1                // vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,204,44,0,0        // vbroadcastss  0x2ccc(%rip),%ymm1        # 5ef4 <_sk_callback_avx+0x378>
+  .byte  196,226,125,24,13,200,44,0,0        // vbroadcastss  0x2cc8(%rip),%ymm1        # 5f10 <_sk_callback_avx+0x374>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -16449,9 +16483,9 @@ _sk_load_a8_avx:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           3247 <_sk_load_a8_avx+0x56>
+  .byte  117,234                             // jne           3267 <_sk_load_a8_avx+0x56>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,161                             // jmp           3205 <_sk_load_a8_avx+0x14>
+  .byte  235,161                             // jmp           3225 <_sk_load_a8_avx+0x14>
 
 HIDDEN _sk_gather_a8_avx
 .globl _sk_gather_a8_avx
@@ -16501,7 +16535,7 @@ _sk_gather_a8_avx:
   .byte  196,226,121,49,201                  // vpmovzxbd     %xmm1,%xmm1
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,193,43,0,0        // vbroadcastss  0x2bc1(%rip),%ymm1        # 5ef8 <_sk_callback_avx+0x37c>
+  .byte  196,226,125,24,13,189,43,0,0        // vbroadcastss  0x2bbd(%rip),%ymm1        # 5f14 <_sk_callback_avx+0x378>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -16519,14 +16553,14 @@ FUNCTION(_sk_store_a8_avx)
 _sk_store_a8_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,156,43,0,0          // vbroadcastss  0x2b9c(%rip),%ymm8        # 5efc <_sk_callback_avx+0x380>
+  .byte  196,98,125,24,5,152,43,0,0          // vbroadcastss  0x2b98(%rip),%ymm8        # 5f18 <_sk_callback_avx+0x37c>
   .byte  196,65,100,89,192                   // vmulps        %ymm8,%ymm3,%ymm8
   .byte  196,65,125,91,192                   // vcvtps2dq     %ymm8,%ymm8
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  196,65,57,103,192                   // vpackuswb     %xmm8,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3389 <_sk_store_a8_avx+0x37>
+  .byte  117,10                              // jne           33a9 <_sk_store_a8_avx+0x37>
   .byte  196,65,123,17,4,58                  // vmovsd        %xmm8,(%r10,%rdi,1)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16534,10 +16568,10 @@ _sk_store_a8_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3385 <_sk_store_a8_avx+0x33>
+  .byte  119,236                             // ja            33a5 <_sk_store_a8_avx+0x33>
   .byte  196,66,121,48,192                   // vpmovzxbw     %xmm8,%xmm8
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 33ec <_sk_store_a8_avx+0x9a>
+  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 340c <_sk_store_a8_avx+0x9a>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16548,7 +16582,7 @@ _sk_store_a8_avx:
   .byte  196,67,121,20,68,58,2,4             // vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   .byte  196,67,121,20,68,58,1,2             // vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   .byte  196,67,121,20,4,58,0                // vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  .byte  235,154                             // jmp           3385 <_sk_store_a8_avx+0x33>
+  .byte  235,154                             // jmp           33a5 <_sk_store_a8_avx+0x33>
   .byte  144                                 // nop
   .byte  246,255                             // idiv          %bh
   .byte  255                                 // (bad)
@@ -16582,17 +16616,17 @@ _sk_load_g8_avx:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,67                              // jne           345b <_sk_load_g8_avx+0x53>
+  .byte  117,67                              // jne           347b <_sk_load_g8_avx+0x53>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,121,49,200                  // vpmovzxbd     %xmm0,%xmm1
   .byte  196,227,121,4,192,229               // vpermilps     $0xe5,%xmm0,%xmm0
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,117,24,192,1                // vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,193,42,0,0        // vbroadcastss  0x2ac1(%rip),%ymm1        # 5f00 <_sk_callback_avx+0x384>
+  .byte  196,226,125,24,13,189,42,0,0        // vbroadcastss  0x2abd(%rip),%ymm1        # 5f1c <_sk_callback_avx+0x380>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,182,42,0,0        // vbroadcastss  0x2ab6(%rip),%ymm3        # 5f04 <_sk_callback_avx+0x388>
+  .byte  196,226,125,24,29,178,42,0,0        // vbroadcastss  0x2ab2(%rip),%ymm3        # 5f20 <_sk_callback_avx+0x384>
   .byte  76,137,193                          // mov           %r8,%rcx
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
@@ -16606,9 +16640,9 @@ _sk_load_g8_avx:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           3463 <_sk_load_g8_avx+0x5b>
+  .byte  117,234                             // jne           3483 <_sk_load_g8_avx+0x5b>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,156                             // jmp           341c <_sk_load_g8_avx+0x14>
+  .byte  235,156                             // jmp           343c <_sk_load_g8_avx+0x14>
 
 HIDDEN _sk_gather_g8_avx
 .globl _sk_gather_g8_avx
@@ -16658,10 +16692,10 @@ _sk_gather_g8_avx:
   .byte  196,226,121,49,201                  // vpmovzxbd     %xmm1,%xmm1
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,181,41,0,0        // vbroadcastss  0x29b5(%rip),%ymm1        # 5f08 <_sk_callback_avx+0x38c>
+  .byte  196,226,125,24,13,177,41,0,0        // vbroadcastss  0x29b1(%rip),%ymm1        # 5f24 <_sk_callback_avx+0x388>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,170,41,0,0        // vbroadcastss  0x29aa(%rip),%ymm3        # 5f0c <_sk_callback_avx+0x390>
+  .byte  196,226,125,24,29,166,41,0,0        // vbroadcastss  0x29a6(%rip),%ymm3        # 5f28 <_sk_callback_avx+0x38c>
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
   .byte  91                                  // pop           %rbx
@@ -16677,9 +16711,9 @@ _sk_gather_i8_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            3582 <_sk_gather_i8_avx+0xf>
+  .byte  116,5                               // je            35a2 <_sk_gather_i8_avx+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           3584 <_sk_gather_i8_avx+0x11>
+  .byte  235,2                               // jmp           35a4 <_sk_gather_i8_avx+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,87                               // push          %r15
   .byte  65,86                               // push          %r14
@@ -16741,10 +16775,10 @@ _sk_gather_i8_avx:
   .byte  196,163,121,34,4,163,2              // vpinsrd       $0x2,(%rbx,%r12,4),%xmm0,%xmm0
   .byte  196,163,121,34,28,19,3              // vpinsrd       $0x3,(%rbx,%r10,1),%xmm0,%xmm3
   .byte  196,227,61,24,195,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  .byte  197,124,40,21,50,42,0,0             // vmovaps       0x2a32(%rip),%ymm10        # 60e0 <_sk_callback_avx+0x564>
+  .byte  197,124,40,21,50,42,0,0             // vmovaps       0x2a32(%rip),%ymm10        # 6100 <_sk_callback_avx+0x564>
   .byte  196,193,124,84,194                  // vandps        %ymm10,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,80,40,0,0          // vbroadcastss  0x2850(%rip),%ymm9        # 5f10 <_sk_callback_avx+0x394>
+  .byte  196,98,125,24,13,76,40,0,0          // vbroadcastss  0x284c(%rip),%ymm9        # 5f2c <_sk_callback_avx+0x390>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,113,114,208,8               // vpsrld        $0x8,%xmm8,%xmm1
   .byte  197,233,114,211,8                   // vpsrld        $0x8,%xmm3,%xmm2
@@ -16778,38 +16812,38 @@ _sk_load_565_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,128,0,0,0                    // jne           37b8 <_sk_load_565_avx+0x8e>
+  .byte  15,133,128,0,0,0                    // jne           37d8 <_sk_load_565_avx+0x8e>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  197,241,239,201                     // vpxor         %xmm1,%xmm1,%xmm1
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,209,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  .byte  196,226,125,24,5,186,39,0,0         // vbroadcastss  0x27ba(%rip),%ymm0        # 5f14 <_sk_callback_avx+0x398>
+  .byte  196,226,125,24,5,182,39,0,0         // vbroadcastss  0x27b6(%rip),%ymm0        # 5f30 <_sk_callback_avx+0x394>
   .byte  197,236,84,192                      // vandps        %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,173,39,0,0        // vbroadcastss  0x27ad(%rip),%ymm1        # 5f18 <_sk_callback_avx+0x39c>
+  .byte  196,226,125,24,13,169,39,0,0        // vbroadcastss  0x27a9(%rip),%ymm1        # 5f34 <_sk_callback_avx+0x398>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,164,39,0,0        // vbroadcastss  0x27a4(%rip),%ymm1        # 5f1c <_sk_callback_avx+0x3a0>
+  .byte  196,226,125,24,13,160,39,0,0        // vbroadcastss  0x27a0(%rip),%ymm1        # 5f38 <_sk_callback_avx+0x39c>
   .byte  197,236,84,201                      // vandps        %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,151,39,0,0        // vbroadcastss  0x2797(%rip),%ymm3        # 5f20 <_sk_callback_avx+0x3a4>
+  .byte  196,226,125,24,29,147,39,0,0        // vbroadcastss  0x2793(%rip),%ymm3        # 5f3c <_sk_callback_avx+0x3a0>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,24,29,142,39,0,0        // vbroadcastss  0x278e(%rip),%ymm3        # 5f24 <_sk_callback_avx+0x3a8>
+  .byte  196,226,125,24,29,138,39,0,0        // vbroadcastss  0x278a(%rip),%ymm3        # 5f40 <_sk_callback_avx+0x3a4>
   .byte  197,236,84,211                      // vandps        %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,129,39,0,0        // vbroadcastss  0x2781(%rip),%ymm3        # 5f28 <_sk_callback_avx+0x3ac>
+  .byte  196,226,125,24,29,125,39,0,0        // vbroadcastss  0x277d(%rip),%ymm3        # 5f44 <_sk_callback_avx+0x3a8>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,118,39,0,0        // vbroadcastss  0x2776(%rip),%ymm3        # 5f2c <_sk_callback_avx+0x3b0>
+  .byte  196,226,125,24,29,114,39,0,0        // vbroadcastss  0x2772(%rip),%ymm3        # 5f48 <_sk_callback_avx+0x3ac>
   .byte  255,224                             // jmpq          *%rax
   .byte  65,137,200                          // mov           %ecx,%r8d
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,110,255,255,255              // ja            373e <_sk_load_565_avx+0x14>
+  .byte  15,135,110,255,255,255              // ja            375e <_sk_load_565_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 3824 <_sk_load_565_avx+0xfa>
+  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 3844 <_sk_load_565_avx+0xfa>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16821,7 +16855,7 @@ _sk_load_565_avx:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,26,255,255,255                  // jmpq          373e <_sk_load_565_avx+0x14>
+  .byte  233,26,255,255,255                  // jmpq          375e <_sk_load_565_avx+0x14>
   .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -16899,23 +16933,23 @@ _sk_gather_565_avx:
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,209,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  .byte  196,226,125,24,5,22,38,0,0          // vbroadcastss  0x2616(%rip),%ymm0        # 5f30 <_sk_callback_avx+0x3b4>
+  .byte  196,226,125,24,5,18,38,0,0          // vbroadcastss  0x2612(%rip),%ymm0        # 5f4c <_sk_callback_avx+0x3b0>
   .byte  197,236,84,192                      // vandps        %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,9,38,0,0          // vbroadcastss  0x2609(%rip),%ymm1        # 5f34 <_sk_callback_avx+0x3b8>
+  .byte  196,226,125,24,13,5,38,0,0          // vbroadcastss  0x2605(%rip),%ymm1        # 5f50 <_sk_callback_avx+0x3b4>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,0,38,0,0          // vbroadcastss  0x2600(%rip),%ymm1        # 5f38 <_sk_callback_avx+0x3bc>
+  .byte  196,226,125,24,13,252,37,0,0        // vbroadcastss  0x25fc(%rip),%ymm1        # 5f54 <_sk_callback_avx+0x3b8>
   .byte  197,236,84,201                      // vandps        %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,243,37,0,0        // vbroadcastss  0x25f3(%rip),%ymm3        # 5f3c <_sk_callback_avx+0x3c0>
+  .byte  196,226,125,24,29,239,37,0,0        // vbroadcastss  0x25ef(%rip),%ymm3        # 5f58 <_sk_callback_avx+0x3bc>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,24,29,234,37,0,0        // vbroadcastss  0x25ea(%rip),%ymm3        # 5f40 <_sk_callback_avx+0x3c4>
+  .byte  196,226,125,24,29,230,37,0,0        // vbroadcastss  0x25e6(%rip),%ymm3        # 5f5c <_sk_callback_avx+0x3c0>
   .byte  197,236,84,211                      // vandps        %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,221,37,0,0        // vbroadcastss  0x25dd(%rip),%ymm3        # 5f44 <_sk_callback_avx+0x3c8>
+  .byte  196,226,125,24,29,217,37,0,0        // vbroadcastss  0x25d9(%rip),%ymm3        # 5f60 <_sk_callback_avx+0x3c4>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,210,37,0,0        // vbroadcastss  0x25d2(%rip),%ymm3        # 5f48 <_sk_callback_avx+0x3cc>
+  .byte  196,226,125,24,29,206,37,0,0        // vbroadcastss  0x25ce(%rip),%ymm3        # 5f64 <_sk_callback_avx+0x3c8>
   .byte  91                                  // pop           %rbx
   .byte  65,92                               // pop           %r12
   .byte  65,94                               // pop           %r14
@@ -16929,14 +16963,14 @@ FUNCTION(_sk_store_565_avx)
 _sk_store_565_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,190,37,0,0          // vbroadcastss  0x25be(%rip),%ymm8        # 5f4c <_sk_callback_avx+0x3d0>
+  .byte  196,98,125,24,5,186,37,0,0          // vbroadcastss  0x25ba(%rip),%ymm8        # 5f68 <_sk_callback_avx+0x3cc>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,41,114,241,11               // vpslld        $0xb,%xmm9,%xmm10
   .byte  196,67,125,25,201,1                 // vextractf128  $0x1,%ymm9,%xmm9
   .byte  196,193,49,114,241,11               // vpslld        $0xb,%xmm9,%xmm9
   .byte  196,67,45,24,201,1                  // vinsertf128   $0x1,%xmm9,%ymm10,%ymm9
-  .byte  196,98,125,24,21,151,37,0,0         // vbroadcastss  0x2597(%rip),%ymm10        # 5f50 <_sk_callback_avx+0x3d4>
+  .byte  196,98,125,24,21,147,37,0,0         // vbroadcastss  0x2593(%rip),%ymm10        # 5f6c <_sk_callback_avx+0x3d0>
   .byte  196,65,116,89,210                   // vmulps        %ymm10,%ymm1,%ymm10
   .byte  196,65,125,91,210                   // vcvtps2dq     %ymm10,%ymm10
   .byte  196,193,33,114,242,5                // vpslld        $0x5,%xmm10,%xmm11
@@ -16950,7 +16984,7 @@ _sk_store_565_avx:
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3a09 <_sk_store_565_avx+0x89>
+  .byte  117,10                              // jne           3a29 <_sk_store_565_avx+0x89>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16958,9 +16992,9 @@ _sk_store_565_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3a05 <_sk_store_565_avx+0x85>
+  .byte  119,236                             // ja            3a25 <_sk_store_565_avx+0x85>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 3a68 <_sk_store_565_avx+0xe8>
+  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 3a88 <_sk_store_565_avx+0xe8>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16971,7 +17005,7 @@ _sk_store_565_avx:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           3a05 <_sk_store_565_avx+0x85>
+  .byte  235,159                             // jmp           3a25 <_sk_store_565_avx+0x85>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -17004,31 +17038,31 @@ _sk_load_4444_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,152,0,0,0                    // jne           3b2a <_sk_load_4444_avx+0xa6>
+  .byte  15,133,152,0,0,0                    // jne           3b4a <_sk_load_4444_avx+0xa6>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  197,241,239,201                     // vpxor         %xmm1,%xmm1,%xmm1
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,217,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  .byte  196,226,125,24,5,160,36,0,0         // vbroadcastss  0x24a0(%rip),%ymm0        # 5f54 <_sk_callback_avx+0x3d8>
+  .byte  196,226,125,24,5,156,36,0,0         // vbroadcastss  0x249c(%rip),%ymm0        # 5f70 <_sk_callback_avx+0x3d4>
   .byte  197,228,84,192                      // vandps        %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,147,36,0,0        // vbroadcastss  0x2493(%rip),%ymm1        # 5f58 <_sk_callback_avx+0x3dc>
+  .byte  196,226,125,24,13,143,36,0,0        // vbroadcastss  0x248f(%rip),%ymm1        # 5f74 <_sk_callback_avx+0x3d8>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,138,36,0,0        // vbroadcastss  0x248a(%rip),%ymm1        # 5f5c <_sk_callback_avx+0x3e0>
+  .byte  196,226,125,24,13,134,36,0,0        // vbroadcastss  0x2486(%rip),%ymm1        # 5f78 <_sk_callback_avx+0x3dc>
   .byte  197,228,84,201                      // vandps        %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,125,36,0,0        // vbroadcastss  0x247d(%rip),%ymm2        # 5f60 <_sk_callback_avx+0x3e4>
+  .byte  196,226,125,24,21,121,36,0,0        // vbroadcastss  0x2479(%rip),%ymm2        # 5f7c <_sk_callback_avx+0x3e0>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,24,21,116,36,0,0        // vbroadcastss  0x2474(%rip),%ymm2        # 5f64 <_sk_callback_avx+0x3e8>
+  .byte  196,226,125,24,21,112,36,0,0        // vbroadcastss  0x2470(%rip),%ymm2        # 5f80 <_sk_callback_avx+0x3e4>
   .byte  197,228,84,210                      // vandps        %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,103,36,0,0          // vbroadcastss  0x2467(%rip),%ymm8        # 5f68 <_sk_callback_avx+0x3ec>
+  .byte  196,98,125,24,5,99,36,0,0           // vbroadcastss  0x2463(%rip),%ymm8        # 5f84 <_sk_callback_avx+0x3e8>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,93,36,0,0           // vbroadcastss  0x245d(%rip),%ymm8        # 5f6c <_sk_callback_avx+0x3f0>
+  .byte  196,98,125,24,5,89,36,0,0           // vbroadcastss  0x2459(%rip),%ymm8        # 5f88 <_sk_callback_avx+0x3ec>
   .byte  196,193,100,84,216                  // vandps        %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,79,36,0,0           // vbroadcastss  0x244f(%rip),%ymm8        # 5f70 <_sk_callback_avx+0x3f4>
+  .byte  196,98,125,24,5,75,36,0,0           // vbroadcastss  0x244b(%rip),%ymm8        # 5f8c <_sk_callback_avx+0x3f0>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17037,9 +17071,9 @@ _sk_load_4444_avx:
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,86,255,255,255               // ja            3a98 <_sk_load_4444_avx+0x14>
+  .byte  15,135,86,255,255,255               // ja            3ab8 <_sk_load_4444_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,75,0,0,0                  // lea           0x4b(%rip),%r9        # 3b98 <_sk_load_4444_avx+0x114>
+  .byte  76,141,13,75,0,0,0                  // lea           0x4b(%rip),%r9        # 3bb8 <_sk_load_4444_avx+0x114>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17051,7 +17085,7 @@ _sk_load_4444_avx:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,2,255,255,255                   // jmpq          3a98 <_sk_load_4444_avx+0x14>
+  .byte  233,2,255,255,255                   // jmpq          3ab8 <_sk_load_4444_avx+0x14>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  242,255                             // repnz         (bad)
   .byte  255                                 // (bad)
@@ -17130,25 +17164,25 @@ _sk_gather_4444_avx:
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,217,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  .byte  196,226,125,24,5,230,34,0,0         // vbroadcastss  0x22e6(%rip),%ymm0        # 5f74 <_sk_callback_avx+0x3f8>
+  .byte  196,226,125,24,5,226,34,0,0         // vbroadcastss  0x22e2(%rip),%ymm0        # 5f90 <_sk_callback_avx+0x3f4>
   .byte  197,228,84,192                      // vandps        %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,217,34,0,0        // vbroadcastss  0x22d9(%rip),%ymm1        # 5f78 <_sk_callback_avx+0x3fc>
+  .byte  196,226,125,24,13,213,34,0,0        // vbroadcastss  0x22d5(%rip),%ymm1        # 5f94 <_sk_callback_avx+0x3f8>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,208,34,0,0        // vbroadcastss  0x22d0(%rip),%ymm1        # 5f7c <_sk_callback_avx+0x400>
+  .byte  196,226,125,24,13,204,34,0,0        // vbroadcastss  0x22cc(%rip),%ymm1        # 5f98 <_sk_callback_avx+0x3fc>
   .byte  197,228,84,201                      // vandps        %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,195,34,0,0        // vbroadcastss  0x22c3(%rip),%ymm2        # 5f80 <_sk_callback_avx+0x404>
+  .byte  196,226,125,24,21,191,34,0,0        // vbroadcastss  0x22bf(%rip),%ymm2        # 5f9c <_sk_callback_avx+0x400>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,24,21,186,34,0,0        // vbroadcastss  0x22ba(%rip),%ymm2        # 5f84 <_sk_callback_avx+0x408>
+  .byte  196,226,125,24,21,182,34,0,0        // vbroadcastss  0x22b6(%rip),%ymm2        # 5fa0 <_sk_callback_avx+0x404>
   .byte  197,228,84,210                      // vandps        %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,173,34,0,0          // vbroadcastss  0x22ad(%rip),%ymm8        # 5f88 <_sk_callback_avx+0x40c>
+  .byte  196,98,125,24,5,169,34,0,0          // vbroadcastss  0x22a9(%rip),%ymm8        # 5fa4 <_sk_callback_avx+0x408>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,163,34,0,0          // vbroadcastss  0x22a3(%rip),%ymm8        # 5f8c <_sk_callback_avx+0x410>
+  .byte  196,98,125,24,5,159,34,0,0          // vbroadcastss  0x229f(%rip),%ymm8        # 5fa8 <_sk_callback_avx+0x40c>
   .byte  196,193,100,84,216                  // vandps        %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,149,34,0,0          // vbroadcastss  0x2295(%rip),%ymm8        # 5f90 <_sk_callback_avx+0x414>
+  .byte  196,98,125,24,5,145,34,0,0          // vbroadcastss  0x2291(%rip),%ymm8        # 5fac <_sk_callback_avx+0x410>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -17164,7 +17198,7 @@ FUNCTION(_sk_store_4444_avx)
 _sk_store_4444_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,122,34,0,0          // vbroadcastss  0x227a(%rip),%ymm8        # 5f94 <_sk_callback_avx+0x418>
+  .byte  196,98,125,24,5,118,34,0,0          // vbroadcastss  0x2276(%rip),%ymm8        # 5fb0 <_sk_callback_avx+0x414>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,41,114,241,12               // vpslld        $0xc,%xmm9,%xmm10
@@ -17191,7 +17225,7 @@ _sk_store_4444_avx:
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3db3 <_sk_store_4444_avx+0xa7>
+  .byte  117,10                              // jne           3dd3 <_sk_store_4444_avx+0xa7>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17199,9 +17233,9 @@ _sk_store_4444_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3daf <_sk_store_4444_avx+0xa3>
+  .byte  119,236                             // ja            3dcf <_sk_store_4444_avx+0xa3>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,66,0,0,0                  // lea           0x42(%rip),%r9        # 3e10 <_sk_store_4444_avx+0x104>
+  .byte  76,141,13,66,0,0,0                  // lea           0x42(%rip),%r9        # 3e30 <_sk_store_4444_avx+0x104>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17212,7 +17246,7 @@ _sk_store_4444_avx:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           3daf <_sk_store_4444_avx+0xa3>
+  .byte  235,159                             // jmp           3dcf <_sk_store_4444_avx+0xa3>
   .byte  247,255                             // idiv          %edi
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -17243,12 +17277,12 @@ _sk_load_8888_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,135,0,0,0                    // jne           3ec1 <_sk_load_8888_avx+0x95>
+  .byte  15,133,135,0,0,0                    // jne           3ee1 <_sk_load_8888_avx+0x95>
   .byte  196,65,124,16,12,186                // vmovups       (%r10,%rdi,4),%ymm9
-  .byte  197,124,40,21,184,34,0,0            // vmovaps       0x22b8(%rip),%ymm10        # 6100 <_sk_callback_avx+0x584>
+  .byte  197,124,40,21,184,34,0,0            // vmovaps       0x22b8(%rip),%ymm10        # 6120 <_sk_callback_avx+0x584>
   .byte  196,193,52,84,194                   // vandps        %ymm10,%ymm9,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,62,33,0,0           // vbroadcastss  0x213e(%rip),%ymm8        # 5f98 <_sk_callback_avx+0x41c>
+  .byte  196,98,125,24,5,58,33,0,0           // vbroadcastss  0x213a(%rip),%ymm8        # 5fb4 <_sk_callback_avx+0x418>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  196,193,113,114,209,8               // vpsrld        $0x8,%xmm9,%xmm1
   .byte  196,99,125,25,203,1                 // vextractf128  $0x1,%ymm9,%xmm3
@@ -17275,9 +17309,9 @@ _sk_load_8888_avx:
   .byte  196,65,52,87,201                    // vxorps        %ymm9,%ymm9,%ymm9
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,102,255,255,255              // ja            3e40 <_sk_load_8888_avx+0x14>
+  .byte  15,135,102,255,255,255              // ja            3e60 <_sk_load_8888_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,139,0,0,0                 // lea           0x8b(%rip),%r9        # 3f70 <_sk_load_8888_avx+0x144>
+  .byte  76,141,13,139,0,0,0                 // lea           0x8b(%rip),%r9        # 3f90 <_sk_load_8888_avx+0x144>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17300,7 +17334,7 @@ _sk_load_8888_avx:
   .byte  196,99,53,12,200,15                 // vblendps      $0xf,%ymm0,%ymm9,%ymm9
   .byte  196,195,49,34,4,186,0               // vpinsrd       $0x0,(%r10,%rdi,4),%xmm9,%xmm0
   .byte  196,99,53,12,200,15                 // vblendps      $0xf,%ymm0,%ymm9,%ymm9
-  .byte  233,210,254,255,255                 // jmpq          3e40 <_sk_load_8888_avx+0x14>
+  .byte  233,210,254,255,255                 // jmpq          3e60 <_sk_load_8888_avx+0x14>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  236                                 // in            (%dx),%al
   .byte  255                                 // (bad)
@@ -17318,7 +17352,7 @@ _sk_load_8888_avx:
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  126,255                             // jle           3f89 <_sk_load_8888_avx+0x15d>
+  .byte  126,255                             // jle           3fa9 <_sk_load_8888_avx+0x15d>
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
 
@@ -17363,10 +17397,10 @@ _sk_gather_8888_avx:
   .byte  196,131,121,34,4,152,2              // vpinsrd       $0x2,(%r8,%r11,4),%xmm0,%xmm0
   .byte  196,131,121,34,28,144,3             // vpinsrd       $0x3,(%r8,%r10,4),%xmm0,%xmm3
   .byte  196,227,61,24,195,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  .byte  197,124,40,21,226,32,0,0            // vmovaps       0x20e2(%rip),%ymm10        # 6120 <_sk_callback_avx+0x5a4>
+  .byte  197,124,40,21,226,32,0,0            // vmovaps       0x20e2(%rip),%ymm10        # 6140 <_sk_callback_avx+0x5a4>
   .byte  196,193,124,84,194                  // vandps        %ymm10,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,76,31,0,0          // vbroadcastss  0x1f4c(%rip),%ymm9        # 5f9c <_sk_callback_avx+0x420>
+  .byte  196,98,125,24,13,72,31,0,0          // vbroadcastss  0x1f48(%rip),%ymm9        # 5fb8 <_sk_callback_avx+0x41c>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,113,114,208,8               // vpsrld        $0x8,%xmm8,%xmm1
   .byte  197,233,114,211,8                   // vpsrld        $0x8,%xmm3,%xmm2
@@ -17398,7 +17432,7 @@ FUNCTION(_sk_store_8888_avx)
 _sk_store_8888_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,218,30,0,0          // vbroadcastss  0x1eda(%rip),%ymm8        # 5fa0 <_sk_callback_avx+0x424>
+  .byte  196,98,125,24,5,214,30,0,0          // vbroadcastss  0x1ed6(%rip),%ymm8        # 5fbc <_sk_callback_avx+0x420>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,65,116,89,208                   // vmulps        %ymm8,%ymm1,%ymm10
@@ -17423,7 +17457,7 @@ _sk_store_8888_avx:
   .byte  196,65,45,86,192                    // vorpd         %ymm8,%ymm10,%ymm8
   .byte  196,65,53,86,192                    // vorpd         %ymm8,%ymm9,%ymm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           4154 <_sk_store_8888_avx+0x9c>
+  .byte  117,10                              // jne           4174 <_sk_store_8888_avx+0x9c>
   .byte  196,65,124,17,4,186                 // vmovups       %ymm8,(%r10,%rdi,4)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17431,9 +17465,9 @@ _sk_store_8888_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            4150 <_sk_store_8888_avx+0x98>
+  .byte  119,236                             // ja            4170 <_sk_store_8888_avx+0x98>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,85,0,0,0                  // lea           0x55(%rip),%r9        # 41c4 <_sk_store_8888_avx+0x10c>
+  .byte  76,141,13,85,0,0,0                  // lea           0x55(%rip),%r9        # 41e4 <_sk_store_8888_avx+0x10c>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17447,7 +17481,7 @@ _sk_store_8888_avx:
   .byte  196,67,121,22,68,186,8,2            // vpextrd       $0x2,%xmm8,0x8(%r10,%rdi,4)
   .byte  196,67,121,22,68,186,4,1            // vpextrd       $0x1,%xmm8,0x4(%r10,%rdi,4)
   .byte  196,65,121,126,4,186                // vmovd         %xmm8,(%r10,%rdi,4)
-  .byte  235,143                             // jmp           4150 <_sk_store_8888_avx+0x98>
+  .byte  235,143                             // jmp           4170 <_sk_store_8888_avx+0x98>
   .byte  15,31,0                             // nopl          (%rax)
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -17485,7 +17519,7 @@ _sk_load_f16_avx:
   .byte  197,252,17,116,36,192               // vmovups       %ymm6,-0x40(%rsp)
   .byte  197,252,17,108,36,160               // vmovups       %ymm5,-0x60(%rsp)
   .byte  197,254,127,100,36,128              // vmovdqu       %ymm4,-0x80(%rsp)
-  .byte  15,133,141,2,0,0                    // jne           4497 <_sk_load_f16_avx+0x2b7>
+  .byte  15,133,141,2,0,0                    // jne           44b7 <_sk_load_f16_avx+0x2b7>
   .byte  197,121,16,4,248                    // vmovupd       (%rax,%rdi,8),%xmm8
   .byte  197,249,16,84,248,16                // vmovupd       0x10(%rax,%rdi,8),%xmm2
   .byte  197,249,16,76,248,32                // vmovupd       0x20(%rax,%rdi,8),%xmm1
@@ -17503,13 +17537,13 @@ _sk_load_f16_avx:
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
-  .byte  196,98,125,24,37,65,29,0,0          // vbroadcastss  0x1d41(%rip),%ymm12        # 5fa4 <_sk_callback_avx+0x428>
+  .byte  196,98,125,24,37,61,29,0,0          // vbroadcastss  0x1d3d(%rip),%ymm12        # 5fc0 <_sk_callback_avx+0x424>
   .byte  196,193,124,84,204                  // vandps        %ymm12,%ymm0,%ymm1
   .byte  197,252,87,193                      // vxorps        %ymm1,%ymm0,%ymm0
   .byte  196,195,125,25,198,1                // vextractf128  $0x1,%ymm0,%xmm14
-  .byte  196,98,121,24,29,45,29,0,0          // vbroadcastss  0x1d2d(%rip),%xmm11        # 5fa8 <_sk_callback_avx+0x42c>
+  .byte  196,98,121,24,29,41,29,0,0          // vbroadcastss  0x1d29(%rip),%xmm11        # 5fc4 <_sk_callback_avx+0x428>
   .byte  196,193,8,87,219                    // vxorps        %xmm11,%xmm14,%xmm3
-  .byte  196,98,121,24,45,35,29,0,0          // vbroadcastss  0x1d23(%rip),%xmm13        # 5fac <_sk_callback_avx+0x430>
+  .byte  196,98,121,24,45,31,29,0,0          // vbroadcastss  0x1d1f(%rip),%xmm13        # 5fc8 <_sk_callback_avx+0x42c>
   .byte  197,145,102,219                     // vpcmpgtd      %xmm3,%xmm13,%xmm3
   .byte  196,65,120,87,211                   // vxorps        %xmm11,%xmm0,%xmm10
   .byte  196,65,17,102,210                   // vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -17523,7 +17557,7 @@ _sk_load_f16_avx:
   .byte  196,227,125,24,195,1                // vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   .byte  197,252,86,193                      // vorps         %ymm1,%ymm0,%ymm0
   .byte  196,227,125,25,193,1                // vextractf128  $0x1,%ymm0,%xmm1
-  .byte  196,226,121,24,29,217,28,0,0        // vbroadcastss  0x1cd9(%rip),%xmm3        # 5fb0 <_sk_callback_avx+0x434>
+  .byte  196,226,121,24,29,213,28,0,0        // vbroadcastss  0x1cd5(%rip),%xmm3        # 5fcc <_sk_callback_avx+0x430>
   .byte  197,241,254,203                     // vpaddd        %xmm3,%xmm1,%xmm1
   .byte  197,249,254,195                     // vpaddd        %xmm3,%xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
@@ -17616,29 +17650,29 @@ _sk_load_f16_avx:
   .byte  197,123,16,4,248                    // vmovsd        (%rax,%rdi,8),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,79                              // je            44f6 <_sk_load_f16_avx+0x316>
+  .byte  116,79                              // je            4516 <_sk_load_f16_avx+0x316>
   .byte  197,57,22,68,248,8                  // vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,67                              // jb            44f6 <_sk_load_f16_avx+0x316>
+  .byte  114,67                              // jb            4516 <_sk_load_f16_avx+0x316>
   .byte  197,251,16,84,248,16                // vmovsd        0x10(%rax,%rdi,8),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,68                              // je            4503 <_sk_load_f16_avx+0x323>
+  .byte  116,68                              // je            4523 <_sk_load_f16_avx+0x323>
   .byte  197,233,22,84,248,24                // vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,56                              // jb            4503 <_sk_load_f16_avx+0x323>
+  .byte  114,56                              // jb            4523 <_sk_load_f16_avx+0x323>
   .byte  197,251,16,76,248,32                // vmovsd        0x20(%rax,%rdi,8),%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,70,253,255,255               // je            4221 <_sk_load_f16_avx+0x41>
+  .byte  15,132,70,253,255,255               // je            4241 <_sk_load_f16_avx+0x41>
   .byte  197,241,22,76,248,40                // vmovhpd       0x28(%rax,%rdi,8),%xmm1,%xmm1
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,54,253,255,255               // jb            4221 <_sk_load_f16_avx+0x41>
+  .byte  15,130,54,253,255,255               // jb            4241 <_sk_load_f16_avx+0x41>
   .byte  197,122,126,76,248,48               // vmovq         0x30(%rax,%rdi,8),%xmm9
-  .byte  233,43,253,255,255                  // jmpq          4221 <_sk_load_f16_avx+0x41>
+  .byte  233,43,253,255,255                  // jmpq          4241 <_sk_load_f16_avx+0x41>
   .byte  197,241,87,201                      // vxorpd        %xmm1,%xmm1,%xmm1
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,30,253,255,255                  // jmpq          4221 <_sk_load_f16_avx+0x41>
+  .byte  233,30,253,255,255                  // jmpq          4241 <_sk_load_f16_avx+0x41>
   .byte  197,241,87,201                      // vxorpd        %xmm1,%xmm1,%xmm1
-  .byte  233,21,253,255,255                  // jmpq          4221 <_sk_load_f16_avx+0x41>
+  .byte  233,21,253,255,255                  // jmpq          4241 <_sk_load_f16_avx+0x41>
 
 HIDDEN _sk_gather_f16_avx
 .globl _sk_gather_f16_avx
@@ -17702,13 +17736,13 @@ _sk_gather_f16_avx:
   .byte  197,249,105,210                     // vpunpckhwd    %xmm2,%xmm0,%xmm2
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,194,1                // vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
-  .byte  196,98,125,24,37,157,25,0,0         // vbroadcastss  0x199d(%rip),%ymm12        # 5fb4 <_sk_callback_avx+0x438>
+  .byte  196,98,125,24,37,153,25,0,0         // vbroadcastss  0x1999(%rip),%ymm12        # 5fd0 <_sk_callback_avx+0x434>
   .byte  196,193,124,84,212                  // vandps        %ymm12,%ymm0,%ymm2
   .byte  197,252,87,194                      // vxorps        %ymm2,%ymm0,%ymm0
   .byte  196,195,125,25,198,1                // vextractf128  $0x1,%ymm0,%xmm14
-  .byte  196,98,121,24,29,137,25,0,0         // vbroadcastss  0x1989(%rip),%xmm11        # 5fb8 <_sk_callback_avx+0x43c>
+  .byte  196,98,121,24,29,133,25,0,0         // vbroadcastss  0x1985(%rip),%xmm11        # 5fd4 <_sk_callback_avx+0x438>
   .byte  196,193,8,87,219                    // vxorps        %xmm11,%xmm14,%xmm3
-  .byte  196,98,121,24,45,127,25,0,0         // vbroadcastss  0x197f(%rip),%xmm13        # 5fbc <_sk_callback_avx+0x440>
+  .byte  196,98,121,24,45,123,25,0,0         // vbroadcastss  0x197b(%rip),%xmm13        # 5fd8 <_sk_callback_avx+0x43c>
   .byte  197,145,102,219                     // vpcmpgtd      %xmm3,%xmm13,%xmm3
   .byte  196,65,120,87,211                   // vxorps        %xmm11,%xmm0,%xmm10
   .byte  196,65,17,102,210                   // vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -17722,7 +17756,7 @@ _sk_gather_f16_avx:
   .byte  196,227,125,24,195,1                // vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   .byte  197,252,86,194                      // vorps         %ymm2,%ymm0,%ymm0
   .byte  196,227,125,25,194,1                // vextractf128  $0x1,%ymm0,%xmm2
-  .byte  196,226,121,24,29,53,25,0,0         // vbroadcastss  0x1935(%rip),%xmm3        # 5fc0 <_sk_callback_avx+0x444>
+  .byte  196,226,121,24,29,49,25,0,0         // vbroadcastss  0x1931(%rip),%xmm3        # 5fdc <_sk_callback_avx+0x440>
   .byte  197,233,254,211                     // vpaddd        %xmm3,%xmm2,%xmm2
   .byte  197,249,254,195                     // vpaddd        %xmm3,%xmm0,%xmm0
   .byte  196,227,125,24,194,1                // vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
@@ -17826,12 +17860,12 @@ _sk_store_f16_avx:
   .byte  197,252,17,52,36                    // vmovups       %ymm6,(%rsp)
   .byte  197,252,17,108,36,224               // vmovups       %ymm5,-0x20(%rsp)
   .byte  197,252,17,100,36,192               // vmovups       %ymm4,-0x40(%rsp)
-  .byte  196,98,125,24,13,78,23,0,0          // vbroadcastss  0x174e(%rip),%ymm9        # 5fc4 <_sk_callback_avx+0x448>
+  .byte  196,98,125,24,13,74,23,0,0          // vbroadcastss  0x174a(%rip),%ymm9        # 5fe0 <_sk_callback_avx+0x444>
   .byte  196,65,124,84,209                   // vandps        %ymm9,%ymm0,%ymm10
   .byte  197,252,17,68,36,128                // vmovups       %ymm0,-0x80(%rsp)
   .byte  196,65,124,87,218                   // vxorps        %ymm10,%ymm0,%ymm11
   .byte  196,67,125,25,220,1                 // vextractf128  $0x1,%ymm11,%xmm12
-  .byte  196,98,121,24,5,51,23,0,0           // vbroadcastss  0x1733(%rip),%xmm8        # 5fc8 <_sk_callback_avx+0x44c>
+  .byte  196,98,121,24,5,47,23,0,0           // vbroadcastss  0x172f(%rip),%xmm8        # 5fe4 <_sk_callback_avx+0x448>
   .byte  196,65,57,102,236                   // vpcmpgtd      %xmm12,%xmm8,%xmm13
   .byte  196,65,57,102,243                   // vpcmpgtd      %xmm11,%xmm8,%xmm14
   .byte  196,67,13,24,237,1                  // vinsertf128   $0x1,%xmm13,%ymm14,%ymm13
@@ -17841,7 +17875,7 @@ _sk_store_f16_avx:
   .byte  196,67,13,24,242,1                  // vinsertf128   $0x1,%xmm10,%ymm14,%ymm14
   .byte  196,193,33,114,211,13               // vpsrld        $0xd,%xmm11,%xmm11
   .byte  196,193,25,114,212,13               // vpsrld        $0xd,%xmm12,%xmm12
-  .byte  196,98,125,24,21,250,22,0,0         // vbroadcastss  0x16fa(%rip),%ymm10        # 5fcc <_sk_callback_avx+0x450>
+  .byte  196,98,125,24,21,246,22,0,0         // vbroadcastss  0x16f6(%rip),%ymm10        # 5fe8 <_sk_callback_avx+0x44c>
   .byte  196,65,12,86,242                    // vorps         %ymm10,%ymm14,%ymm14
   .byte  196,67,125,25,247,1                 // vextractf128  $0x1,%ymm14,%xmm15
   .byte  196,65,1,254,228                    // vpaddd        %xmm12,%xmm15,%xmm12
@@ -17923,7 +17957,7 @@ _sk_store_f16_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,66                              // jne           4ab0 <_sk_store_f16_avx+0x25e>
+  .byte  117,66                              // jne           4ad0 <_sk_store_f16_avx+0x25e>
   .byte  197,120,17,28,248                   // vmovups       %xmm11,(%rax,%rdi,8)
   .byte  197,120,17,84,248,16                // vmovups       %xmm10,0x10(%rax,%rdi,8)
   .byte  197,120,17,76,248,32                // vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -17939,22 +17973,22 @@ _sk_store_f16_avx:
   .byte  255,224                             // jmpq          *%rax
   .byte  197,121,214,28,248                  // vmovq         %xmm11,(%rax,%rdi,8)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,202                             // je            4a85 <_sk_store_f16_avx+0x233>
+  .byte  116,202                             // je            4aa5 <_sk_store_f16_avx+0x233>
   .byte  197,121,23,92,248,8                 // vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,190                             // jb            4a85 <_sk_store_f16_avx+0x233>
+  .byte  114,190                             // jb            4aa5 <_sk_store_f16_avx+0x233>
   .byte  197,121,214,84,248,16               // vmovq         %xmm10,0x10(%rax,%rdi,8)
-  .byte  116,182                             // je            4a85 <_sk_store_f16_avx+0x233>
+  .byte  116,182                             // je            4aa5 <_sk_store_f16_avx+0x233>
   .byte  197,121,23,84,248,24                // vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,170                             // jb            4a85 <_sk_store_f16_avx+0x233>
+  .byte  114,170                             // jb            4aa5 <_sk_store_f16_avx+0x233>
   .byte  197,121,214,76,248,32               // vmovq         %xmm9,0x20(%rax,%rdi,8)
-  .byte  116,162                             // je            4a85 <_sk_store_f16_avx+0x233>
+  .byte  116,162                             // je            4aa5 <_sk_store_f16_avx+0x233>
   .byte  197,121,23,76,248,40                // vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,150                             // jb            4a85 <_sk_store_f16_avx+0x233>
+  .byte  114,150                             // jb            4aa5 <_sk_store_f16_avx+0x233>
   .byte  197,121,214,68,248,48               // vmovq         %xmm8,0x30(%rax,%rdi,8)
-  .byte  235,142                             // jmp           4a85 <_sk_store_f16_avx+0x233>
+  .byte  235,142                             // jmp           4aa5 <_sk_store_f16_avx+0x233>
 
 HIDDEN _sk_load_u16_be_avx
 .globl _sk_load_u16_be_avx
@@ -17964,7 +17998,7 @@ _sk_load_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,253,0,0,0                    // jne           4c0a <_sk_load_u16_be_avx+0x113>
+  .byte  15,133,253,0,0,0                    // jne           4c2a <_sk_load_u16_be_avx+0x113>
   .byte  196,65,121,16,4,64                  // vmovupd       (%r8,%rax,2),%xmm8
   .byte  196,193,121,16,84,64,16             // vmovupd       0x10(%r8,%rax,2),%xmm2
   .byte  196,193,121,16,92,64,32             // vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -17986,7 +18020,7 @@ _sk_load_u16_be_avx:
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,29,82,20,0,0          // vbroadcastss  0x1452(%rip),%ymm11        # 5fd0 <_sk_callback_avx+0x454>
+  .byte  196,98,125,24,29,78,20,0,0          // vbroadcastss  0x144e(%rip),%ymm11        # 5fec <_sk_callback_avx+0x450>
   .byte  196,193,124,89,195                  // vmulps        %ymm11,%ymm0,%ymm0
   .byte  197,177,109,202                     // vpunpckhqdq   %xmm2,%xmm9,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -18020,29 +18054,29 @@ _sk_load_u16_be_avx:
   .byte  196,65,123,16,4,64                  // vmovsd        (%r8,%rax,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            4c70 <_sk_load_u16_be_avx+0x179>
+  .byte  116,85                              // je            4c90 <_sk_load_u16_be_avx+0x179>
   .byte  196,65,57,22,68,64,8                // vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            4c70 <_sk_load_u16_be_avx+0x179>
+  .byte  114,72                              // jb            4c90 <_sk_load_u16_be_avx+0x179>
   .byte  196,193,123,16,84,64,16             // vmovsd        0x10(%r8,%rax,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            4c7d <_sk_load_u16_be_avx+0x186>
+  .byte  116,72                              // je            4c9d <_sk_load_u16_be_avx+0x186>
   .byte  196,193,105,22,84,64,24             // vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            4c7d <_sk_load_u16_be_avx+0x186>
+  .byte  114,59                              // jb            4c9d <_sk_load_u16_be_avx+0x186>
   .byte  196,193,123,16,92,64,32             // vmovsd        0x20(%r8,%rax,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,213,254,255,255              // je            4b28 <_sk_load_u16_be_avx+0x31>
+  .byte  15,132,213,254,255,255              // je            4b48 <_sk_load_u16_be_avx+0x31>
   .byte  196,193,97,22,92,64,40              // vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,196,254,255,255              // jb            4b28 <_sk_load_u16_be_avx+0x31>
+  .byte  15,130,196,254,255,255              // jb            4b48 <_sk_load_u16_be_avx+0x31>
   .byte  196,65,122,126,76,64,48             // vmovq         0x30(%r8,%rax,2),%xmm9
-  .byte  233,184,254,255,255                 // jmpq          4b28 <_sk_load_u16_be_avx+0x31>
+  .byte  233,184,254,255,255                 // jmpq          4b48 <_sk_load_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,171,254,255,255                 // jmpq          4b28 <_sk_load_u16_be_avx+0x31>
+  .byte  233,171,254,255,255                 // jmpq          4b48 <_sk_load_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,162,254,255,255                 // jmpq          4b28 <_sk_load_u16_be_avx+0x31>
+  .byte  233,162,254,255,255                 // jmpq          4b48 <_sk_load_u16_be_avx+0x31>
 
 HIDDEN _sk_load_rgb_u16_be_avx
 .globl _sk_load_rgb_u16_be_avx
@@ -18052,7 +18086,7 @@ _sk_load_rgb_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,127                        // lea           (%rdi,%rdi,2),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,243,0,0,0                    // jne           4d8b <_sk_load_rgb_u16_be_avx+0x105>
+  .byte  15,133,243,0,0,0                    // jne           4dab <_sk_load_rgb_u16_be_avx+0x105>
   .byte  196,193,122,111,4,64                // vmovdqu       (%r8,%rax,2),%xmm0
   .byte  196,193,122,111,84,64,12            // vmovdqu       0xc(%r8,%rax,2),%xmm2
   .byte  196,193,122,111,76,64,24            // vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -18079,7 +18113,7 @@ _sk_load_rgb_u16_be_avx:
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,29,178,18,0,0         // vbroadcastss  0x12b2(%rip),%ymm11        # 5fd4 <_sk_callback_avx+0x458>
+  .byte  196,98,125,24,29,174,18,0,0         // vbroadcastss  0x12ae(%rip),%ymm11        # 5ff0 <_sk_callback_avx+0x454>
   .byte  196,193,124,89,195                  // vmulps        %ymm11,%ymm0,%ymm0
   .byte  197,185,109,202                     // vpunpckhqdq   %xmm2,%xmm8,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -18100,41 +18134,41 @@ _sk_load_rgb_u16_be_avx:
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,211                  // vmulps        %ymm11,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,79,18,0,0         // vbroadcastss  0x124f(%rip),%ymm3        # 5fd8 <_sk_callback_avx+0x45c>
+  .byte  196,226,125,24,29,75,18,0,0         // vbroadcastss  0x124b(%rip),%ymm3        # 5ff4 <_sk_callback_avx+0x458>
   .byte  255,224                             // jmpq          *%rax
   .byte  196,193,121,110,4,64                // vmovd         (%r8,%rax,2),%xmm0
   .byte  196,193,121,196,68,64,4,2           // vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           4da4 <_sk_load_rgb_u16_be_avx+0x11e>
-  .byte  233,40,255,255,255                  // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  117,5                               // jne           4dc4 <_sk_load_rgb_u16_be_avx+0x11e>
+  .byte  233,40,255,255,255                  // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
   .byte  196,193,121,110,76,64,6             // vmovd         0x6(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,68,64,10,2           // vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            4dd3 <_sk_load_rgb_u16_be_avx+0x14d>
+  .byte  114,26                              // jb            4df3 <_sk_load_rgb_u16_be_avx+0x14d>
   .byte  196,193,121,110,76,64,12            // vmovd         0xc(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,84,64,16,2          // vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           4dd8 <_sk_load_rgb_u16_be_avx+0x152>
-  .byte  233,249,254,255,255                 // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
-  .byte  233,244,254,255,255                 // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           4df8 <_sk_load_rgb_u16_be_avx+0x152>
+  .byte  233,249,254,255,255                 // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,244,254,255,255                 // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
   .byte  196,193,121,110,76,64,18            // vmovd         0x12(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,76,64,22,2           // vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            4e07 <_sk_load_rgb_u16_be_avx+0x181>
+  .byte  114,26                              // jb            4e27 <_sk_load_rgb_u16_be_avx+0x181>
   .byte  196,193,121,110,76,64,24            // vmovd         0x18(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,76,64,28,2          // vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           4e0c <_sk_load_rgb_u16_be_avx+0x186>
-  .byte  233,197,254,255,255                 // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
-  .byte  233,192,254,255,255                 // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           4e2c <_sk_load_rgb_u16_be_avx+0x186>
+  .byte  233,197,254,255,255                 // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,192,254,255,255                 // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
   .byte  196,193,121,110,92,64,30            // vmovd         0x1e(%r8,%rax,2),%xmm3
   .byte  196,65,97,196,92,64,34,2            // vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            4e35 <_sk_load_rgb_u16_be_avx+0x1af>
+  .byte  114,20                              // jb            4e55 <_sk_load_rgb_u16_be_avx+0x1af>
   .byte  196,193,121,110,92,64,36            // vmovd         0x24(%r8,%rax,2),%xmm3
   .byte  196,193,97,196,92,64,40,2           // vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  .byte  233,151,254,255,255                 // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
-  .byte  233,146,254,255,255                 // jmpq          4ccc <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,151,254,255,255                 // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,146,254,255,255                 // jmpq          4cec <_sk_load_rgb_u16_be_avx+0x46>
 
 HIDDEN _sk_store_u16_be_avx
 .globl _sk_store_u16_be_avx
@@ -18143,7 +18177,7 @@ _sk_store_u16_be_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
-  .byte  196,98,125,24,5,140,17,0,0          // vbroadcastss  0x118c(%rip),%ymm8        # 5fdc <_sk_callback_avx+0x460>
+  .byte  196,98,125,24,5,136,17,0,0          // vbroadcastss  0x1188(%rip),%ymm8        # 5ff8 <_sk_callback_avx+0x45c>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,67,125,25,202,1                 // vextractf128  $0x1,%ymm9,%xmm10
@@ -18181,7 +18215,7 @@ _sk_store_u16_be_avx:
   .byte  196,65,17,98,200                    // vpunpckldq    %xmm8,%xmm13,%xmm9
   .byte  196,65,17,106,192                   // vpunpckhdq    %xmm8,%xmm13,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,31                              // jne           4f34 <_sk_store_u16_be_avx+0xfa>
+  .byte  117,31                              // jne           4f54 <_sk_store_u16_be_avx+0xfa>
   .byte  196,65,120,17,28,64                 // vmovups       %xmm11,(%r8,%rax,2)
   .byte  196,65,120,17,84,64,16              // vmovups       %xmm10,0x10(%r8,%rax,2)
   .byte  196,65,120,17,76,64,32              // vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -18190,22 +18224,22 @@ _sk_store_u16_be_avx:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,214,28,64                // vmovq         %xmm11,(%r8,%rax,2)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  116,240                             // je            4f50 <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,23,92,64,8               // vmovhpd       %xmm11,0x8(%r8,%rax,2)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  114,227                             // jb            4f50 <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,214,84,64,16             // vmovq         %xmm10,0x10(%r8,%rax,2)
-  .byte  116,218                             // je            4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  116,218                             // je            4f50 <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,23,84,64,24              // vmovhpd       %xmm10,0x18(%r8,%rax,2)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  114,205                             // jb            4f50 <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,214,76,64,32             // vmovq         %xmm9,0x20(%r8,%rax,2)
-  .byte  116,196                             // je            4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  116,196                             // je            4f50 <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,23,76,64,40              // vmovhpd       %xmm9,0x28(%r8,%rax,2)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,183                             // jb            4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  114,183                             // jb            4f50 <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,214,68,64,48             // vmovq         %xmm8,0x30(%r8,%rax,2)
-  .byte  235,174                             // jmp           4f30 <_sk_store_u16_be_avx+0xf6>
+  .byte  235,174                             // jmp           4f50 <_sk_store_u16_be_avx+0xf6>
 
 HIDDEN _sk_load_f32_avx
 .globl _sk_load_f32_avx
@@ -18213,10 +18247,10 @@ FUNCTION(_sk_load_f32_avx)
 _sk_load_f32_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  119,110                             // ja            4ff8 <_sk_load_f32_avx+0x76>
+  .byte  119,110                             // ja            5018 <_sk_load_f32_avx+0x76>
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
-  .byte  76,141,21,132,0,0,0                 // lea           0x84(%rip),%r10        # 5020 <_sk_load_f32_avx+0x9e>
+  .byte  76,141,21,132,0,0,0                 // lea           0x84(%rip),%r10        # 5040 <_sk_load_f32_avx+0x9e>
   .byte  73,99,4,138                         // movslq        (%r10,%rcx,4),%rax
   .byte  76,1,208                            // add           %r10,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -18275,7 +18309,7 @@ _sk_store_f32_avx:
   .byte  196,65,37,20,196                    // vunpcklpd     %ymm12,%ymm11,%ymm8
   .byte  196,65,37,21,220                    // vunpckhpd     %ymm12,%ymm11,%ymm11
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,55                              // jne           50ad <_sk_store_f32_avx+0x6d>
+  .byte  117,55                              // jne           50cd <_sk_store_f32_avx+0x6d>
   .byte  196,67,45,24,225,1                  // vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   .byte  196,67,61,24,235,1                  // vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   .byte  196,67,45,6,201,49                  // vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -18288,22 +18322,22 @@ _sk_store_f32_avx:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,17,20,128                // vmovupd       %xmm10,(%r8,%rax,4)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            50a9 <_sk_store_f32_avx+0x69>
+  .byte  116,240                             // je            50c9 <_sk_store_f32_avx+0x69>
   .byte  196,65,121,17,76,128,16             // vmovupd       %xmm9,0x10(%r8,%rax,4)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            50a9 <_sk_store_f32_avx+0x69>
+  .byte  114,227                             // jb            50c9 <_sk_store_f32_avx+0x69>
   .byte  196,65,121,17,68,128,32             // vmovupd       %xmm8,0x20(%r8,%rax,4)
-  .byte  116,218                             // je            50a9 <_sk_store_f32_avx+0x69>
+  .byte  116,218                             // je            50c9 <_sk_store_f32_avx+0x69>
   .byte  196,65,121,17,92,128,48             // vmovupd       %xmm11,0x30(%r8,%rax,4)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            50a9 <_sk_store_f32_avx+0x69>
+  .byte  114,205                             // jb            50c9 <_sk_store_f32_avx+0x69>
   .byte  196,67,125,25,84,128,64,1           // vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  .byte  116,195                             // je            50a9 <_sk_store_f32_avx+0x69>
+  .byte  116,195                             // je            50c9 <_sk_store_f32_avx+0x69>
   .byte  196,67,125,25,76,128,80,1           // vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,181                             // jb            50a9 <_sk_store_f32_avx+0x69>
+  .byte  114,181                             // jb            50c9 <_sk_store_f32_avx+0x69>
   .byte  196,67,125,25,68,128,96,1           // vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  .byte  235,171                             // jmp           50a9 <_sk_store_f32_avx+0x69>
+  .byte  235,171                             // jmp           50c9 <_sk_store_f32_avx+0x69>
 
 HIDDEN _sk_clamp_x_avx
 .globl _sk_clamp_x_avx
@@ -18439,12 +18473,12 @@ HIDDEN _sk_luminance_to_alpha_avx
 .globl _sk_luminance_to_alpha_avx
 FUNCTION(_sk_luminance_to_alpha_avx)
 _sk_luminance_to_alpha_avx:
-  .byte  196,226,125,24,29,23,13,0,0         // vbroadcastss  0xd17(%rip),%ymm3        # 5fe0 <_sk_callback_avx+0x464>
+  .byte  196,226,125,24,29,19,13,0,0         // vbroadcastss  0xd13(%rip),%ymm3        # 5ffc <_sk_callback_avx+0x460>
   .byte  197,252,89,195                      // vmulps        %ymm3,%ymm0,%ymm0
-  .byte  196,226,125,24,29,14,13,0,0         // vbroadcastss  0xd0e(%rip),%ymm3        # 5fe4 <_sk_callback_avx+0x468>
+  .byte  196,226,125,24,29,10,13,0,0         // vbroadcastss  0xd0a(%rip),%ymm3        # 6000 <_sk_callback_avx+0x464>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,1,13,0,0          // vbroadcastss  0xd01(%rip),%ymm1        # 5fe8 <_sk_callback_avx+0x46c>
+  .byte  196,226,125,24,13,253,12,0,0        // vbroadcastss  0xcfd(%rip),%ymm1        # 6004 <_sk_callback_avx+0x468>
   .byte  197,236,89,201                      // vmulps        %ymm1,%ymm2,%ymm1
   .byte  197,252,88,217                      // vaddps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18622,7 +18656,7 @@ _sk_linear_gradient_avx:
   .byte  196,226,125,24,88,28                // vbroadcastss  0x1c(%rax),%ymm3
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  15,132,146,0,0,0                    // je            563d <_sk_linear_gradient_avx+0xb8>
+  .byte  15,132,146,0,0,0                    // je            565d <_sk_linear_gradient_avx+0xb8>
   .byte  72,139,64,8                         // mov           0x8(%rax),%rax
   .byte  72,131,192,32                       // add           $0x20,%rax
   .byte  196,65,28,87,228                    // vxorps        %ymm12,%ymm12,%ymm12
@@ -18649,8 +18683,8 @@ _sk_linear_gradient_avx:
   .byte  196,227,13,74,219,208               // vblendvps     %ymm13,%ymm3,%ymm14,%ymm3
   .byte  72,131,192,36                       // add           $0x24,%rax
   .byte  73,255,200                          // dec           %r8
-  .byte  117,140                             // jne           55c7 <_sk_linear_gradient_avx+0x42>
-  .byte  235,20                              // jmp           5651 <_sk_linear_gradient_avx+0xcc>
+  .byte  117,140                             // jne           55e7 <_sk_linear_gradient_avx+0x42>
+  .byte  235,20                              // jmp           5671 <_sk_linear_gradient_avx+0xcc>
   .byte  196,65,36,87,219                    // vxorps        %ymm11,%ymm11,%ymm11
   .byte  196,65,44,87,210                    // vxorps        %ymm10,%ymm10,%ymm10
   .byte  196,65,52,87,201                    // vxorps        %ymm9,%ymm9,%ymm9
@@ -18705,27 +18739,27 @@ _sk_xy_to_polar_unit_avx:
   .byte  196,65,52,95,226                    // vmaxps        %ymm10,%ymm9,%ymm12
   .byte  196,65,36,94,220                    // vdivps        %ymm12,%ymm11,%ymm11
   .byte  196,65,36,89,227                    // vmulps        %ymm11,%ymm11,%ymm12
-  .byte  196,98,125,24,45,230,8,0,0          // vbroadcastss  0x8e6(%rip),%ymm13        # 5fec <_sk_callback_avx+0x470>
+  .byte  196,98,125,24,45,226,8,0,0          // vbroadcastss  0x8e2(%rip),%ymm13        # 6008 <_sk_callback_avx+0x46c>
   .byte  196,65,28,89,237                    // vmulps        %ymm13,%ymm12,%ymm13
-  .byte  196,98,125,24,53,220,8,0,0          // vbroadcastss  0x8dc(%rip),%ymm14        # 5ff0 <_sk_callback_avx+0x474>
+  .byte  196,98,125,24,53,216,8,0,0          // vbroadcastss  0x8d8(%rip),%ymm14        # 600c <_sk_callback_avx+0x470>
   .byte  196,65,20,88,238                    // vaddps        %ymm14,%ymm13,%ymm13
   .byte  196,65,28,89,237                    // vmulps        %ymm13,%ymm12,%ymm13
-  .byte  196,98,125,24,53,205,8,0,0          // vbroadcastss  0x8cd(%rip),%ymm14        # 5ff4 <_sk_callback_avx+0x478>
+  .byte  196,98,125,24,53,201,8,0,0          // vbroadcastss  0x8c9(%rip),%ymm14        # 6010 <_sk_callback_avx+0x474>
   .byte  196,65,20,88,238                    // vaddps        %ymm14,%ymm13,%ymm13
   .byte  196,65,28,89,229                    // vmulps        %ymm13,%ymm12,%ymm12
-  .byte  196,98,125,24,45,190,8,0,0          // vbroadcastss  0x8be(%rip),%ymm13        # 5ff8 <_sk_callback_avx+0x47c>
+  .byte  196,98,125,24,45,186,8,0,0          // vbroadcastss  0x8ba(%rip),%ymm13        # 6014 <_sk_callback_avx+0x478>
   .byte  196,65,28,88,229                    // vaddps        %ymm13,%ymm12,%ymm12
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
   .byte  196,65,52,194,202,1                 // vcmpltps      %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,21,169,8,0,0          // vbroadcastss  0x8a9(%rip),%ymm10        # 5ffc <_sk_callback_avx+0x480>
+  .byte  196,98,125,24,21,165,8,0,0          // vbroadcastss  0x8a5(%rip),%ymm10        # 6018 <_sk_callback_avx+0x47c>
   .byte  196,65,44,92,211                    // vsubps        %ymm11,%ymm10,%ymm10
   .byte  196,67,37,74,202,144                // vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   .byte  196,193,124,194,192,1               // vcmpltps      %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,21,147,8,0,0          // vbroadcastss  0x893(%rip),%ymm10        # 6000 <_sk_callback_avx+0x484>
+  .byte  196,98,125,24,21,143,8,0,0          // vbroadcastss  0x88f(%rip),%ymm10        # 601c <_sk_callback_avx+0x480>
   .byte  196,65,44,92,209                    // vsubps        %ymm9,%ymm10,%ymm10
   .byte  196,195,53,74,194,0                 // vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   .byte  196,65,116,194,200,1                // vcmpltps      %ymm8,%ymm1,%ymm9
-  .byte  196,98,125,24,21,125,8,0,0          // vbroadcastss  0x87d(%rip),%ymm10        # 6004 <_sk_callback_avx+0x488>
+  .byte  196,98,125,24,21,121,8,0,0          // vbroadcastss  0x879(%rip),%ymm10        # 6020 <_sk_callback_avx+0x484>
   .byte  197,44,92,208                       // vsubps        %ymm0,%ymm10,%ymm10
   .byte  196,195,125,74,194,144              // vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   .byte  196,65,124,194,200,3                // vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -18750,7 +18784,7 @@ HIDDEN _sk_save_xy_avx
 FUNCTION(_sk_save_xy_avx)
 _sk_save_xy_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,67,8,0,0            // vbroadcastss  0x843(%rip),%ymm8        # 6008 <_sk_callback_avx+0x48c>
+  .byte  196,98,125,24,5,63,8,0,0            // vbroadcastss  0x83f(%rip),%ymm8        # 6024 <_sk_callback_avx+0x488>
   .byte  196,65,124,88,200                   // vaddps        %ymm8,%ymm0,%ymm9
   .byte  196,67,125,8,209,1                  // vroundps      $0x1,%ymm9,%ymm10
   .byte  196,65,52,92,202                    // vsubps        %ymm10,%ymm9,%ymm9
@@ -18787,9 +18821,9 @@ HIDDEN _sk_bilinear_nx_avx
 FUNCTION(_sk_bilinear_nx_avx)
 _sk_bilinear_nx_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,207,7,0,0          // vbroadcastss  0x7cf(%rip),%ymm0        # 600c <_sk_callback_avx+0x490>
+  .byte  196,226,125,24,5,203,7,0,0          // vbroadcastss  0x7cb(%rip),%ymm0        # 6028 <_sk_callback_avx+0x48c>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,198,7,0,0           // vbroadcastss  0x7c6(%rip),%ymm8        # 6010 <_sk_callback_avx+0x494>
+  .byte  196,98,125,24,5,194,7,0,0           // vbroadcastss  0x7c2(%rip),%ymm8        # 602c <_sk_callback_avx+0x490>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18800,7 +18834,7 @@ HIDDEN _sk_bilinear_px_avx
 FUNCTION(_sk_bilinear_px_avx)
 _sk_bilinear_px_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,174,7,0,0          // vbroadcastss  0x7ae(%rip),%ymm0        # 6014 <_sk_callback_avx+0x498>
+  .byte  196,226,125,24,5,170,7,0,0          // vbroadcastss  0x7aa(%rip),%ymm0        # 6030 <_sk_callback_avx+0x494>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -18812,9 +18846,9 @@ HIDDEN _sk_bilinear_ny_avx
 FUNCTION(_sk_bilinear_ny_avx)
 _sk_bilinear_ny_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,146,7,0,0         // vbroadcastss  0x792(%rip),%ymm1        # 6018 <_sk_callback_avx+0x49c>
+  .byte  196,226,125,24,13,142,7,0,0         // vbroadcastss  0x78e(%rip),%ymm1        # 6034 <_sk_callback_avx+0x498>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,136,7,0,0           // vbroadcastss  0x788(%rip),%ymm8        # 601c <_sk_callback_avx+0x4a0>
+  .byte  196,98,125,24,5,132,7,0,0           // vbroadcastss  0x784(%rip),%ymm8        # 6038 <_sk_callback_avx+0x49c>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18825,7 +18859,7 @@ HIDDEN _sk_bilinear_py_avx
 FUNCTION(_sk_bilinear_py_avx)
 _sk_bilinear_py_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,112,7,0,0         // vbroadcastss  0x770(%rip),%ymm1        # 6020 <_sk_callback_avx+0x4a4>
+  .byte  196,226,125,24,13,108,7,0,0         // vbroadcastss  0x76c(%rip),%ymm1        # 603c <_sk_callback_avx+0x4a0>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -18837,14 +18871,14 @@ HIDDEN _sk_bicubic_n3x_avx
 FUNCTION(_sk_bicubic_n3x_avx)
 _sk_bicubic_n3x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,83,7,0,0           // vbroadcastss  0x753(%rip),%ymm0        # 6024 <_sk_callback_avx+0x4a8>
+  .byte  196,226,125,24,5,79,7,0,0           // vbroadcastss  0x74f(%rip),%ymm0        # 6040 <_sk_callback_avx+0x4a4>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,74,7,0,0            // vbroadcastss  0x74a(%rip),%ymm8        # 6028 <_sk_callback_avx+0x4ac>
+  .byte  196,98,125,24,5,70,7,0,0            // vbroadcastss  0x746(%rip),%ymm8        # 6044 <_sk_callback_avx+0x4a8>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,59,7,0,0           // vbroadcastss  0x73b(%rip),%ymm10        # 602c <_sk_callback_avx+0x4b0>
+  .byte  196,98,125,24,21,55,7,0,0           // vbroadcastss  0x737(%rip),%ymm10        # 6048 <_sk_callback_avx+0x4ac>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,49,7,0,0           // vbroadcastss  0x731(%rip),%ymm10        # 6030 <_sk_callback_avx+0x4b4>
+  .byte  196,98,125,24,21,45,7,0,0           // vbroadcastss  0x72d(%rip),%ymm10        # 604c <_sk_callback_avx+0x4b0>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -18856,19 +18890,19 @@ HIDDEN _sk_bicubic_n1x_avx
 FUNCTION(_sk_bicubic_n1x_avx)
 _sk_bicubic_n1x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,20,7,0,0           // vbroadcastss  0x714(%rip),%ymm0        # 6034 <_sk_callback_avx+0x4b8>
+  .byte  196,226,125,24,5,16,7,0,0           // vbroadcastss  0x710(%rip),%ymm0        # 6050 <_sk_callback_avx+0x4b4>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,11,7,0,0            // vbroadcastss  0x70b(%rip),%ymm8        # 6038 <_sk_callback_avx+0x4bc>
+  .byte  196,98,125,24,5,7,7,0,0             // vbroadcastss  0x707(%rip),%ymm8        # 6054 <_sk_callback_avx+0x4b8>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,1,7,0,0            // vbroadcastss  0x701(%rip),%ymm9        # 603c <_sk_callback_avx+0x4c0>
+  .byte  196,98,125,24,13,253,6,0,0          // vbroadcastss  0x6fd(%rip),%ymm9        # 6058 <_sk_callback_avx+0x4bc>
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,247,6,0,0          // vbroadcastss  0x6f7(%rip),%ymm10        # 6040 <_sk_callback_avx+0x4c4>
+  .byte  196,98,125,24,21,243,6,0,0          // vbroadcastss  0x6f3(%rip),%ymm10        # 605c <_sk_callback_avx+0x4c0>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,232,6,0,0          // vbroadcastss  0x6e8(%rip),%ymm10        # 6044 <_sk_callback_avx+0x4c8>
+  .byte  196,98,125,24,21,228,6,0,0          // vbroadcastss  0x6e4(%rip),%ymm10        # 6060 <_sk_callback_avx+0x4c4>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,217,6,0,0          // vbroadcastss  0x6d9(%rip),%ymm9        # 6048 <_sk_callback_avx+0x4cc>
+  .byte  196,98,125,24,13,213,6,0,0          // vbroadcastss  0x6d5(%rip),%ymm9        # 6064 <_sk_callback_avx+0x4c8>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18879,17 +18913,17 @@ HIDDEN _sk_bicubic_p1x_avx
 FUNCTION(_sk_bicubic_p1x_avx)
 _sk_bicubic_p1x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,193,6,0,0           // vbroadcastss  0x6c1(%rip),%ymm8        # 604c <_sk_callback_avx+0x4d0>
+  .byte  196,98,125,24,5,189,6,0,0           // vbroadcastss  0x6bd(%rip),%ymm8        # 6068 <_sk_callback_avx+0x4cc>
   .byte  197,188,88,0                        // vaddps        (%rax),%ymm8,%ymm0
   .byte  197,124,16,72,64                    // vmovups       0x40(%rax),%ymm9
-  .byte  196,98,125,24,21,179,6,0,0          // vbroadcastss  0x6b3(%rip),%ymm10        # 6050 <_sk_callback_avx+0x4d4>
+  .byte  196,98,125,24,21,175,6,0,0          // vbroadcastss  0x6af(%rip),%ymm10        # 606c <_sk_callback_avx+0x4d0>
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
-  .byte  196,98,125,24,29,169,6,0,0          // vbroadcastss  0x6a9(%rip),%ymm11        # 6054 <_sk_callback_avx+0x4d8>
+  .byte  196,98,125,24,29,165,6,0,0          // vbroadcastss  0x6a5(%rip),%ymm11        # 6070 <_sk_callback_avx+0x4d4>
   .byte  196,65,44,88,211                    // vaddps        %ymm11,%ymm10,%ymm10
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
   .byte  196,65,44,88,192                    // vaddps        %ymm8,%ymm10,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
-  .byte  196,98,125,24,13,144,6,0,0          // vbroadcastss  0x690(%rip),%ymm9        # 6058 <_sk_callback_avx+0x4dc>
+  .byte  196,98,125,24,13,140,6,0,0          // vbroadcastss  0x68c(%rip),%ymm9        # 6074 <_sk_callback_avx+0x4d8>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18900,13 +18934,13 @@ HIDDEN _sk_bicubic_p3x_avx
 FUNCTION(_sk_bicubic_p3x_avx)
 _sk_bicubic_p3x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,120,6,0,0          // vbroadcastss  0x678(%rip),%ymm0        # 605c <_sk_callback_avx+0x4e0>
+  .byte  196,226,125,24,5,116,6,0,0          // vbroadcastss  0x674(%rip),%ymm0        # 6078 <_sk_callback_avx+0x4dc>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,101,6,0,0          // vbroadcastss  0x665(%rip),%ymm10        # 6060 <_sk_callback_avx+0x4e4>
+  .byte  196,98,125,24,21,97,6,0,0           // vbroadcastss  0x661(%rip),%ymm10        # 607c <_sk_callback_avx+0x4e0>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,91,6,0,0           // vbroadcastss  0x65b(%rip),%ymm10        # 6064 <_sk_callback_avx+0x4e8>
+  .byte  196,98,125,24,21,87,6,0,0           // vbroadcastss  0x657(%rip),%ymm10        # 6080 <_sk_callback_avx+0x4e4>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -18918,14 +18952,14 @@ HIDDEN _sk_bicubic_n3y_avx
 FUNCTION(_sk_bicubic_n3y_avx)
 _sk_bicubic_n3y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,62,6,0,0          // vbroadcastss  0x63e(%rip),%ymm1        # 6068 <_sk_callback_avx+0x4ec>
+  .byte  196,226,125,24,13,58,6,0,0          // vbroadcastss  0x63a(%rip),%ymm1        # 6084 <_sk_callback_avx+0x4e8>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,52,6,0,0            // vbroadcastss  0x634(%rip),%ymm8        # 606c <_sk_callback_avx+0x4f0>
+  .byte  196,98,125,24,5,48,6,0,0            // vbroadcastss  0x630(%rip),%ymm8        # 6088 <_sk_callback_avx+0x4ec>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,37,6,0,0           // vbroadcastss  0x625(%rip),%ymm10        # 6070 <_sk_callback_avx+0x4f4>
+  .byte  196,98,125,24,21,33,6,0,0           // vbroadcastss  0x621(%rip),%ymm10        # 608c <_sk_callback_avx+0x4f0>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,27,6,0,0           // vbroadcastss  0x61b(%rip),%ymm10        # 6074 <_sk_callback_avx+0x4f8>
+  .byte  196,98,125,24,21,23,6,0,0           // vbroadcastss  0x617(%rip),%ymm10        # 6090 <_sk_callback_avx+0x4f4>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -18937,19 +18971,19 @@ HIDDEN _sk_bicubic_n1y_avx
 FUNCTION(_sk_bicubic_n1y_avx)
 _sk_bicubic_n1y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,254,5,0,0         // vbroadcastss  0x5fe(%rip),%ymm1        # 6078 <_sk_callback_avx+0x4fc>
+  .byte  196,226,125,24,13,250,5,0,0         // vbroadcastss  0x5fa(%rip),%ymm1        # 6094 <_sk_callback_avx+0x4f8>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,244,5,0,0           // vbroadcastss  0x5f4(%rip),%ymm8        # 607c <_sk_callback_avx+0x500>
+  .byte  196,98,125,24,5,240,5,0,0           // vbroadcastss  0x5f0(%rip),%ymm8        # 6098 <_sk_callback_avx+0x4fc>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,234,5,0,0          // vbroadcastss  0x5ea(%rip),%ymm9        # 6080 <_sk_callback_avx+0x504>
+  .byte  196,98,125,24,13,230,5,0,0          // vbroadcastss  0x5e6(%rip),%ymm9        # 609c <_sk_callback_avx+0x500>
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,224,5,0,0          // vbroadcastss  0x5e0(%rip),%ymm10        # 6084 <_sk_callback_avx+0x508>
+  .byte  196,98,125,24,21,220,5,0,0          // vbroadcastss  0x5dc(%rip),%ymm10        # 60a0 <_sk_callback_avx+0x504>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,209,5,0,0          // vbroadcastss  0x5d1(%rip),%ymm10        # 6088 <_sk_callback_avx+0x50c>
+  .byte  196,98,125,24,21,205,5,0,0          // vbroadcastss  0x5cd(%rip),%ymm10        # 60a4 <_sk_callback_avx+0x508>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,194,5,0,0          // vbroadcastss  0x5c2(%rip),%ymm9        # 608c <_sk_callback_avx+0x510>
+  .byte  196,98,125,24,13,190,5,0,0          // vbroadcastss  0x5be(%rip),%ymm9        # 60a8 <_sk_callback_avx+0x50c>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18960,17 +18994,17 @@ HIDDEN _sk_bicubic_p1y_avx
 FUNCTION(_sk_bicubic_p1y_avx)
 _sk_bicubic_p1y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,170,5,0,0           // vbroadcastss  0x5aa(%rip),%ymm8        # 6090 <_sk_callback_avx+0x514>
+  .byte  196,98,125,24,5,166,5,0,0           // vbroadcastss  0x5a6(%rip),%ymm8        # 60ac <_sk_callback_avx+0x510>
   .byte  197,188,88,72,32                    // vaddps        0x20(%rax),%ymm8,%ymm1
   .byte  197,124,16,72,96                    // vmovups       0x60(%rax),%ymm9
-  .byte  196,98,125,24,21,155,5,0,0          // vbroadcastss  0x59b(%rip),%ymm10        # 6094 <_sk_callback_avx+0x518>
+  .byte  196,98,125,24,21,151,5,0,0          // vbroadcastss  0x597(%rip),%ymm10        # 60b0 <_sk_callback_avx+0x514>
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
-  .byte  196,98,125,24,29,145,5,0,0          // vbroadcastss  0x591(%rip),%ymm11        # 6098 <_sk_callback_avx+0x51c>
+  .byte  196,98,125,24,29,141,5,0,0          // vbroadcastss  0x58d(%rip),%ymm11        # 60b4 <_sk_callback_avx+0x518>
   .byte  196,65,44,88,211                    // vaddps        %ymm11,%ymm10,%ymm10
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
   .byte  196,65,44,88,192                    // vaddps        %ymm8,%ymm10,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
-  .byte  196,98,125,24,13,120,5,0,0          // vbroadcastss  0x578(%rip),%ymm9        # 609c <_sk_callback_avx+0x520>
+  .byte  196,98,125,24,13,116,5,0,0          // vbroadcastss  0x574(%rip),%ymm9        # 60b8 <_sk_callback_avx+0x51c>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -18981,13 +19015,13 @@ HIDDEN _sk_bicubic_p3y_avx
 FUNCTION(_sk_bicubic_p3y_avx)
 _sk_bicubic_p3y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,96,5,0,0          // vbroadcastss  0x560(%rip),%ymm1        # 60a0 <_sk_callback_avx+0x524>
+  .byte  196,226,125,24,13,92,5,0,0          // vbroadcastss  0x55c(%rip),%ymm1        # 60bc <_sk_callback_avx+0x520>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,76,5,0,0           // vbroadcastss  0x54c(%rip),%ymm10        # 60a4 <_sk_callback_avx+0x528>
+  .byte  196,98,125,24,21,72,5,0,0           // vbroadcastss  0x548(%rip),%ymm10        # 60c0 <_sk_callback_avx+0x524>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,66,5,0,0           // vbroadcastss  0x542(%rip),%ymm10        # 60a8 <_sk_callback_avx+0x52c>
+  .byte  196,98,125,24,21,62,5,0,0           // vbroadcastss  0x53e(%rip),%ymm10        # 60c4 <_sk_callback_avx+0x528>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -19112,25 +19146,25 @@ BALIGN4
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 5d59 <.literal4+0xb5>
+  .byte  71,225,61                           // rex.RXB       loope 5d79 <.literal4+0xb5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 5d69 <.literal4+0xc5>
+  .byte  71,225,61                           // rex.RXB       loope 5d89 <.literal4+0xc5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 5d79 <.literal4+0xd5>
+  .byte  71,225,61                           // rex.RXB       loope 5d99 <.literal4+0xd5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 5d89 <.literal4+0xe5>
+  .byte  71,225,61                           // rex.RXB       loope 5da9 <.literal4+0xe5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -19179,24 +19213,26 @@ BALIGN4
   .byte  190,129,128,128,59                  // mov           $0x3b808081,%esi
   .byte  129,128,128,59,0,248,0,0,8,33       // addl          $0x21080000,-0x7ffc480(%rax)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5dd1 <.literal4+0x12d>
+  .byte  224,7                               // loopne        5df1 <.literal4+0x12d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
   .byte  31                                  // (bad)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,8                                 // add           %cl,(%rax)
-  .byte  33,4,61,0,0,128,63                  // and           %eax,0x3f800000(,%rdi,1)
-  .byte  129,128,128,59,128,0,128,55,0,0     // addl          $0x3780,0x803b80(%rax)
+  .byte  33,4,61,129,128,128,59              // and           %eax,0x3b808081(,%rdi,1)
+  .byte  128,0,128                           // addb          $0x80,(%rax)
+  .byte  55                                  // (bad)
+  .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
   .byte  0,52,255                            // add           %dh,(%rdi,%rdi,8)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5dfc <.literal4+0x158>
+  .byte  127,0                               // jg            5e18 <.literal4+0x154>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5e75 <.literal4+0x1d1>
+  .byte  119,115                             // ja            5e91 <.literal4+0x1cd>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -19210,10 +19246,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5e30 <.literal4+0x18c>
+  .byte  127,0                               // jg            5e4c <.literal4+0x188>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5ea9 <.literal4+0x205>
+  .byte  119,115                             // ja            5ec5 <.literal4+0x201>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -19227,10 +19263,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5e64 <.literal4+0x1c0>
+  .byte  127,0                               // jg            5e80 <.literal4+0x1bc>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5edd <.literal4+0x239>
+  .byte  119,115                             // ja            5ef9 <.literal4+0x235>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -19244,10 +19280,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5e98 <.literal4+0x1f4>
+  .byte  127,0                               // jg            5eb4 <.literal4+0x1f0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5f11 <.literal4+0x26d>
+  .byte  119,115                             // ja            5f2d <.literal4+0x269>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -19260,7 +19296,7 @@ BALIGN4
   .byte  0,75,0                              // add           %cl,0x0(%rbx)
   .byte  0,128,63,0,0,200                    // add           %al,-0x37ffffc1(%rax)
   .byte  66,0,0                              // rex.X         add %al,(%rax)
-  .byte  127,67                              // jg            5f0f <.literal4+0x26b>
+  .byte  127,67                              // jg            5f2b <.literal4+0x267>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -19272,10 +19308,10 @@ BALIGN4
   .byte  190,80,128,3,62                     // mov           $0x3e038050,%esi
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           5f2f <.literal4+0x28b>
+  .byte  118,63                              // jbe           5f4b <.literal4+0x287>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            5f43 <.literal4+0x29f>
+  .byte  127,67                              // jg            5f5f <.literal4+0x29b>
   .byte  129,128,128,59,0,0,128,63,129,128   // addl          $0x80813f80,0x3b80(%rax)
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,128,63,129,128,128                // add           %al,-0x7f7f7ec1(%rax)
@@ -19284,7 +19320,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5f25 <.literal4+0x281>
+  .byte  224,7                               // loopne        5f41 <.literal4+0x27d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -19296,7 +19332,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5f41 <.literal4+0x29d>
+  .byte  224,7                               // loopne        5f5d <.literal4+0x299>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -19307,7 +19343,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            5f96 <.literal4+0x2f2>
+  .byte  124,66                              // jl            5fb2 <.literal4+0x2ee>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,55,0,15                 // mov           %ecx,0xf003788(%rax)
@@ -19325,9 +19361,9 @@ BALIGN4
   .byte  137,136,136,59,15,0                 // mov           %ecx,0xf3b88(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,61,0,0                  // mov           %ecx,0x3d88(%rax)
-  .byte  112,65                              // jo            5fd9 <.literal4+0x335>
+  .byte  112,65                              // jo            5ff5 <.literal4+0x331>
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            5fe7 <.literal4+0x343>
+  .byte  127,67                              // jg            6003 <.literal4+0x33f>
   .byte  0,128,0,0,0,0                       // add           %al,0x0(%rax)
   .byte  0,128,0,4,0,128                     // add           %al,-0x7ffffc00(%rax)
   .byte  0,0                                 // add           %al,(%rax)
@@ -19343,7 +19379,7 @@ BALIGN4
   .byte  0,128,55,0,0,128                    // add           %al,-0x7fffffc9(%rax)
   .byte  63                                  // (bad)
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            6027 <.literal4+0x383>
+  .byte  127,71                              // jg            6043 <.literal4+0x37f>
   .byte  208                                 // (bad)
   .byte  179,89                              // mov           $0x59,%bl
   .byte  62,89                               // ds            pop %rcx
@@ -19570,7 +19606,7 @@ _sk_seed_shader_sse41:
   .byte  102,15,110,199                      // movd          %edi,%xmm0
   .byte  102,15,112,192,0                    // pshufd        $0x0,%xmm0,%xmm0
   .byte  15,91,200                           // cvtdq2ps      %xmm0,%xmm1
-  .byte  15,40,21,212,66,0,0                 // movaps        0x42d4(%rip),%xmm2        # 4350 <_sk_callback_sse41+0xe4>
+  .byte  15,40,21,244,66,0,0                 // movaps        0x42f4(%rip),%xmm2        # 4370 <_sk_callback_sse41+0xe0>
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  15,16,2                             // movups        (%rdx),%xmm0
   .byte  15,88,193                           // addps         %xmm1,%xmm0
@@ -19579,7 +19615,7 @@ _sk_seed_shader_sse41:
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,21,195,66,0,0                 // movaps        0x42c3(%rip),%xmm2        # 4360 <_sk_callback_sse41+0xf4>
+  .byte  15,40,21,227,66,0,0                 // movaps        0x42e3(%rip),%xmm2        # 4380 <_sk_callback_sse41+0xf0>
   .byte  15,87,219                           // xorps         %xmm3,%xmm3
   .byte  15,87,228                           // xorps         %xmm4,%xmm4
   .byte  15,87,237                           // xorps         %xmm5,%xmm5
@@ -19602,14 +19638,14 @@ _sk_dither_sse41:
   .byte  102,68,15,110,1                     // movd          (%rcx),%xmm8
   .byte  102,69,15,112,192,0                 // pshufd        $0x0,%xmm8,%xmm8
   .byte  102,69,15,239,193                   // pxor          %xmm9,%xmm8
-  .byte  102,68,15,111,21,136,66,0,0         // movdqa        0x4288(%rip),%xmm10        # 4370 <_sk_callback_sse41+0x104>
+  .byte  102,68,15,111,21,168,66,0,0         // movdqa        0x42a8(%rip),%xmm10        # 4390 <_sk_callback_sse41+0x100>
   .byte  102,69,15,111,216                   // movdqa        %xmm8,%xmm11
   .byte  102,69,15,219,218                   // pand          %xmm10,%xmm11
   .byte  102,65,15,114,243,5                 // pslld         $0x5,%xmm11
   .byte  102,69,15,219,209                   // pand          %xmm9,%xmm10
   .byte  102,65,15,114,242,4                 // pslld         $0x4,%xmm10
-  .byte  102,68,15,111,37,116,66,0,0         // movdqa        0x4274(%rip),%xmm12        # 4380 <_sk_callback_sse41+0x114>
-  .byte  102,68,15,111,45,123,66,0,0         // movdqa        0x427b(%rip),%xmm13        # 4390 <_sk_callback_sse41+0x124>
+  .byte  102,68,15,111,37,148,66,0,0         // movdqa        0x4294(%rip),%xmm12        # 43a0 <_sk_callback_sse41+0x110>
+  .byte  102,68,15,111,45,155,66,0,0         // movdqa        0x429b(%rip),%xmm13        # 43b0 <_sk_callback_sse41+0x120>
   .byte  102,69,15,111,240                   // movdqa        %xmm8,%xmm14
   .byte  102,69,15,219,245                   // pand          %xmm13,%xmm14
   .byte  102,65,15,114,246,2                 // pslld         $0x2,%xmm14
@@ -19625,8 +19661,8 @@ _sk_dither_sse41:
   .byte  102,69,15,235,245                   // por           %xmm13,%xmm14
   .byte  102,69,15,235,240                   // por           %xmm8,%xmm14
   .byte  69,15,91,198                        // cvtdq2ps      %xmm14,%xmm8
-  .byte  68,15,89,5,54,66,0,0                // mulps         0x4236(%rip),%xmm8        # 43a0 <_sk_callback_sse41+0x134>
-  .byte  68,15,88,5,62,66,0,0                // addps         0x423e(%rip),%xmm8        # 43b0 <_sk_callback_sse41+0x144>
+  .byte  68,15,89,5,86,66,0,0                // mulps         0x4256(%rip),%xmm8        # 43c0 <_sk_callback_sse41+0x130>
+  .byte  68,15,88,5,94,66,0,0                // addps         0x425e(%rip),%xmm8        # 43d0 <_sk_callback_sse41+0x140>
   .byte  243,68,15,16,72,8                   // movss         0x8(%rax),%xmm9
   .byte  69,15,198,201,0                     // shufps        $0x0,%xmm9,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
@@ -19668,7 +19704,7 @@ HIDDEN _sk_srcatop_sse41
 FUNCTION(_sk_srcatop_sse41)
 _sk_srcatop_sse41:
   .byte  15,89,199                           // mulps         %xmm7,%xmm0
-  .byte  68,15,40,5,235,65,0,0               // movaps        0x41eb(%rip),%xmm8        # 43c0 <_sk_callback_sse41+0x154>
+  .byte  68,15,40,5,11,66,0,0                // movaps        0x420b(%rip),%xmm8        # 43e0 <_sk_callback_sse41+0x150>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -19693,7 +19729,7 @@ FUNCTION(_sk_dstatop_sse41)
 _sk_dstatop_sse41:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
   .byte  68,15,89,196                        // mulps         %xmm4,%xmm8
-  .byte  68,15,40,13,174,65,0,0              // movaps        0x41ae(%rip),%xmm9        # 43d0 <_sk_callback_sse41+0x164>
+  .byte  68,15,40,13,206,65,0,0              // movaps        0x41ce(%rip),%xmm9        # 43f0 <_sk_callback_sse41+0x160>
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
@@ -19740,7 +19776,7 @@ HIDDEN _sk_srcout_sse41
 .globl _sk_srcout_sse41
 FUNCTION(_sk_srcout_sse41)
 _sk_srcout_sse41:
-  .byte  68,15,40,5,82,65,0,0                // movaps        0x4152(%rip),%xmm8        # 43e0 <_sk_callback_sse41+0x174>
+  .byte  68,15,40,5,114,65,0,0               // movaps        0x4172(%rip),%xmm8        # 4400 <_sk_callback_sse41+0x170>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
@@ -19753,7 +19789,7 @@ HIDDEN _sk_dstout_sse41
 .globl _sk_dstout_sse41
 FUNCTION(_sk_dstout_sse41)
 _sk_dstout_sse41:
-  .byte  68,15,40,5,66,65,0,0                // movaps        0x4142(%rip),%xmm8        # 43f0 <_sk_callback_sse41+0x184>
+  .byte  68,15,40,5,98,65,0,0                // movaps        0x4162(%rip),%xmm8        # 4410 <_sk_callback_sse41+0x180>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  15,89,196                           // mulps         %xmm4,%xmm0
@@ -19770,7 +19806,7 @@ HIDDEN _sk_srcover_sse41
 .globl _sk_srcover_sse41
 FUNCTION(_sk_srcover_sse41)
 _sk_srcover_sse41:
-  .byte  68,15,40,5,37,65,0,0                // movaps        0x4125(%rip),%xmm8        # 4400 <_sk_callback_sse41+0x194>
+  .byte  68,15,40,5,69,65,0,0                // movaps        0x4145(%rip),%xmm8        # 4420 <_sk_callback_sse41+0x190>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -19790,7 +19826,7 @@ HIDDEN _sk_dstover_sse41
 .globl _sk_dstover_sse41
 FUNCTION(_sk_dstover_sse41)
 _sk_dstover_sse41:
-  .byte  68,15,40,5,249,64,0,0               // movaps        0x40f9(%rip),%xmm8        # 4410 <_sk_callback_sse41+0x1a4>
+  .byte  68,15,40,5,25,65,0,0                // movaps        0x4119(%rip),%xmm8        # 4430 <_sk_callback_sse41+0x1a0>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -19818,7 +19854,7 @@ HIDDEN _sk_multiply_sse41
 .globl _sk_multiply_sse41
 FUNCTION(_sk_multiply_sse41)
 _sk_multiply_sse41:
-  .byte  68,15,40,5,205,64,0,0               // movaps        0x40cd(%rip),%xmm8        # 4420 <_sk_callback_sse41+0x1b4>
+  .byte  68,15,40,5,237,64,0,0               // movaps        0x40ed(%rip),%xmm8        # 4440 <_sk_callback_sse41+0x1b0>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
@@ -19894,7 +19930,7 @@ HIDDEN _sk_xor__sse41
 FUNCTION(_sk_xor__sse41)
 _sk_xor__sse41:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
-  .byte  15,40,29,254,63,0,0                 // movaps        0x3ffe(%rip),%xmm3        # 4430 <_sk_callback_sse41+0x1c4>
+  .byte  15,40,29,30,64,0,0                  // movaps        0x401e(%rip),%xmm3        # 4450 <_sk_callback_sse41+0x1c0>
   .byte  68,15,40,203                        // movaps        %xmm3,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
@@ -19942,7 +19978,7 @@ _sk_darken_sse41:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,95,209                        // maxps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,105,63,0,0                 // movaps        0x3f69(%rip),%xmm2        # 4440 <_sk_callback_sse41+0x1d4>
+  .byte  15,40,21,137,63,0,0                 // movaps        0x3f89(%rip),%xmm2        # 4460 <_sk_callback_sse41+0x1d0>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -19976,7 +20012,7 @@ _sk_lighten_sse41:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,14,63,0,0                  // movaps        0x3f0e(%rip),%xmm2        # 4450 <_sk_callback_sse41+0x1e4>
+  .byte  15,40,21,46,63,0,0                  // movaps        0x3f2e(%rip),%xmm2        # 4470 <_sk_callback_sse41+0x1e0>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -20013,7 +20049,7 @@ _sk_difference_sse41:
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,168,62,0,0                 // movaps        0x3ea8(%rip),%xmm2        # 4460 <_sk_callback_sse41+0x1f4>
+  .byte  15,40,21,200,62,0,0                 // movaps        0x3ec8(%rip),%xmm2        # 4480 <_sk_callback_sse41+0x1f0>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -20040,7 +20076,7 @@ _sk_exclusion_sse41:
   .byte  15,89,214                           // mulps         %xmm6,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,202                        // subps         %xmm2,%xmm9
-  .byte  15,40,13,105,62,0,0                 // movaps        0x3e69(%rip),%xmm1        # 4470 <_sk_callback_sse41+0x204>
+  .byte  15,40,13,137,62,0,0                 // movaps        0x3e89(%rip),%xmm1        # 4490 <_sk_callback_sse41+0x200>
   .byte  15,92,203                           // subps         %xmm3,%xmm1
   .byte  15,89,207                           // mulps         %xmm7,%xmm1
   .byte  15,88,217                           // addps         %xmm1,%xmm3
@@ -20054,7 +20090,7 @@ HIDDEN _sk_colorburn_sse41
 FUNCTION(_sk_colorburn_sse41)
 _sk_colorburn_sse41:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,40,21,88,62,0,0               // movaps        0x3e58(%rip),%xmm10        # 4480 <_sk_callback_sse41+0x214>
+  .byte  68,15,40,21,120,62,0,0              // movaps        0x3e78(%rip),%xmm10        # 44a0 <_sk_callback_sse41+0x210>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  69,15,40,203                        // movaps        %xmm11,%xmm9
@@ -20136,7 +20172,7 @@ HIDDEN _sk_colordodge_sse41
 FUNCTION(_sk_colordodge_sse41)
 _sk_colordodge_sse41:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,40,21,54,61,0,0               // movaps        0x3d36(%rip),%xmm10        # 4490 <_sk_callback_sse41+0x224>
+  .byte  68,15,40,21,86,61,0,0               // movaps        0x3d56(%rip),%xmm10        # 44b0 <_sk_callback_sse41+0x220>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  69,15,40,227                        // movaps        %xmm11,%xmm12
@@ -20218,7 +20254,7 @@ _sk_hardlight_sse41:
   .byte  15,40,244                           // movaps        %xmm4,%xmm6
   .byte  15,40,227                           // movaps        %xmm3,%xmm4
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
-  .byte  68,15,40,21,15,60,0,0               // movaps        0x3c0f(%rip),%xmm10        # 44a0 <_sk_callback_sse41+0x234>
+  .byte  68,15,40,21,47,60,0,0               // movaps        0x3c2f(%rip),%xmm10        # 44c0 <_sk_callback_sse41+0x230>
   .byte  65,15,40,234                        // movaps        %xmm10,%xmm5
   .byte  15,92,239                           // subps         %xmm7,%xmm5
   .byte  15,40,197                           // movaps        %xmm5,%xmm0
@@ -20301,7 +20337,7 @@ FUNCTION(_sk_overlay_sse41)
 _sk_overlay_sse41:
   .byte  68,15,40,201                        // movaps        %xmm1,%xmm9
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
-  .byte  68,15,40,21,244,58,0,0              // movaps        0x3af4(%rip),%xmm10        # 44b0 <_sk_callback_sse41+0x244>
+  .byte  68,15,40,21,20,59,0,0               // movaps        0x3b14(%rip),%xmm10        # 44d0 <_sk_callback_sse41+0x240>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
@@ -20386,7 +20422,7 @@ _sk_softlight_sse41:
   .byte  15,40,198                           // movaps        %xmm6,%xmm0
   .byte  15,94,199                           // divps         %xmm7,%xmm0
   .byte  65,15,84,193                        // andps         %xmm9,%xmm0
-  .byte  15,40,13,203,57,0,0                 // movaps        0x39cb(%rip),%xmm1        # 44c0 <_sk_callback_sse41+0x254>
+  .byte  15,40,13,235,57,0,0                 // movaps        0x39eb(%rip),%xmm1        # 44e0 <_sk_callback_sse41+0x250>
   .byte  68,15,40,209                        // movaps        %xmm1,%xmm10
   .byte  68,15,92,208                        // subps         %xmm0,%xmm10
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
@@ -20399,10 +20435,10 @@ _sk_softlight_sse41:
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  15,89,210                           // mulps         %xmm2,%xmm2
   .byte  15,88,208                           // addps         %xmm0,%xmm2
-  .byte  68,15,40,45,169,57,0,0              // movaps        0x39a9(%rip),%xmm13        # 44d0 <_sk_callback_sse41+0x264>
+  .byte  68,15,40,45,201,57,0,0              // movaps        0x39c9(%rip),%xmm13        # 44f0 <_sk_callback_sse41+0x260>
   .byte  69,15,88,245                        // addps         %xmm13,%xmm14
   .byte  68,15,89,242                        // mulps         %xmm2,%xmm14
-  .byte  68,15,40,37,169,57,0,0              // movaps        0x39a9(%rip),%xmm12        # 44e0 <_sk_callback_sse41+0x274>
+  .byte  68,15,40,37,201,57,0,0              // movaps        0x39c9(%rip),%xmm12        # 4500 <_sk_callback_sse41+0x270>
   .byte  69,15,89,252                        // mulps         %xmm12,%xmm15
   .byte  69,15,88,254                        // addps         %xmm14,%xmm15
   .byte  15,40,198                           // movaps        %xmm6,%xmm0
@@ -20545,7 +20581,7 @@ _sk_hue_sse41:
   .byte  15,40,243                           // movaps        %xmm3,%xmm6
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,87,246                        // xorps         %xmm14,%xmm14
-  .byte  68,15,40,45,184,55,0,0              // movaps        0x37b8(%rip),%xmm13        # 44f0 <_sk_callback_sse41+0x284>
+  .byte  68,15,40,45,216,55,0,0              // movaps        0x37d8(%rip),%xmm13        # 4510 <_sk_callback_sse41+0x280>
   .byte  65,15,40,221                        // movaps        %xmm13,%xmm3
   .byte  15,94,222                           // divps         %xmm6,%xmm3
   .byte  15,40,198                           // movaps        %xmm6,%xmm0
@@ -20589,12 +20625,12 @@ _sk_hue_sse41:
   .byte  68,15,84,194                        // andps         %xmm2,%xmm8
   .byte  15,84,202                           // andps         %xmm2,%xmm1
   .byte  15,84,194                           // andps         %xmm2,%xmm0
-  .byte  68,15,40,13,39,55,0,0               // movaps        0x3727(%rip),%xmm9        # 4500 <_sk_callback_sse41+0x294>
+  .byte  68,15,40,13,71,55,0,0               // movaps        0x3747(%rip),%xmm9        # 4520 <_sk_callback_sse41+0x290>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  15,40,29,44,55,0,0                  // movaps        0x372c(%rip),%xmm3        # 4510 <_sk_callback_sse41+0x2a4>
+  .byte  15,40,29,76,55,0,0                  // movaps        0x374c(%rip),%xmm3        # 4530 <_sk_callback_sse41+0x2a0>
   .byte  68,15,89,219                        // mulps         %xmm3,%xmm11
   .byte  69,15,88,218                        // addps         %xmm10,%xmm11
-  .byte  68,15,40,53,44,55,0,0               // movaps        0x372c(%rip),%xmm14        # 4520 <_sk_callback_sse41+0x2b4>
+  .byte  68,15,40,53,76,55,0,0               // movaps        0x374c(%rip),%xmm14        # 4540 <_sk_callback_sse41+0x2b0>
   .byte  68,15,40,253                        // movaps        %xmm5,%xmm15
   .byte  69,15,89,254                        // mulps         %xmm14,%xmm15
   .byte  69,15,88,251                        // addps         %xmm11,%xmm15
@@ -20702,7 +20738,7 @@ _sk_saturation_sse41:
   .byte  68,15,40,220                        // movaps        %xmm4,%xmm11
   .byte  15,40,243                           // movaps        %xmm3,%xmm6
   .byte  69,15,87,246                        // xorps         %xmm14,%xmm14
-  .byte  68,15,40,37,165,53,0,0              // movaps        0x35a5(%rip),%xmm12        # 4530 <_sk_callback_sse41+0x2c4>
+  .byte  68,15,40,37,197,53,0,0              // movaps        0x35c5(%rip),%xmm12        # 4550 <_sk_callback_sse41+0x2c0>
   .byte  65,15,40,220                        // movaps        %xmm12,%xmm3
   .byte  15,94,223                           // divps         %xmm7,%xmm3
   .byte  68,15,40,199                        // movaps        %xmm7,%xmm8
@@ -20744,14 +20780,14 @@ _sk_saturation_sse41:
   .byte  68,15,84,202                        // andps         %xmm2,%xmm9
   .byte  68,15,84,234                        // andps         %xmm2,%xmm13
   .byte  68,15,84,194                        // andps         %xmm2,%xmm8
-  .byte  15,40,13,16,53,0,0                  // movaps        0x3510(%rip),%xmm1        # 4540 <_sk_callback_sse41+0x2d4>
+  .byte  15,40,13,48,53,0,0                  // movaps        0x3530(%rip),%xmm1        # 4560 <_sk_callback_sse41+0x2d0>
   .byte  65,15,40,211                        // movaps        %xmm11,%xmm2
   .byte  15,89,209                           // mulps         %xmm1,%xmm2
-  .byte  15,40,5,18,53,0,0                   // movaps        0x3512(%rip),%xmm0        # 4550 <_sk_callback_sse41+0x2e4>
+  .byte  15,40,5,50,53,0,0                   // movaps        0x3532(%rip),%xmm0        # 4570 <_sk_callback_sse41+0x2e0>
   .byte  15,40,221                           // movaps        %xmm5,%xmm3
   .byte  15,89,216                           // mulps         %xmm0,%xmm3
   .byte  15,88,218                           // addps         %xmm2,%xmm3
-  .byte  68,15,40,53,17,53,0,0               // movaps        0x3511(%rip),%xmm14        # 4560 <_sk_callback_sse41+0x2f4>
+  .byte  68,15,40,53,49,53,0,0               // movaps        0x3531(%rip),%xmm14        # 4580 <_sk_callback_sse41+0x2f0>
   .byte  69,15,40,250                        // movaps        %xmm10,%xmm15
   .byte  69,15,89,254                        // mulps         %xmm14,%xmm15
   .byte  68,15,88,251                        // addps         %xmm3,%xmm15
@@ -20859,7 +20895,7 @@ _sk_color_sse41:
   .byte  15,40,227                           // movaps        %xmm3,%xmm4
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,87,201                        // xorps         %xmm9,%xmm9
-  .byte  68,15,40,45,138,51,0,0              // movaps        0x338a(%rip),%xmm13        # 4570 <_sk_callback_sse41+0x304>
+  .byte  68,15,40,45,170,51,0,0              // movaps        0x33aa(%rip),%xmm13        # 4590 <_sk_callback_sse41+0x300>
   .byte  65,15,40,197                        // movaps        %xmm13,%xmm0
   .byte  15,94,196                           // divps         %xmm4,%xmm0
   .byte  65,15,194,217,4                     // cmpneqps      %xmm9,%xmm3
@@ -20867,13 +20903,13 @@ _sk_color_sse41:
   .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
   .byte  15,89,203                           // mulps         %xmm3,%xmm1
   .byte  15,89,218                           // mulps         %xmm2,%xmm3
-  .byte  68,15,40,13,121,51,0,0              // movaps        0x3379(%rip),%xmm9        # 4580 <_sk_callback_sse41+0x314>
+  .byte  68,15,40,13,153,51,0,0              // movaps        0x3399(%rip),%xmm9        # 45a0 <_sk_callback_sse41+0x310>
   .byte  15,40,213                           // movaps        %xmm5,%xmm2
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
-  .byte  68,15,40,21,122,51,0,0              // movaps        0x337a(%rip),%xmm10        # 4590 <_sk_callback_sse41+0x324>
+  .byte  68,15,40,21,154,51,0,0              // movaps        0x339a(%rip),%xmm10        # 45b0 <_sk_callback_sse41+0x320>
   .byte  69,15,89,218                        // mulps         %xmm10,%xmm11
   .byte  68,15,88,218                        // addps         %xmm2,%xmm11
-  .byte  68,15,40,53,122,51,0,0              // movaps        0x337a(%rip),%xmm14        # 45a0 <_sk_callback_sse41+0x334>
+  .byte  68,15,40,53,154,51,0,0              // movaps        0x339a(%rip),%xmm14        # 45c0 <_sk_callback_sse41+0x330>
   .byte  68,15,40,254                        // movaps        %xmm6,%xmm15
   .byte  69,15,89,254                        // mulps         %xmm14,%xmm15
   .byte  69,15,88,251                        // addps         %xmm11,%xmm15
@@ -20982,7 +21018,7 @@ _sk_luminosity_sse41:
   .byte  15,40,244                           // movaps        %xmm4,%xmm6
   .byte  15,40,235                           // movaps        %xmm3,%xmm5
   .byte  69,15,87,228                        // xorps         %xmm12,%xmm12
-  .byte  68,15,40,45,234,49,0,0              // movaps        0x31ea(%rip),%xmm13        # 45b0 <_sk_callback_sse41+0x344>
+  .byte  68,15,40,45,10,50,0,0               // movaps        0x320a(%rip),%xmm13        # 45d0 <_sk_callback_sse41+0x340>
   .byte  69,15,40,197                        // movaps        %xmm13,%xmm8
   .byte  68,15,94,199                        // divps         %xmm7,%xmm8
   .byte  15,40,223                           // movaps        %xmm7,%xmm3
@@ -20993,12 +21029,12 @@ _sk_luminosity_sse41:
   .byte  68,15,40,219                        // movaps        %xmm3,%xmm11
   .byte  69,15,89,222                        // mulps         %xmm14,%xmm11
   .byte  65,15,89,217                        // mulps         %xmm9,%xmm3
-  .byte  68,15,40,5,202,49,0,0               // movaps        0x31ca(%rip),%xmm8        # 45c0 <_sk_callback_sse41+0x354>
+  .byte  68,15,40,5,234,49,0,0               // movaps        0x31ea(%rip),%xmm8        # 45e0 <_sk_callback_sse41+0x350>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
-  .byte  68,15,40,13,206,49,0,0              // movaps        0x31ce(%rip),%xmm9        # 45d0 <_sk_callback_sse41+0x364>
+  .byte  68,15,40,13,238,49,0,0              // movaps        0x31ee(%rip),%xmm9        # 45f0 <_sk_callback_sse41+0x360>
   .byte  65,15,89,201                        // mulps         %xmm9,%xmm1
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  68,15,40,53,207,49,0,0              // movaps        0x31cf(%rip),%xmm14        # 45e0 <_sk_callback_sse41+0x374>
+  .byte  68,15,40,53,239,49,0,0              // movaps        0x31ef(%rip),%xmm14        # 4600 <_sk_callback_sse41+0x370>
   .byte  65,15,89,214                        // mulps         %xmm14,%xmm2
   .byte  15,88,209                           // addps         %xmm1,%xmm2
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
@@ -21111,7 +21147,7 @@ HIDDEN _sk_clamp_1_sse41
 .globl _sk_clamp_1_sse41
 FUNCTION(_sk_clamp_1_sse41)
 _sk_clamp_1_sse41:
-  .byte  68,15,40,5,73,48,0,0                // movaps        0x3049(%rip),%xmm8        # 45f0 <_sk_callback_sse41+0x384>
+  .byte  68,15,40,5,105,48,0,0               // movaps        0x3069(%rip),%xmm8        # 4610 <_sk_callback_sse41+0x380>
   .byte  65,15,93,192                        // minps         %xmm8,%xmm0
   .byte  65,15,93,200                        // minps         %xmm8,%xmm1
   .byte  65,15,93,208                        // minps         %xmm8,%xmm2
@@ -21123,7 +21159,7 @@ HIDDEN _sk_clamp_a_sse41
 .globl _sk_clamp_a_sse41
 FUNCTION(_sk_clamp_a_sse41)
 _sk_clamp_a_sse41:
-  .byte  15,93,29,62,48,0,0                  // minps         0x303e(%rip),%xmm3        # 4600 <_sk_callback_sse41+0x394>
+  .byte  15,93,29,94,48,0,0                  // minps         0x305e(%rip),%xmm3        # 4620 <_sk_callback_sse41+0x390>
   .byte  15,93,195                           // minps         %xmm3,%xmm0
   .byte  15,93,203                           // minps         %xmm3,%xmm1
   .byte  15,93,211                           // minps         %xmm3,%xmm2
@@ -21210,7 +21246,7 @@ HIDDEN _sk_unpremul_sse41
 FUNCTION(_sk_unpremul_sse41)
 _sk_unpremul_sse41:
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
-  .byte  68,15,40,13,169,47,0,0              // movaps        0x2fa9(%rip),%xmm9        # 4610 <_sk_callback_sse41+0x3a4>
+  .byte  68,15,40,13,201,47,0,0              // movaps        0x2fc9(%rip),%xmm9        # 4630 <_sk_callback_sse41+0x3a0>
   .byte  68,15,94,203                        // divps         %xmm3,%xmm9
   .byte  68,15,194,195,4                     // cmpneqps      %xmm3,%xmm8
   .byte  69,15,84,193                        // andps         %xmm9,%xmm8
@@ -21224,20 +21260,20 @@ HIDDEN _sk_from_srgb_sse41
 .globl _sk_from_srgb_sse41
 FUNCTION(_sk_from_srgb_sse41)
 _sk_from_srgb_sse41:
-  .byte  68,15,40,29,148,47,0,0              // movaps        0x2f94(%rip),%xmm11        # 4620 <_sk_callback_sse41+0x3b4>
+  .byte  68,15,40,29,180,47,0,0              // movaps        0x2fb4(%rip),%xmm11        # 4640 <_sk_callback_sse41+0x3b0>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
   .byte  68,15,40,208                        // movaps        %xmm0,%xmm10
   .byte  69,15,89,210                        // mulps         %xmm10,%xmm10
-  .byte  68,15,40,37,140,47,0,0              // movaps        0x2f8c(%rip),%xmm12        # 4630 <_sk_callback_sse41+0x3c4>
+  .byte  68,15,40,37,172,47,0,0              // movaps        0x2fac(%rip),%xmm12        # 4650 <_sk_callback_sse41+0x3c0>
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,89,196                        // mulps         %xmm12,%xmm8
-  .byte  68,15,40,45,140,47,0,0              // movaps        0x2f8c(%rip),%xmm13        # 4640 <_sk_callback_sse41+0x3d4>
+  .byte  68,15,40,45,172,47,0,0              // movaps        0x2fac(%rip),%xmm13        # 4660 <_sk_callback_sse41+0x3d0>
   .byte  69,15,88,197                        // addps         %xmm13,%xmm8
   .byte  69,15,89,194                        // mulps         %xmm10,%xmm8
-  .byte  68,15,40,53,140,47,0,0              // movaps        0x2f8c(%rip),%xmm14        # 4650 <_sk_callback_sse41+0x3e4>
+  .byte  68,15,40,53,172,47,0,0              // movaps        0x2fac(%rip),%xmm14        # 4670 <_sk_callback_sse41+0x3e0>
   .byte  69,15,88,198                        // addps         %xmm14,%xmm8
-  .byte  68,15,40,61,144,47,0,0              // movaps        0x2f90(%rip),%xmm15        # 4660 <_sk_callback_sse41+0x3f4>
+  .byte  68,15,40,61,176,47,0,0              // movaps        0x2fb0(%rip),%xmm15        # 4680 <_sk_callback_sse41+0x3f0>
   .byte  65,15,194,199,1                     // cmpltps       %xmm15,%xmm0
   .byte  102,69,15,56,20,193                 // blendvps      %xmm0,%xmm9,%xmm8
   .byte  68,15,40,209                        // movaps        %xmm1,%xmm10
@@ -21282,20 +21318,20 @@ _sk_to_srgb_sse41:
   .byte  68,15,82,192                        // rsqrtps       %xmm0,%xmm8
   .byte  69,15,83,200                        // rcpps         %xmm8,%xmm9
   .byte  69,15,82,208                        // rsqrtps       %xmm8,%xmm10
-  .byte  68,15,40,29,0,47,0,0                // movaps        0x2f00(%rip),%xmm11        # 4670 <_sk_callback_sse41+0x404>
+  .byte  68,15,40,29,32,47,0,0               // movaps        0x2f20(%rip),%xmm11        # 4690 <_sk_callback_sse41+0x400>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
-  .byte  68,15,40,37,1,47,0,0                // movaps        0x2f01(%rip),%xmm12        # 4680 <_sk_callback_sse41+0x414>
+  .byte  68,15,40,37,33,47,0,0               // movaps        0x2f21(%rip),%xmm12        # 46a0 <_sk_callback_sse41+0x410>
   .byte  69,15,89,204                        // mulps         %xmm12,%xmm9
-  .byte  68,15,40,45,5,47,0,0                // movaps        0x2f05(%rip),%xmm13        # 4690 <_sk_callback_sse41+0x424>
+  .byte  68,15,40,45,37,47,0,0               // movaps        0x2f25(%rip),%xmm13        # 46b0 <_sk_callback_sse41+0x420>
   .byte  69,15,88,205                        // addps         %xmm13,%xmm9
-  .byte  68,15,40,53,9,47,0,0                // movaps        0x2f09(%rip),%xmm14        # 46a0 <_sk_callback_sse41+0x434>
+  .byte  68,15,40,53,41,47,0,0               // movaps        0x2f29(%rip),%xmm14        # 46c0 <_sk_callback_sse41+0x430>
   .byte  69,15,89,214                        // mulps         %xmm14,%xmm10
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
-  .byte  68,15,40,5,9,47,0,0                 // movaps        0x2f09(%rip),%xmm8        # 46b0 <_sk_callback_sse41+0x444>
+  .byte  68,15,40,5,41,47,0,0                // movaps        0x2f29(%rip),%xmm8        # 46d0 <_sk_callback_sse41+0x440>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,93,202                        // minps         %xmm10,%xmm9
-  .byte  68,15,40,61,9,47,0,0                // movaps        0x2f09(%rip),%xmm15        # 46c0 <_sk_callback_sse41+0x454>
+  .byte  68,15,40,61,41,47,0,0               // movaps        0x2f29(%rip),%xmm15        # 46e0 <_sk_callback_sse41+0x450>
   .byte  65,15,194,199,1                     // cmpltps       %xmm15,%xmm0
   .byte  102,68,15,56,20,201                 // blendvps      %xmm0,%xmm1,%xmm9
   .byte  15,82,194                           // rsqrtps       %xmm2,%xmm0
@@ -21349,7 +21385,7 @@ _sk_rgb_to_hsl_sse41:
   .byte  68,15,93,226                        // minps         %xmm2,%xmm12
   .byte  65,15,40,203                        // movaps        %xmm11,%xmm1
   .byte  65,15,92,204                        // subps         %xmm12,%xmm1
-  .byte  68,15,40,53,90,46,0,0               // movaps        0x2e5a(%rip),%xmm14        # 46d0 <_sk_callback_sse41+0x464>
+  .byte  68,15,40,53,122,46,0,0              // movaps        0x2e7a(%rip),%xmm14        # 46f0 <_sk_callback_sse41+0x460>
   .byte  68,15,94,241                        // divps         %xmm1,%xmm14
   .byte  69,15,40,211                        // movaps        %xmm11,%xmm10
   .byte  69,15,194,208,0                     // cmpeqps       %xmm8,%xmm10
@@ -21358,27 +21394,27 @@ _sk_rgb_to_hsl_sse41:
   .byte  65,15,89,198                        // mulps         %xmm14,%xmm0
   .byte  69,15,40,249                        // movaps        %xmm9,%xmm15
   .byte  68,15,194,250,1                     // cmpltps       %xmm2,%xmm15
-  .byte  68,15,84,61,65,46,0,0               // andps         0x2e41(%rip),%xmm15        # 46e0 <_sk_callback_sse41+0x474>
+  .byte  68,15,84,61,97,46,0,0               // andps         0x2e61(%rip),%xmm15        # 4700 <_sk_callback_sse41+0x470>
   .byte  68,15,88,248                        // addps         %xmm0,%xmm15
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  65,15,194,193,0                     // cmpeqps       %xmm9,%xmm0
   .byte  65,15,92,208                        // subps         %xmm8,%xmm2
   .byte  65,15,89,214                        // mulps         %xmm14,%xmm2
-  .byte  68,15,40,45,52,46,0,0               // movaps        0x2e34(%rip),%xmm13        # 46f0 <_sk_callback_sse41+0x484>
+  .byte  68,15,40,45,84,46,0,0               // movaps        0x2e54(%rip),%xmm13        # 4710 <_sk_callback_sse41+0x480>
   .byte  65,15,88,213                        // addps         %xmm13,%xmm2
   .byte  69,15,92,193                        // subps         %xmm9,%xmm8
   .byte  69,15,89,198                        // mulps         %xmm14,%xmm8
-  .byte  68,15,88,5,48,46,0,0                // addps         0x2e30(%rip),%xmm8        # 4700 <_sk_callback_sse41+0x494>
+  .byte  68,15,88,5,80,46,0,0                // addps         0x2e50(%rip),%xmm8        # 4720 <_sk_callback_sse41+0x490>
   .byte  102,68,15,56,20,194                 // blendvps      %xmm0,%xmm2,%xmm8
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  102,69,15,56,20,199                 // blendvps      %xmm0,%xmm15,%xmm8
-  .byte  68,15,89,5,40,46,0,0                // mulps         0x2e28(%rip),%xmm8        # 4710 <_sk_callback_sse41+0x4a4>
+  .byte  68,15,89,5,72,46,0,0                // mulps         0x2e48(%rip),%xmm8        # 4730 <_sk_callback_sse41+0x4a0>
   .byte  69,15,40,203                        // movaps        %xmm11,%xmm9
   .byte  69,15,194,204,4                     // cmpneqps      %xmm12,%xmm9
   .byte  69,15,84,193                        // andps         %xmm9,%xmm8
   .byte  69,15,92,235                        // subps         %xmm11,%xmm13
   .byte  69,15,88,220                        // addps         %xmm12,%xmm11
-  .byte  15,40,5,28,46,0,0                   // movaps        0x2e1c(%rip),%xmm0        # 4720 <_sk_callback_sse41+0x4b4>
+  .byte  15,40,5,60,46,0,0                   // movaps        0x2e3c(%rip),%xmm0        # 4740 <_sk_callback_sse41+0x4b0>
   .byte  65,15,40,211                        // movaps        %xmm11,%xmm2
   .byte  15,89,208                           // mulps         %xmm0,%xmm2
   .byte  15,194,194,1                        // cmpltps       %xmm2,%xmm0
@@ -21400,7 +21436,7 @@ _sk_hsl_to_rgb_sse41:
   .byte  15,41,100,36,184                    // movaps        %xmm4,-0x48(%rsp)
   .byte  15,41,92,36,168                     // movaps        %xmm3,-0x58(%rsp)
   .byte  68,15,40,208                        // movaps        %xmm0,%xmm10
-  .byte  68,15,40,13,226,45,0,0              // movaps        0x2de2(%rip),%xmm9        # 4730 <_sk_callback_sse41+0x4c4>
+  .byte  68,15,40,13,2,46,0,0                // movaps        0x2e02(%rip),%xmm9        # 4750 <_sk_callback_sse41+0x4c0>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  15,194,194,2                        // cmpleps       %xmm2,%xmm0
   .byte  15,40,217                           // movaps        %xmm1,%xmm3
@@ -21413,19 +21449,19 @@ _sk_hsl_to_rgb_sse41:
   .byte  15,41,84,36,152                     // movaps        %xmm2,-0x68(%rsp)
   .byte  69,15,88,192                        // addps         %xmm8,%xmm8
   .byte  68,15,92,197                        // subps         %xmm5,%xmm8
-  .byte  68,15,40,53,189,45,0,0              // movaps        0x2dbd(%rip),%xmm14        # 4740 <_sk_callback_sse41+0x4d4>
+  .byte  68,15,40,53,221,45,0,0              // movaps        0x2ddd(%rip),%xmm14        # 4760 <_sk_callback_sse41+0x4d0>
   .byte  69,15,88,242                        // addps         %xmm10,%xmm14
   .byte  102,65,15,58,8,198,1                // roundps       $0x1,%xmm14,%xmm0
   .byte  68,15,92,240                        // subps         %xmm0,%xmm14
-  .byte  68,15,40,29,182,45,0,0              // movaps        0x2db6(%rip),%xmm11        # 4750 <_sk_callback_sse41+0x4e4>
+  .byte  68,15,40,29,214,45,0,0              // movaps        0x2dd6(%rip),%xmm11        # 4770 <_sk_callback_sse41+0x4e0>
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  65,15,194,198,2                     // cmpleps       %xmm14,%xmm0
   .byte  15,40,245                           // movaps        %xmm5,%xmm6
   .byte  65,15,92,240                        // subps         %xmm8,%xmm6
-  .byte  15,40,61,175,45,0,0                 // movaps        0x2daf(%rip),%xmm7        # 4760 <_sk_callback_sse41+0x4f4>
+  .byte  15,40,61,207,45,0,0                 // movaps        0x2dcf(%rip),%xmm7        # 4780 <_sk_callback_sse41+0x4f0>
   .byte  69,15,40,238                        // movaps        %xmm14,%xmm13
   .byte  68,15,89,239                        // mulps         %xmm7,%xmm13
-  .byte  15,40,29,176,45,0,0                 // movaps        0x2db0(%rip),%xmm3        # 4770 <_sk_callback_sse41+0x504>
+  .byte  15,40,29,208,45,0,0                 // movaps        0x2dd0(%rip),%xmm3        # 4790 <_sk_callback_sse41+0x500>
   .byte  68,15,40,227                        // movaps        %xmm3,%xmm12
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  68,15,89,230                        // mulps         %xmm6,%xmm12
@@ -21435,7 +21471,7 @@ _sk_hsl_to_rgb_sse41:
   .byte  65,15,194,198,2                     // cmpleps       %xmm14,%xmm0
   .byte  68,15,40,253                        // movaps        %xmm5,%xmm15
   .byte  102,69,15,56,20,252                 // blendvps      %xmm0,%xmm12,%xmm15
-  .byte  68,15,40,37,143,45,0,0              // movaps        0x2d8f(%rip),%xmm12        # 4780 <_sk_callback_sse41+0x514>
+  .byte  68,15,40,37,175,45,0,0              // movaps        0x2daf(%rip),%xmm12        # 47a0 <_sk_callback_sse41+0x510>
   .byte  65,15,40,196                        // movaps        %xmm12,%xmm0
   .byte  65,15,194,198,2                     // cmpleps       %xmm14,%xmm0
   .byte  68,15,89,238                        // mulps         %xmm6,%xmm13
@@ -21469,7 +21505,7 @@ _sk_hsl_to_rgb_sse41:
   .byte  65,15,40,198                        // movaps        %xmm14,%xmm0
   .byte  15,40,84,36,152                     // movaps        -0x68(%rsp),%xmm2
   .byte  102,15,56,20,202                    // blendvps      %xmm0,%xmm2,%xmm1
-  .byte  68,15,88,21,7,45,0,0                // addps         0x2d07(%rip),%xmm10        # 4790 <_sk_callback_sse41+0x524>
+  .byte  68,15,88,21,39,45,0,0               // addps         0x2d27(%rip),%xmm10        # 47b0 <_sk_callback_sse41+0x520>
   .byte  102,65,15,58,8,194,1                // roundps       $0x1,%xmm10,%xmm0
   .byte  68,15,92,208                        // subps         %xmm0,%xmm10
   .byte  69,15,194,218,2                     // cmpleps       %xmm10,%xmm11
@@ -21521,7 +21557,7 @@ _sk_scale_u8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,68,15,56,49,4,56                // pmovzxbd      (%rax,%rdi,1),%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,100,44,0,0               // mulps         0x2c64(%rip),%xmm8        # 47a0 <_sk_callback_sse41+0x534>
+  .byte  68,15,89,5,132,44,0,0               // mulps         0x2c84(%rip),%xmm8        # 47c0 <_sk_callback_sse41+0x530>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
@@ -21559,7 +21595,7 @@ _sk_lerp_u8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,68,15,56,49,4,56                // pmovzxbd      (%rax,%rdi,1),%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,16,44,0,0                // mulps         0x2c10(%rip),%xmm8        # 47b0 <_sk_callback_sse41+0x544>
+  .byte  68,15,89,5,48,44,0,0                // mulps         0x2c30(%rip),%xmm8        # 47d0 <_sk_callback_sse41+0x540>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -21581,29 +21617,38 @@ FUNCTION(_sk_lerp_565_sse41)
 _sk_lerp_565_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  102,68,15,56,51,4,120               // pmovzxwd      (%rax,%rdi,2),%xmm8
-  .byte  102,15,111,29,224,43,0,0            // movdqa        0x2be0(%rip),%xmm3        # 47c0 <_sk_callback_sse41+0x554>
-  .byte  102,65,15,219,216                   // pand          %xmm8,%xmm3
-  .byte  68,15,91,203                        // cvtdq2ps      %xmm3,%xmm9
-  .byte  68,15,89,13,223,43,0,0              // mulps         0x2bdf(%rip),%xmm9        # 47d0 <_sk_callback_sse41+0x564>
-  .byte  102,15,111,29,231,43,0,0            // movdqa        0x2be7(%rip),%xmm3        # 47e0 <_sk_callback_sse41+0x574>
-  .byte  102,65,15,219,216                   // pand          %xmm8,%xmm3
-  .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,232,43,0,0                 // mulps         0x2be8(%rip),%xmm3        # 47f0 <_sk_callback_sse41+0x584>
-  .byte  102,68,15,219,5,239,43,0,0          // pand          0x2bef(%rip),%xmm8        # 4800 <_sk_callback_sse41+0x594>
+  .byte  102,68,15,56,51,20,120              // pmovzxwd      (%rax,%rdi,2),%xmm10
+  .byte  102,68,15,111,5,255,43,0,0          // movdqa        0x2bff(%rip),%xmm8        # 47e0 <_sk_callback_sse41+0x550>
+  .byte  102,69,15,219,194                   // pand          %xmm10,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,243,43,0,0               // mulps         0x2bf3(%rip),%xmm8        # 4810 <_sk_callback_sse41+0x5a4>
+  .byte  68,15,89,5,254,43,0,0               // mulps         0x2bfe(%rip),%xmm8        # 47f0 <_sk_callback_sse41+0x560>
+  .byte  102,68,15,111,13,5,44,0,0           // movdqa        0x2c05(%rip),%xmm9        # 4800 <_sk_callback_sse41+0x570>
+  .byte  102,69,15,219,202                   // pand          %xmm10,%xmm9
+  .byte  69,15,91,201                        // cvtdq2ps      %xmm9,%xmm9
+  .byte  68,15,89,13,4,44,0,0                // mulps         0x2c04(%rip),%xmm9        # 4810 <_sk_callback_sse41+0x580>
+  .byte  102,68,15,219,21,11,44,0,0          // pand          0x2c0b(%rip),%xmm10        # 4820 <_sk_callback_sse41+0x590>
+  .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
+  .byte  68,15,89,21,15,44,0,0               // mulps         0x2c0f(%rip),%xmm10        # 4830 <_sk_callback_sse41+0x5a0>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
-  .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
+  .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
   .byte  15,92,205                           // subps         %xmm5,%xmm1
-  .byte  15,89,203                           // mulps         %xmm3,%xmm1
+  .byte  65,15,89,201                        // mulps         %xmm9,%xmm1
   .byte  15,88,205                           // addps         %xmm5,%xmm1
   .byte  15,92,214                           // subps         %xmm6,%xmm2
-  .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
+  .byte  65,15,89,210                        // mulps         %xmm10,%xmm2
   .byte  15,88,214                           // addps         %xmm6,%xmm2
+  .byte  15,92,223                           // subps         %xmm7,%xmm3
+  .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
+  .byte  68,15,88,199                        // addps         %xmm7,%xmm8
+  .byte  68,15,89,203                        // mulps         %xmm3,%xmm9
+  .byte  68,15,88,207                        // addps         %xmm7,%xmm9
+  .byte  65,15,89,218                        // mulps         %xmm10,%xmm3
+  .byte  15,88,223                           // addps         %xmm7,%xmm3
+  .byte  68,15,95,203                        // maxps         %xmm3,%xmm9
+  .byte  69,15,95,193                        // maxps         %xmm9,%xmm8
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,221,43,0,0                 // movaps        0x2bdd(%rip),%xmm3        # 4820 <_sk_callback_sse41+0x5b4>
+  .byte  65,15,40,216                        // movaps        %xmm8,%xmm3
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_load_tables_sse41
@@ -21614,7 +21659,7 @@ _sk_load_tables_sse41:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,139,72,8                         // mov           0x8(%rax),%r9
   .byte  243,69,15,111,4,184                 // movdqu        (%r8,%rdi,4),%xmm8
-  .byte  102,15,111,5,212,43,0,0             // movdqa        0x2bd4(%rip),%xmm0        # 4830 <_sk_callback_sse41+0x5c4>
+  .byte  102,15,111,5,192,43,0,0             // movdqa        0x2bc0(%rip),%xmm0        # 4840 <_sk_callback_sse41+0x5b0>
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,73,15,58,22,192,1               // pextrq        $0x1,%xmm0,%r8
   .byte  102,72,15,126,193                   // movq          %xmm0,%rcx
@@ -21629,7 +21674,7 @@ _sk_load_tables_sse41:
   .byte  102,15,58,33,193,48                 // insertps      $0x30,%xmm1,%xmm0
   .byte  76,139,64,16                        // mov           0x10(%rax),%r8
   .byte  102,65,15,111,200                   // movdqa        %xmm8,%xmm1
-  .byte  102,15,56,0,13,143,43,0,0           // pshufb        0x2b8f(%rip),%xmm1        # 4840 <_sk_callback_sse41+0x5d4>
+  .byte  102,15,56,0,13,123,43,0,0           // pshufb        0x2b7b(%rip),%xmm1        # 4850 <_sk_callback_sse41+0x5c0>
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
   .byte  102,72,15,126,201                   // movq          %xmm1,%rcx
   .byte  68,15,182,209                       // movzbl        %cl,%r10d
@@ -21644,7 +21689,7 @@ _sk_load_tables_sse41:
   .byte  102,15,58,33,202,48                 // insertps      $0x30,%xmm2,%xmm1
   .byte  76,139,64,24                        // mov           0x18(%rax),%r8
   .byte  102,65,15,111,208                   // movdqa        %xmm8,%xmm2
-  .byte  102,15,56,0,21,75,43,0,0            // pshufb        0x2b4b(%rip),%xmm2        # 4850 <_sk_callback_sse41+0x5e4>
+  .byte  102,15,56,0,21,55,43,0,0            // pshufb        0x2b37(%rip),%xmm2        # 4860 <_sk_callback_sse41+0x5d0>
   .byte  102,72,15,58,22,209,1               // pextrq        $0x1,%xmm2,%rcx
   .byte  102,72,15,126,208                   // movq          %xmm2,%rax
   .byte  68,15,182,200                       // movzbl        %al,%r9d
@@ -21659,7 +21704,7 @@ _sk_load_tables_sse41:
   .byte  102,15,58,33,211,48                 // insertps      $0x30,%xmm3,%xmm2
   .byte  102,65,15,114,208,24                // psrld         $0x18,%xmm8
   .byte  65,15,91,216                        // cvtdq2ps      %xmm8,%xmm3
-  .byte  15,89,29,8,43,0,0                   // mulps         0x2b08(%rip),%xmm3        # 4860 <_sk_callback_sse41+0x5f4>
+  .byte  15,89,29,244,42,0,0                 // mulps         0x2af4(%rip),%xmm3        # 4870 <_sk_callback_sse41+0x5e0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -21678,7 +21723,7 @@ _sk_load_tables_u16_be_sse41:
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,97,200                       // punpcklwd     %xmm0,%xmm1
   .byte  102,68,15,105,200                   // punpckhwd     %xmm0,%xmm9
-  .byte  102,68,15,111,5,219,42,0,0          // movdqa        0x2adb(%rip),%xmm8        # 4870 <_sk_callback_sse41+0x604>
+  .byte  102,68,15,111,5,199,42,0,0          // movdqa        0x2ac7(%rip),%xmm8        # 4880 <_sk_callback_sse41+0x5f0>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,15,56,51,192                    // pmovzxwd      %xmm0,%xmm0
@@ -21695,7 +21740,7 @@ _sk_load_tables_u16_be_sse41:
   .byte  243,67,15,16,20,8                   // movss         (%r8,%r9,1),%xmm2
   .byte  102,15,58,33,194,48                 // insertps      $0x30,%xmm2,%xmm0
   .byte  76,139,64,16                        // mov           0x10(%rax),%r8
-  .byte  102,15,56,0,13,142,42,0,0           // pshufb        0x2a8e(%rip),%xmm1        # 4880 <_sk_callback_sse41+0x614>
+  .byte  102,15,56,0,13,122,42,0,0           // pshufb        0x2a7a(%rip),%xmm1        # 4890 <_sk_callback_sse41+0x600>
   .byte  102,15,56,51,201                    // pmovzxwd      %xmm1,%xmm1
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
   .byte  102,72,15,126,201                   // movq          %xmm1,%rcx
@@ -21731,7 +21776,7 @@ _sk_load_tables_u16_be_sse41:
   .byte  102,65,15,235,216                   // por           %xmm8,%xmm3
   .byte  102,15,56,51,219                    // pmovzxwd      %xmm3,%xmm3
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,220,41,0,0                 // mulps         0x29dc(%rip),%xmm3        # 4890 <_sk_callback_sse41+0x624>
+  .byte  15,89,29,200,41,0,0                 // mulps         0x29c8(%rip),%xmm3        # 48a0 <_sk_callback_sse41+0x610>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -21753,7 +21798,7 @@ _sk_load_tables_rgb_u16_be_sse41:
   .byte  102,68,15,97,200                    // punpcklwd     %xmm0,%xmm9
   .byte  102,15,111,202                      // movdqa        %xmm2,%xmm1
   .byte  102,65,15,97,201                    // punpcklwd     %xmm9,%xmm1
-  .byte  102,68,15,111,5,158,41,0,0          // movdqa        0x299e(%rip),%xmm8        # 48a0 <_sk_callback_sse41+0x634>
+  .byte  102,68,15,111,5,138,41,0,0          // movdqa        0x298a(%rip),%xmm8        # 48b0 <_sk_callback_sse41+0x620>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,15,56,51,192                    // pmovzxwd      %xmm0,%xmm0
@@ -21770,7 +21815,7 @@ _sk_load_tables_rgb_u16_be_sse41:
   .byte  243,67,15,16,28,8                   // movss         (%r8,%r9,1),%xmm3
   .byte  102,15,58,33,195,48                 // insertps      $0x30,%xmm3,%xmm0
   .byte  76,139,64,16                        // mov           0x10(%rax),%r8
-  .byte  102,15,56,0,13,81,41,0,0            // pshufb        0x2951(%rip),%xmm1        # 48b0 <_sk_callback_sse41+0x644>
+  .byte  102,15,56,0,13,61,41,0,0            // pshufb        0x293d(%rip),%xmm1        # 48c0 <_sk_callback_sse41+0x630>
   .byte  102,15,56,51,201                    // pmovzxwd      %xmm1,%xmm1
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
   .byte  102,72,15,126,201                   // movq          %xmm1,%rcx
@@ -21801,7 +21846,7 @@ _sk_load_tables_rgb_u16_be_sse41:
   .byte  243,65,15,16,28,8                   // movss         (%r8,%rcx,1),%xmm3
   .byte  102,15,58,33,211,48                 // insertps      $0x30,%xmm3,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,188,40,0,0                 // movaps        0x28bc(%rip),%xmm3        # 48c0 <_sk_callback_sse41+0x654>
+  .byte  15,40,29,168,40,0,0                 // movaps        0x28a8(%rip),%xmm3        # 48d0 <_sk_callback_sse41+0x640>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_byte_tables_sse41
@@ -21811,7 +21856,7 @@ _sk_byte_tables_sse41:
   .byte  65,86                               // push          %r14
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,189,40,0,0               // movaps        0x28bd(%rip),%xmm8        # 48d0 <_sk_callback_sse41+0x664>
+  .byte  68,15,40,5,169,40,0,0               // movaps        0x28a9(%rip),%xmm8        # 48e0 <_sk_callback_sse41+0x650>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,91,192                       // cvtps2dq      %xmm0,%xmm0
   .byte  102,72,15,58,22,193,1               // pextrq        $0x1,%xmm0,%rcx
@@ -21830,7 +21875,7 @@ _sk_byte_tables_sse41:
   .byte  102,15,58,32,193,3                  // pinsrb        $0x3,%ecx,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,13,110,40,0,0              // movaps        0x286e(%rip),%xmm9        # 48e0 <_sk_callback_sse41+0x674>
+  .byte  68,15,40,13,90,40,0,0               // movaps        0x285a(%rip),%xmm9        # 48f0 <_sk_callback_sse41+0x660>
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -21921,7 +21966,7 @@ _sk_byte_tables_rgb_sse41:
   .byte  102,15,58,32,193,3                  // pinsrb        $0x3,%ecx,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,13,246,38,0,0              // movaps        0x26f6(%rip),%xmm9        # 48f0 <_sk_callback_sse41+0x684>
+  .byte  68,15,40,13,226,38,0,0              // movaps        0x26e2(%rip),%xmm9        # 4900 <_sk_callback_sse41+0x670>
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -22098,31 +22143,31 @@ _sk_parametric_r_sse41:
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,194                        // cvtdq2ps      %xmm10,%xmm8
-  .byte  68,15,89,5,77,36,0,0                // mulps         0x244d(%rip),%xmm8        # 4900 <_sk_callback_sse41+0x694>
-  .byte  68,15,84,21,85,36,0,0               // andps         0x2455(%rip),%xmm10        # 4910 <_sk_callback_sse41+0x6a4>
-  .byte  68,15,86,21,93,36,0,0               // orps          0x245d(%rip),%xmm10        # 4920 <_sk_callback_sse41+0x6b4>
-  .byte  68,15,88,5,101,36,0,0               // addps         0x2465(%rip),%xmm8        # 4930 <_sk_callback_sse41+0x6c4>
-  .byte  68,15,40,37,109,36,0,0              // movaps        0x246d(%rip),%xmm12        # 4940 <_sk_callback_sse41+0x6d4>
+  .byte  68,15,89,5,57,36,0,0                // mulps         0x2439(%rip),%xmm8        # 4910 <_sk_callback_sse41+0x680>
+  .byte  68,15,84,21,65,36,0,0               // andps         0x2441(%rip),%xmm10        # 4920 <_sk_callback_sse41+0x690>
+  .byte  68,15,86,21,73,36,0,0               // orps          0x2449(%rip),%xmm10        # 4930 <_sk_callback_sse41+0x6a0>
+  .byte  68,15,88,5,81,36,0,0                // addps         0x2451(%rip),%xmm8        # 4940 <_sk_callback_sse41+0x6b0>
+  .byte  68,15,40,37,89,36,0,0               // movaps        0x2459(%rip),%xmm12        # 4950 <_sk_callback_sse41+0x6c0>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,196                        // subps         %xmm12,%xmm8
-  .byte  68,15,88,21,109,36,0,0              // addps         0x246d(%rip),%xmm10        # 4950 <_sk_callback_sse41+0x6e4>
-  .byte  68,15,40,37,117,36,0,0              // movaps        0x2475(%rip),%xmm12        # 4960 <_sk_callback_sse41+0x6f4>
+  .byte  68,15,88,21,89,36,0,0               // addps         0x2459(%rip),%xmm10        # 4960 <_sk_callback_sse41+0x6d0>
+  .byte  68,15,40,37,97,36,0,0               // movaps        0x2461(%rip),%xmm12        # 4970 <_sk_callback_sse41+0x6e0>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,196                        // subps         %xmm12,%xmm8
   .byte  69,15,89,195                        // mulps         %xmm11,%xmm8
   .byte  102,69,15,58,8,208,1                // roundps       $0x1,%xmm8,%xmm10
   .byte  69,15,40,216                        // movaps        %xmm8,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,5,98,36,0,0                // addps         0x2462(%rip),%xmm8        # 4970 <_sk_callback_sse41+0x704>
-  .byte  68,15,40,21,106,36,0,0              // movaps        0x246a(%rip),%xmm10        # 4980 <_sk_callback_sse41+0x714>
+  .byte  68,15,88,5,78,36,0,0                // addps         0x244e(%rip),%xmm8        # 4980 <_sk_callback_sse41+0x6f0>
+  .byte  68,15,40,21,86,36,0,0               // movaps        0x2456(%rip),%xmm10        # 4990 <_sk_callback_sse41+0x700>
   .byte  69,15,89,211                        // mulps         %xmm11,%xmm10
   .byte  69,15,92,194                        // subps         %xmm10,%xmm8
-  .byte  68,15,40,21,106,36,0,0              // movaps        0x246a(%rip),%xmm10        # 4990 <_sk_callback_sse41+0x724>
+  .byte  68,15,40,21,86,36,0,0               // movaps        0x2456(%rip),%xmm10        # 49a0 <_sk_callback_sse41+0x710>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  68,15,40,29,110,36,0,0              // movaps        0x246e(%rip),%xmm11        # 49a0 <_sk_callback_sse41+0x734>
+  .byte  68,15,40,29,90,36,0,0               // movaps        0x245a(%rip),%xmm11        # 49b0 <_sk_callback_sse41+0x720>
   .byte  69,15,94,218                        // divps         %xmm10,%xmm11
   .byte  69,15,88,216                        // addps         %xmm8,%xmm11
-  .byte  68,15,89,29,110,36,0,0              // mulps         0x246e(%rip),%xmm11        # 49b0 <_sk_callback_sse41+0x744>
+  .byte  68,15,89,29,90,36,0,0               // mulps         0x245a(%rip),%xmm11        # 49c0 <_sk_callback_sse41+0x730>
   .byte  102,69,15,91,211                    // cvtps2dq      %xmm11,%xmm10
   .byte  243,68,15,16,64,20                  // movss         0x14(%rax),%xmm8
   .byte  69,15,198,192,0                     // shufps        $0x0,%xmm8,%xmm8
@@ -22130,7 +22175,7 @@ _sk_parametric_r_sse41:
   .byte  102,69,15,56,20,193                 // blendvps      %xmm0,%xmm9,%xmm8
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  68,15,95,192                        // maxps         %xmm0,%xmm8
-  .byte  68,15,93,5,85,36,0,0                // minps         0x2455(%rip),%xmm8        # 49c0 <_sk_callback_sse41+0x754>
+  .byte  68,15,93,5,65,36,0,0                // minps         0x2441(%rip),%xmm8        # 49d0 <_sk_callback_sse41+0x740>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -22160,31 +22205,31 @@ _sk_parametric_g_sse41:
   .byte  68,15,88,217                        // addps         %xmm1,%xmm11
   .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
   .byte  69,15,91,227                        // cvtdq2ps      %xmm11,%xmm12
-  .byte  68,15,89,37,246,35,0,0              // mulps         0x23f6(%rip),%xmm12        # 49d0 <_sk_callback_sse41+0x764>
-  .byte  68,15,84,29,254,35,0,0              // andps         0x23fe(%rip),%xmm11        # 49e0 <_sk_callback_sse41+0x774>
-  .byte  68,15,86,29,6,36,0,0                // orps          0x2406(%rip),%xmm11        # 49f0 <_sk_callback_sse41+0x784>
-  .byte  68,15,88,37,14,36,0,0               // addps         0x240e(%rip),%xmm12        # 4a00 <_sk_callback_sse41+0x794>
-  .byte  15,40,13,23,36,0,0                  // movaps        0x2417(%rip),%xmm1        # 4a10 <_sk_callback_sse41+0x7a4>
+  .byte  68,15,89,37,226,35,0,0              // mulps         0x23e2(%rip),%xmm12        # 49e0 <_sk_callback_sse41+0x750>
+  .byte  68,15,84,29,234,35,0,0              // andps         0x23ea(%rip),%xmm11        # 49f0 <_sk_callback_sse41+0x760>
+  .byte  68,15,86,29,242,35,0,0              // orps          0x23f2(%rip),%xmm11        # 4a00 <_sk_callback_sse41+0x770>
+  .byte  68,15,88,37,250,35,0,0              // addps         0x23fa(%rip),%xmm12        # 4a10 <_sk_callback_sse41+0x780>
+  .byte  15,40,13,3,36,0,0                   // movaps        0x2403(%rip),%xmm1        # 4a20 <_sk_callback_sse41+0x790>
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
   .byte  68,15,92,225                        // subps         %xmm1,%xmm12
-  .byte  68,15,88,29,23,36,0,0               // addps         0x2417(%rip),%xmm11        # 4a20 <_sk_callback_sse41+0x7b4>
-  .byte  15,40,13,32,36,0,0                  // movaps        0x2420(%rip),%xmm1        # 4a30 <_sk_callback_sse41+0x7c4>
+  .byte  68,15,88,29,3,36,0,0                // addps         0x2403(%rip),%xmm11        # 4a30 <_sk_callback_sse41+0x7a0>
+  .byte  15,40,13,12,36,0,0                  // movaps        0x240c(%rip),%xmm1        # 4a40 <_sk_callback_sse41+0x7b0>
   .byte  65,15,94,203                        // divps         %xmm11,%xmm1
   .byte  68,15,92,225                        // subps         %xmm1,%xmm12
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  102,69,15,58,8,212,1                // roundps       $0x1,%xmm12,%xmm10
   .byte  69,15,40,220                        // movaps        %xmm12,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,37,13,36,0,0               // addps         0x240d(%rip),%xmm12        # 4a40 <_sk_callback_sse41+0x7d4>
-  .byte  15,40,13,22,36,0,0                  // movaps        0x2416(%rip),%xmm1        # 4a50 <_sk_callback_sse41+0x7e4>
+  .byte  68,15,88,37,249,35,0,0              // addps         0x23f9(%rip),%xmm12        # 4a50 <_sk_callback_sse41+0x7c0>
+  .byte  15,40,13,2,36,0,0                   // movaps        0x2402(%rip),%xmm1        # 4a60 <_sk_callback_sse41+0x7d0>
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
   .byte  68,15,92,225                        // subps         %xmm1,%xmm12
-  .byte  68,15,40,21,22,36,0,0               // movaps        0x2416(%rip),%xmm10        # 4a60 <_sk_callback_sse41+0x7f4>
+  .byte  68,15,40,21,2,36,0,0                // movaps        0x2402(%rip),%xmm10        # 4a70 <_sk_callback_sse41+0x7e0>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  15,40,13,27,36,0,0                  // movaps        0x241b(%rip),%xmm1        # 4a70 <_sk_callback_sse41+0x804>
+  .byte  15,40,13,7,36,0,0                   // movaps        0x2407(%rip),%xmm1        # 4a80 <_sk_callback_sse41+0x7f0>
   .byte  65,15,94,202                        // divps         %xmm10,%xmm1
   .byte  65,15,88,204                        // addps         %xmm12,%xmm1
-  .byte  15,89,13,28,36,0,0                  // mulps         0x241c(%rip),%xmm1        # 4a80 <_sk_callback_sse41+0x814>
+  .byte  15,89,13,8,36,0,0                   // mulps         0x2408(%rip),%xmm1        # 4a90 <_sk_callback_sse41+0x800>
   .byte  102,68,15,91,209                    // cvtps2dq      %xmm1,%xmm10
   .byte  243,15,16,72,20                     // movss         0x14(%rax),%xmm1
   .byte  15,198,201,0                        // shufps        $0x0,%xmm1,%xmm1
@@ -22192,7 +22237,7 @@ _sk_parametric_g_sse41:
   .byte  102,65,15,56,20,201                 // blendvps      %xmm0,%xmm9,%xmm1
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,95,200                           // maxps         %xmm0,%xmm1
-  .byte  15,93,13,7,36,0,0                   // minps         0x2407(%rip),%xmm1        # 4a90 <_sk_callback_sse41+0x824>
+  .byte  15,93,13,243,35,0,0                 // minps         0x23f3(%rip),%xmm1        # 4aa0 <_sk_callback_sse41+0x810>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -22222,31 +22267,31 @@ _sk_parametric_b_sse41:
   .byte  68,15,88,218                        // addps         %xmm2,%xmm11
   .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
   .byte  69,15,91,227                        // cvtdq2ps      %xmm11,%xmm12
-  .byte  68,15,89,37,168,35,0,0              // mulps         0x23a8(%rip),%xmm12        # 4aa0 <_sk_callback_sse41+0x834>
-  .byte  68,15,84,29,176,35,0,0              // andps         0x23b0(%rip),%xmm11        # 4ab0 <_sk_callback_sse41+0x844>
-  .byte  68,15,86,29,184,35,0,0              // orps          0x23b8(%rip),%xmm11        # 4ac0 <_sk_callback_sse41+0x854>
-  .byte  68,15,88,37,192,35,0,0              // addps         0x23c0(%rip),%xmm12        # 4ad0 <_sk_callback_sse41+0x864>
-  .byte  15,40,21,201,35,0,0                 // movaps        0x23c9(%rip),%xmm2        # 4ae0 <_sk_callback_sse41+0x874>
+  .byte  68,15,89,37,148,35,0,0              // mulps         0x2394(%rip),%xmm12        # 4ab0 <_sk_callback_sse41+0x820>
+  .byte  68,15,84,29,156,35,0,0              // andps         0x239c(%rip),%xmm11        # 4ac0 <_sk_callback_sse41+0x830>
+  .byte  68,15,86,29,164,35,0,0              // orps          0x23a4(%rip),%xmm11        # 4ad0 <_sk_callback_sse41+0x840>
+  .byte  68,15,88,37,172,35,0,0              // addps         0x23ac(%rip),%xmm12        # 4ae0 <_sk_callback_sse41+0x850>
+  .byte  15,40,21,181,35,0,0                 // movaps        0x23b5(%rip),%xmm2        # 4af0 <_sk_callback_sse41+0x860>
   .byte  65,15,89,211                        // mulps         %xmm11,%xmm2
   .byte  68,15,92,226                        // subps         %xmm2,%xmm12
-  .byte  68,15,88,29,201,35,0,0              // addps         0x23c9(%rip),%xmm11        # 4af0 <_sk_callback_sse41+0x884>
-  .byte  15,40,21,210,35,0,0                 // movaps        0x23d2(%rip),%xmm2        # 4b00 <_sk_callback_sse41+0x894>
+  .byte  68,15,88,29,181,35,0,0              // addps         0x23b5(%rip),%xmm11        # 4b00 <_sk_callback_sse41+0x870>
+  .byte  15,40,21,190,35,0,0                 // movaps        0x23be(%rip),%xmm2        # 4b10 <_sk_callback_sse41+0x880>
   .byte  65,15,94,211                        // divps         %xmm11,%xmm2
   .byte  68,15,92,226                        // subps         %xmm2,%xmm12
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  102,69,15,58,8,212,1                // roundps       $0x1,%xmm12,%xmm10
   .byte  69,15,40,220                        // movaps        %xmm12,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,37,191,35,0,0              // addps         0x23bf(%rip),%xmm12        # 4b10 <_sk_callback_sse41+0x8a4>
-  .byte  15,40,21,200,35,0,0                 // movaps        0x23c8(%rip),%xmm2        # 4b20 <_sk_callback_sse41+0x8b4>
+  .byte  68,15,88,37,171,35,0,0              // addps         0x23ab(%rip),%xmm12        # 4b20 <_sk_callback_sse41+0x890>
+  .byte  15,40,21,180,35,0,0                 // movaps        0x23b4(%rip),%xmm2        # 4b30 <_sk_callback_sse41+0x8a0>
   .byte  65,15,89,211                        // mulps         %xmm11,%xmm2
   .byte  68,15,92,226                        // subps         %xmm2,%xmm12
-  .byte  68,15,40,21,200,35,0,0              // movaps        0x23c8(%rip),%xmm10        # 4b30 <_sk_callback_sse41+0x8c4>
+  .byte  68,15,40,21,180,35,0,0              // movaps        0x23b4(%rip),%xmm10        # 4b40 <_sk_callback_sse41+0x8b0>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  15,40,21,205,35,0,0                 // movaps        0x23cd(%rip),%xmm2        # 4b40 <_sk_callback_sse41+0x8d4>
+  .byte  15,40,21,185,35,0,0                 // movaps        0x23b9(%rip),%xmm2        # 4b50 <_sk_callback_sse41+0x8c0>
   .byte  65,15,94,210                        // divps         %xmm10,%xmm2
   .byte  65,15,88,212                        // addps         %xmm12,%xmm2
-  .byte  15,89,21,206,35,0,0                 // mulps         0x23ce(%rip),%xmm2        # 4b50 <_sk_callback_sse41+0x8e4>
+  .byte  15,89,21,186,35,0,0                 // mulps         0x23ba(%rip),%xmm2        # 4b60 <_sk_callback_sse41+0x8d0>
   .byte  102,68,15,91,210                    // cvtps2dq      %xmm2,%xmm10
   .byte  243,15,16,80,20                     // movss         0x14(%rax),%xmm2
   .byte  15,198,210,0                        // shufps        $0x0,%xmm2,%xmm2
@@ -22254,7 +22299,7 @@ _sk_parametric_b_sse41:
   .byte  102,65,15,56,20,209                 // blendvps      %xmm0,%xmm9,%xmm2
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,95,208                           // maxps         %xmm0,%xmm2
-  .byte  15,93,21,185,35,0,0                 // minps         0x23b9(%rip),%xmm2        # 4b60 <_sk_callback_sse41+0x8f4>
+  .byte  15,93,21,165,35,0,0                 // minps         0x23a5(%rip),%xmm2        # 4b70 <_sk_callback_sse41+0x8e0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -22284,31 +22329,31 @@ _sk_parametric_a_sse41:
   .byte  68,15,88,219                        // addps         %xmm3,%xmm11
   .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
   .byte  69,15,91,227                        // cvtdq2ps      %xmm11,%xmm12
-  .byte  68,15,89,37,90,35,0,0               // mulps         0x235a(%rip),%xmm12        # 4b70 <_sk_callback_sse41+0x904>
-  .byte  68,15,84,29,98,35,0,0               // andps         0x2362(%rip),%xmm11        # 4b80 <_sk_callback_sse41+0x914>
-  .byte  68,15,86,29,106,35,0,0              // orps          0x236a(%rip),%xmm11        # 4b90 <_sk_callback_sse41+0x924>
-  .byte  68,15,88,37,114,35,0,0              // addps         0x2372(%rip),%xmm12        # 4ba0 <_sk_callback_sse41+0x934>
-  .byte  15,40,29,123,35,0,0                 // movaps        0x237b(%rip),%xmm3        # 4bb0 <_sk_callback_sse41+0x944>
+  .byte  68,15,89,37,70,35,0,0               // mulps         0x2346(%rip),%xmm12        # 4b80 <_sk_callback_sse41+0x8f0>
+  .byte  68,15,84,29,78,35,0,0               // andps         0x234e(%rip),%xmm11        # 4b90 <_sk_callback_sse41+0x900>
+  .byte  68,15,86,29,86,35,0,0               // orps          0x2356(%rip),%xmm11        # 4ba0 <_sk_callback_sse41+0x910>
+  .byte  68,15,88,37,94,35,0,0               // addps         0x235e(%rip),%xmm12        # 4bb0 <_sk_callback_sse41+0x920>
+  .byte  15,40,29,103,35,0,0                 // movaps        0x2367(%rip),%xmm3        # 4bc0 <_sk_callback_sse41+0x930>
   .byte  65,15,89,219                        // mulps         %xmm11,%xmm3
   .byte  68,15,92,227                        // subps         %xmm3,%xmm12
-  .byte  68,15,88,29,123,35,0,0              // addps         0x237b(%rip),%xmm11        # 4bc0 <_sk_callback_sse41+0x954>
-  .byte  15,40,29,132,35,0,0                 // movaps        0x2384(%rip),%xmm3        # 4bd0 <_sk_callback_sse41+0x964>
+  .byte  68,15,88,29,103,35,0,0              // addps         0x2367(%rip),%xmm11        # 4bd0 <_sk_callback_sse41+0x940>
+  .byte  15,40,29,112,35,0,0                 // movaps        0x2370(%rip),%xmm3        # 4be0 <_sk_callback_sse41+0x950>
   .byte  65,15,94,219                        // divps         %xmm11,%xmm3
   .byte  68,15,92,227                        // subps         %xmm3,%xmm12
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  102,69,15,58,8,212,1                // roundps       $0x1,%xmm12,%xmm10
   .byte  69,15,40,220                        // movaps        %xmm12,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,37,113,35,0,0              // addps         0x2371(%rip),%xmm12        # 4be0 <_sk_callback_sse41+0x974>
-  .byte  15,40,29,122,35,0,0                 // movaps        0x237a(%rip),%xmm3        # 4bf0 <_sk_callback_sse41+0x984>
+  .byte  68,15,88,37,93,35,0,0               // addps         0x235d(%rip),%xmm12        # 4bf0 <_sk_callback_sse41+0x960>
+  .byte  15,40,29,102,35,0,0                 // movaps        0x2366(%rip),%xmm3        # 4c00 <_sk_callback_sse41+0x970>
   .byte  65,15,89,219                        // mulps         %xmm11,%xmm3
   .byte  68,15,92,227                        // subps         %xmm3,%xmm12
-  .byte  68,15,40,21,122,35,0,0              // movaps        0x237a(%rip),%xmm10        # 4c00 <_sk_callback_sse41+0x994>
+  .byte  68,15,40,21,102,35,0,0              // movaps        0x2366(%rip),%xmm10        # 4c10 <_sk_callback_sse41+0x980>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  15,40,29,127,35,0,0                 // movaps        0x237f(%rip),%xmm3        # 4c10 <_sk_callback_sse41+0x9a4>
+  .byte  15,40,29,107,35,0,0                 // movaps        0x236b(%rip),%xmm3        # 4c20 <_sk_callback_sse41+0x990>
   .byte  65,15,94,218                        // divps         %xmm10,%xmm3
   .byte  65,15,88,220                        // addps         %xmm12,%xmm3
-  .byte  15,89,29,128,35,0,0                 // mulps         0x2380(%rip),%xmm3        # 4c20 <_sk_callback_sse41+0x9b4>
+  .byte  15,89,29,108,35,0,0                 // mulps         0x236c(%rip),%xmm3        # 4c30 <_sk_callback_sse41+0x9a0>
   .byte  102,68,15,91,211                    // cvtps2dq      %xmm3,%xmm10
   .byte  243,15,16,88,20                     // movss         0x14(%rax),%xmm3
   .byte  15,198,219,0                        // shufps        $0x0,%xmm3,%xmm3
@@ -22316,7 +22361,7 @@ _sk_parametric_a_sse41:
   .byte  102,65,15,56,20,217                 // blendvps      %xmm0,%xmm9,%xmm3
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,95,216                           // maxps         %xmm0,%xmm3
-  .byte  15,93,29,107,35,0,0                 // minps         0x236b(%rip),%xmm3        # 4c30 <_sk_callback_sse41+0x9c4>
+  .byte  15,93,29,87,35,0,0                  // minps         0x2357(%rip),%xmm3        # 4c40 <_sk_callback_sse41+0x9b0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -22326,29 +22371,29 @@ HIDDEN _sk_lab_to_xyz_sse41
 FUNCTION(_sk_lab_to_xyz_sse41)
 _sk_lab_to_xyz_sse41:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,89,5,103,35,0,0               // mulps         0x2367(%rip),%xmm8        # 4c40 <_sk_callback_sse41+0x9d4>
-  .byte  68,15,40,13,111,35,0,0              // movaps        0x236f(%rip),%xmm9        # 4c50 <_sk_callback_sse41+0x9e4>
+  .byte  68,15,89,5,83,35,0,0                // mulps         0x2353(%rip),%xmm8        # 4c50 <_sk_callback_sse41+0x9c0>
+  .byte  68,15,40,13,91,35,0,0               // movaps        0x235b(%rip),%xmm9        # 4c60 <_sk_callback_sse41+0x9d0>
   .byte  65,15,89,201                        // mulps         %xmm9,%xmm1
-  .byte  15,40,5,116,35,0,0                  // movaps        0x2374(%rip),%xmm0        # 4c60 <_sk_callback_sse41+0x9f4>
+  .byte  15,40,5,96,35,0,0                   // movaps        0x2360(%rip),%xmm0        # 4c70 <_sk_callback_sse41+0x9e0>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
   .byte  15,88,208                           // addps         %xmm0,%xmm2
-  .byte  68,15,88,5,114,35,0,0               // addps         0x2372(%rip),%xmm8        # 4c70 <_sk_callback_sse41+0xa04>
-  .byte  68,15,89,5,122,35,0,0               // mulps         0x237a(%rip),%xmm8        # 4c80 <_sk_callback_sse41+0xa14>
-  .byte  15,89,13,131,35,0,0                 // mulps         0x2383(%rip),%xmm1        # 4c90 <_sk_callback_sse41+0xa24>
+  .byte  68,15,88,5,94,35,0,0                // addps         0x235e(%rip),%xmm8        # 4c80 <_sk_callback_sse41+0x9f0>
+  .byte  68,15,89,5,102,35,0,0               // mulps         0x2366(%rip),%xmm8        # 4c90 <_sk_callback_sse41+0xa00>
+  .byte  15,89,13,111,35,0,0                 // mulps         0x236f(%rip),%xmm1        # 4ca0 <_sk_callback_sse41+0xa10>
   .byte  65,15,88,200                        // addps         %xmm8,%xmm1
-  .byte  15,89,21,136,35,0,0                 // mulps         0x2388(%rip),%xmm2        # 4ca0 <_sk_callback_sse41+0xa34>
+  .byte  15,89,21,116,35,0,0                 // mulps         0x2374(%rip),%xmm2        # 4cb0 <_sk_callback_sse41+0xa20>
   .byte  69,15,40,208                        // movaps        %xmm8,%xmm10
   .byte  68,15,92,210                        // subps         %xmm2,%xmm10
   .byte  68,15,40,217                        // movaps        %xmm1,%xmm11
   .byte  69,15,89,219                        // mulps         %xmm11,%xmm11
   .byte  68,15,89,217                        // mulps         %xmm1,%xmm11
-  .byte  68,15,40,13,124,35,0,0              // movaps        0x237c(%rip),%xmm9        # 4cb0 <_sk_callback_sse41+0xa44>
+  .byte  68,15,40,13,104,35,0,0              // movaps        0x2368(%rip),%xmm9        # 4cc0 <_sk_callback_sse41+0xa30>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  65,15,194,195,1                     // cmpltps       %xmm11,%xmm0
-  .byte  15,40,21,124,35,0,0                 // movaps        0x237c(%rip),%xmm2        # 4cc0 <_sk_callback_sse41+0xa54>
+  .byte  15,40,21,104,35,0,0                 // movaps        0x2368(%rip),%xmm2        # 4cd0 <_sk_callback_sse41+0xa40>
   .byte  15,88,202                           // addps         %xmm2,%xmm1
-  .byte  68,15,40,37,129,35,0,0              // movaps        0x2381(%rip),%xmm12        # 4cd0 <_sk_callback_sse41+0xa64>
+  .byte  68,15,40,37,109,35,0,0              // movaps        0x236d(%rip),%xmm12        # 4ce0 <_sk_callback_sse41+0xa50>
   .byte  65,15,89,204                        // mulps         %xmm12,%xmm1
   .byte  102,65,15,56,20,203                 // blendvps      %xmm0,%xmm11,%xmm1
   .byte  69,15,40,216                        // movaps        %xmm8,%xmm11
@@ -22367,8 +22412,8 @@ _sk_lab_to_xyz_sse41:
   .byte  65,15,89,212                        // mulps         %xmm12,%xmm2
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  102,65,15,56,20,211                 // blendvps      %xmm0,%xmm11,%xmm2
-  .byte  15,89,13,58,35,0,0                  // mulps         0x233a(%rip),%xmm1        # 4ce0 <_sk_callback_sse41+0xa74>
-  .byte  15,89,21,67,35,0,0                  // mulps         0x2343(%rip),%xmm2        # 4cf0 <_sk_callback_sse41+0xa84>
+  .byte  15,89,13,38,35,0,0                  // mulps         0x2326(%rip),%xmm1        # 4cf0 <_sk_callback_sse41+0xa60>
+  .byte  15,89,21,47,35,0,0                  // mulps         0x232f(%rip),%xmm2        # 4d00 <_sk_callback_sse41+0xa70>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,40,193                           // movaps        %xmm1,%xmm0
   .byte  65,15,40,200                        // movaps        %xmm8,%xmm1
@@ -22382,7 +22427,7 @@ _sk_load_a8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,49,4,56                   // pmovzxbd      (%rax,%rdi,1),%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,51,35,0,0                  // mulps         0x2333(%rip),%xmm3        # 4d00 <_sk_callback_sse41+0xa94>
+  .byte  15,89,29,31,35,0,0                  // mulps         0x231f(%rip),%xmm3        # 4d10 <_sk_callback_sse41+0xa80>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,87,201                           // xorps         %xmm1,%xmm1
@@ -22415,7 +22460,7 @@ _sk_gather_a8_sse41:
   .byte  102,15,58,32,192,3                  // pinsrb        $0x3,%eax,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,199,34,0,0                 // mulps         0x22c7(%rip),%xmm3        # 4d10 <_sk_callback_sse41+0xaa4>
+  .byte  15,89,29,179,34,0,0                 // mulps         0x22b3(%rip),%xmm3        # 4d20 <_sk_callback_sse41+0xa90>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
@@ -22428,7 +22473,7 @@ FUNCTION(_sk_store_a8_sse41)
 _sk_store_a8_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,187,34,0,0               // movaps        0x22bb(%rip),%xmm8        # 4d20 <_sk_callback_sse41+0xab4>
+  .byte  68,15,40,5,167,34,0,0               // movaps        0x22a7(%rip),%xmm8        # 4d30 <_sk_callback_sse41+0xaa0>
   .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
   .byte  102,69,15,56,43,192                 // packusdw      %xmm8,%xmm8
@@ -22445,9 +22490,9 @@ _sk_load_g8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,49,4,56                   // pmovzxbd      (%rax,%rdi,1),%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,152,34,0,0                  // mulps         0x2298(%rip),%xmm0        # 4d30 <_sk_callback_sse41+0xac4>
+  .byte  15,89,5,132,34,0,0                  // mulps         0x2284(%rip),%xmm0        # 4d40 <_sk_callback_sse41+0xab0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,159,34,0,0                 // movaps        0x229f(%rip),%xmm3        # 4d40 <_sk_callback_sse41+0xad4>
+  .byte  15,40,29,139,34,0,0                 // movaps        0x228b(%rip),%xmm3        # 4d50 <_sk_callback_sse41+0xac0>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -22478,9 +22523,9 @@ _sk_gather_g8_sse41:
   .byte  102,15,58,32,192,3                  // pinsrb        $0x3,%eax,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,56,34,0,0                   // mulps         0x2238(%rip),%xmm0        # 4d50 <_sk_callback_sse41+0xae4>
+  .byte  15,89,5,36,34,0,0                   // mulps         0x2224(%rip),%xmm0        # 4d60 <_sk_callback_sse41+0xad0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,63,34,0,0                  // movaps        0x223f(%rip),%xmm3        # 4d60 <_sk_callback_sse41+0xaf4>
+  .byte  15,40,29,43,34,0,0                  // movaps        0x222b(%rip),%xmm3        # 4d70 <_sk_callback_sse41+0xae0>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -22492,9 +22537,9 @@ _sk_gather_i8_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            2b38 <_sk_gather_i8_sse41+0xf>
+  .byte  116,5                               // je            2b5c <_sk_gather_i8_sse41+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           2b3a <_sk_gather_i8_sse41+0x11>
+  .byte  235,2                               // jmp           2b5e <_sk_gather_i8_sse41+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  243,15,91,201                       // cvttps2dq     %xmm1,%xmm1
@@ -22525,17 +22570,17 @@ _sk_gather_i8_sse41:
   .byte  102,15,58,34,28,8,1                 // pinsrd        $0x1,(%rax,%rcx,1),%xmm3
   .byte  102,66,15,58,34,28,144,2            // pinsrd        $0x2,(%rax,%r10,4),%xmm3
   .byte  102,66,15,58,34,28,8,3              // pinsrd        $0x3,(%rax,%r9,1),%xmm3
-  .byte  102,15,111,5,150,33,0,0             // movdqa        0x2196(%rip),%xmm0        # 4d70 <_sk_callback_sse41+0xb04>
+  .byte  102,15,111,5,130,33,0,0             // movdqa        0x2182(%rip),%xmm0        # 4d80 <_sk_callback_sse41+0xaf0>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,151,33,0,0               // movaps        0x2197(%rip),%xmm8        # 4d80 <_sk_callback_sse41+0xb14>
+  .byte  68,15,40,5,131,33,0,0               // movaps        0x2183(%rip),%xmm8        # 4d90 <_sk_callback_sse41+0xb00>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
-  .byte  102,15,56,0,13,150,33,0,0           // pshufb        0x2196(%rip),%xmm1        # 4d90 <_sk_callback_sse41+0xb24>
+  .byte  102,15,56,0,13,130,33,0,0           // pshufb        0x2182(%rip),%xmm1        # 4da0 <_sk_callback_sse41+0xb10>
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,111,211                      // movdqa        %xmm3,%xmm2
-  .byte  102,15,56,0,21,146,33,0,0           // pshufb        0x2192(%rip),%xmm2        # 4da0 <_sk_callback_sse41+0xb34>
+  .byte  102,15,56,0,21,126,33,0,0           // pshufb        0x217e(%rip),%xmm2        # 4db0 <_sk_callback_sse41+0xb20>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  102,15,114,211,24                   // psrld         $0x18,%xmm3
@@ -22551,19 +22596,19 @@ _sk_load_565_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,51,20,120                 // pmovzxwd      (%rax,%rdi,2),%xmm2
-  .byte  102,15,111,5,120,33,0,0             // movdqa        0x2178(%rip),%xmm0        # 4db0 <_sk_callback_sse41+0xb44>
+  .byte  102,15,111,5,100,33,0,0             // movdqa        0x2164(%rip),%xmm0        # 4dc0 <_sk_callback_sse41+0xb30>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,122,33,0,0                  // mulps         0x217a(%rip),%xmm0        # 4dc0 <_sk_callback_sse41+0xb54>
-  .byte  102,15,111,13,130,33,0,0            // movdqa        0x2182(%rip),%xmm1        # 4dd0 <_sk_callback_sse41+0xb64>
+  .byte  15,89,5,102,33,0,0                  // mulps         0x2166(%rip),%xmm0        # 4dd0 <_sk_callback_sse41+0xb40>
+  .byte  102,15,111,13,110,33,0,0            // movdqa        0x216e(%rip),%xmm1        # 4de0 <_sk_callback_sse41+0xb50>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,132,33,0,0                 // mulps         0x2184(%rip),%xmm1        # 4de0 <_sk_callback_sse41+0xb74>
-  .byte  102,15,219,21,140,33,0,0            // pand          0x218c(%rip),%xmm2        # 4df0 <_sk_callback_sse41+0xb84>
+  .byte  15,89,13,112,33,0,0                 // mulps         0x2170(%rip),%xmm1        # 4df0 <_sk_callback_sse41+0xb60>
+  .byte  102,15,219,21,120,33,0,0            // pand          0x2178(%rip),%xmm2        # 4e00 <_sk_callback_sse41+0xb70>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,146,33,0,0                 // mulps         0x2192(%rip),%xmm2        # 4e00 <_sk_callback_sse41+0xb94>
+  .byte  15,89,21,126,33,0,0                 // mulps         0x217e(%rip),%xmm2        # 4e10 <_sk_callback_sse41+0xb80>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,153,33,0,0                 // movaps        0x2199(%rip),%xmm3        # 4e10 <_sk_callback_sse41+0xba4>
+  .byte  15,40,29,133,33,0,0                 // movaps        0x2185(%rip),%xmm3        # 4e20 <_sk_callback_sse41+0xb90>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_gather_565_sse41
@@ -22591,19 +22636,19 @@ _sk_gather_565_sse41:
   .byte  65,15,183,4,65                      // movzwl        (%r9,%rax,2),%eax
   .byte  102,15,196,192,3                    // pinsrw        $0x3,%eax,%xmm0
   .byte  102,15,56,51,208                    // pmovzxwd      %xmm0,%xmm2
-  .byte  102,15,111,5,62,33,0,0              // movdqa        0x213e(%rip),%xmm0        # 4e20 <_sk_callback_sse41+0xbb4>
+  .byte  102,15,111,5,42,33,0,0              // movdqa        0x212a(%rip),%xmm0        # 4e30 <_sk_callback_sse41+0xba0>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,64,33,0,0                   // mulps         0x2140(%rip),%xmm0        # 4e30 <_sk_callback_sse41+0xbc4>
-  .byte  102,15,111,13,72,33,0,0             // movdqa        0x2148(%rip),%xmm1        # 4e40 <_sk_callback_sse41+0xbd4>
+  .byte  15,89,5,44,33,0,0                   // mulps         0x212c(%rip),%xmm0        # 4e40 <_sk_callback_sse41+0xbb0>
+  .byte  102,15,111,13,52,33,0,0             // movdqa        0x2134(%rip),%xmm1        # 4e50 <_sk_callback_sse41+0xbc0>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,74,33,0,0                  // mulps         0x214a(%rip),%xmm1        # 4e50 <_sk_callback_sse41+0xbe4>
-  .byte  102,15,219,21,82,33,0,0             // pand          0x2152(%rip),%xmm2        # 4e60 <_sk_callback_sse41+0xbf4>
+  .byte  15,89,13,54,33,0,0                  // mulps         0x2136(%rip),%xmm1        # 4e60 <_sk_callback_sse41+0xbd0>
+  .byte  102,15,219,21,62,33,0,0             // pand          0x213e(%rip),%xmm2        # 4e70 <_sk_callback_sse41+0xbe0>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,88,33,0,0                  // mulps         0x2158(%rip),%xmm2        # 4e70 <_sk_callback_sse41+0xc04>
+  .byte  15,89,21,68,33,0,0                  // mulps         0x2144(%rip),%xmm2        # 4e80 <_sk_callback_sse41+0xbf0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,95,33,0,0                  // movaps        0x215f(%rip),%xmm3        # 4e80 <_sk_callback_sse41+0xc14>
+  .byte  15,40,29,75,33,0,0                  // movaps        0x214b(%rip),%xmm3        # 4e90 <_sk_callback_sse41+0xc00>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_565_sse41
@@ -22612,12 +22657,12 @@ FUNCTION(_sk_store_565_sse41)
 _sk_store_565_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,96,33,0,0                // movaps        0x2160(%rip),%xmm8        # 4e90 <_sk_callback_sse41+0xc24>
+  .byte  68,15,40,5,76,33,0,0                // movaps        0x214c(%rip),%xmm8        # 4ea0 <_sk_callback_sse41+0xc10>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
   .byte  102,65,15,114,241,11                // pslld         $0xb,%xmm9
-  .byte  68,15,40,21,85,33,0,0               // movaps        0x2155(%rip),%xmm10        # 4ea0 <_sk_callback_sse41+0xc34>
+  .byte  68,15,40,21,65,33,0,0               // movaps        0x2141(%rip),%xmm10        # 4eb0 <_sk_callback_sse41+0xc20>
   .byte  68,15,89,209                        // mulps         %xmm1,%xmm10
   .byte  102,69,15,91,210                    // cvtps2dq      %xmm10,%xmm10
   .byte  102,65,15,114,242,5                 // pslld         $0x5,%xmm10
@@ -22637,21 +22682,21 @@ _sk_load_4444_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,51,28,120                 // pmovzxwd      (%rax,%rdi,2),%xmm3
-  .byte  102,15,111,5,32,33,0,0              // movdqa        0x2120(%rip),%xmm0        # 4eb0 <_sk_callback_sse41+0xc44>
+  .byte  102,15,111,5,12,33,0,0              // movdqa        0x210c(%rip),%xmm0        # 4ec0 <_sk_callback_sse41+0xc30>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,34,33,0,0                   // mulps         0x2122(%rip),%xmm0        # 4ec0 <_sk_callback_sse41+0xc54>
-  .byte  102,15,111,13,42,33,0,0             // movdqa        0x212a(%rip),%xmm1        # 4ed0 <_sk_callback_sse41+0xc64>
+  .byte  15,89,5,14,33,0,0                   // mulps         0x210e(%rip),%xmm0        # 4ed0 <_sk_callback_sse41+0xc40>
+  .byte  102,15,111,13,22,33,0,0             // movdqa        0x2116(%rip),%xmm1        # 4ee0 <_sk_callback_sse41+0xc50>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,44,33,0,0                  // mulps         0x212c(%rip),%xmm1        # 4ee0 <_sk_callback_sse41+0xc74>
-  .byte  102,15,111,21,52,33,0,0             // movdqa        0x2134(%rip),%xmm2        # 4ef0 <_sk_callback_sse41+0xc84>
+  .byte  15,89,13,24,33,0,0                  // mulps         0x2118(%rip),%xmm1        # 4ef0 <_sk_callback_sse41+0xc60>
+  .byte  102,15,111,21,32,33,0,0             // movdqa        0x2120(%rip),%xmm2        # 4f00 <_sk_callback_sse41+0xc70>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,54,33,0,0                  // mulps         0x2136(%rip),%xmm2        # 4f00 <_sk_callback_sse41+0xc94>
-  .byte  102,15,219,29,62,33,0,0             // pand          0x213e(%rip),%xmm3        # 4f10 <_sk_callback_sse41+0xca4>
+  .byte  15,89,21,34,33,0,0                  // mulps         0x2122(%rip),%xmm2        # 4f10 <_sk_callback_sse41+0xc80>
+  .byte  102,15,219,29,42,33,0,0             // pand          0x212a(%rip),%xmm3        # 4f20 <_sk_callback_sse41+0xc90>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,68,33,0,0                  // mulps         0x2144(%rip),%xmm3        # 4f20 <_sk_callback_sse41+0xcb4>
+  .byte  15,89,29,48,33,0,0                  // mulps         0x2130(%rip),%xmm3        # 4f30 <_sk_callback_sse41+0xca0>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -22680,21 +22725,21 @@ _sk_gather_4444_sse41:
   .byte  65,15,183,4,65                      // movzwl        (%r9,%rax,2),%eax
   .byte  102,15,196,192,3                    // pinsrw        $0x3,%eax,%xmm0
   .byte  102,15,56,51,216                    // pmovzxwd      %xmm0,%xmm3
-  .byte  102,15,111,5,231,32,0,0             // movdqa        0x20e7(%rip),%xmm0        # 4f30 <_sk_callback_sse41+0xcc4>
+  .byte  102,15,111,5,211,32,0,0             // movdqa        0x20d3(%rip),%xmm0        # 4f40 <_sk_callback_sse41+0xcb0>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,233,32,0,0                  // mulps         0x20e9(%rip),%xmm0        # 4f40 <_sk_callback_sse41+0xcd4>
-  .byte  102,15,111,13,241,32,0,0            // movdqa        0x20f1(%rip),%xmm1        # 4f50 <_sk_callback_sse41+0xce4>
+  .byte  15,89,5,213,32,0,0                  // mulps         0x20d5(%rip),%xmm0        # 4f50 <_sk_callback_sse41+0xcc0>
+  .byte  102,15,111,13,221,32,0,0            // movdqa        0x20dd(%rip),%xmm1        # 4f60 <_sk_callback_sse41+0xcd0>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,243,32,0,0                 // mulps         0x20f3(%rip),%xmm1        # 4f60 <_sk_callback_sse41+0xcf4>
-  .byte  102,15,111,21,251,32,0,0            // movdqa        0x20fb(%rip),%xmm2        # 4f70 <_sk_callback_sse41+0xd04>
+  .byte  15,89,13,223,32,0,0                 // mulps         0x20df(%rip),%xmm1        # 4f70 <_sk_callback_sse41+0xce0>
+  .byte  102,15,111,21,231,32,0,0            // movdqa        0x20e7(%rip),%xmm2        # 4f80 <_sk_callback_sse41+0xcf0>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,253,32,0,0                 // mulps         0x20fd(%rip),%xmm2        # 4f80 <_sk_callback_sse41+0xd14>
-  .byte  102,15,219,29,5,33,0,0              // pand          0x2105(%rip),%xmm3        # 4f90 <_sk_callback_sse41+0xd24>
+  .byte  15,89,21,233,32,0,0                 // mulps         0x20e9(%rip),%xmm2        # 4f90 <_sk_callback_sse41+0xd00>
+  .byte  102,15,219,29,241,32,0,0            // pand          0x20f1(%rip),%xmm3        # 4fa0 <_sk_callback_sse41+0xd10>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,11,33,0,0                  // mulps         0x210b(%rip),%xmm3        # 4fa0 <_sk_callback_sse41+0xd34>
+  .byte  15,89,29,247,32,0,0                 // mulps         0x20f7(%rip),%xmm3        # 4fb0 <_sk_callback_sse41+0xd20>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -22704,7 +22749,7 @@ FUNCTION(_sk_store_4444_sse41)
 _sk_store_4444_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,10,33,0,0                // movaps        0x210a(%rip),%xmm8        # 4fb0 <_sk_callback_sse41+0xd44>
+  .byte  68,15,40,5,246,32,0,0               // movaps        0x20f6(%rip),%xmm8        # 4fc0 <_sk_callback_sse41+0xd30>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -22734,17 +22779,17 @@ _sk_load_8888_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  15,16,28,184                        // movups        (%rax,%rdi,4),%xmm3
-  .byte  15,40,5,169,32,0,0                  // movaps        0x20a9(%rip),%xmm0        # 4fc0 <_sk_callback_sse41+0xd54>
+  .byte  15,40,5,149,32,0,0                  // movaps        0x2095(%rip),%xmm0        # 4fd0 <_sk_callback_sse41+0xd40>
   .byte  15,84,195                           // andps         %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,171,32,0,0               // movaps        0x20ab(%rip),%xmm8        # 4fd0 <_sk_callback_sse41+0xd64>
+  .byte  68,15,40,5,151,32,0,0               // movaps        0x2097(%rip),%xmm8        # 4fe0 <_sk_callback_sse41+0xd50>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,40,203                           // movaps        %xmm3,%xmm1
-  .byte  102,15,56,0,13,171,32,0,0           // pshufb        0x20ab(%rip),%xmm1        # 4fe0 <_sk_callback_sse41+0xd74>
+  .byte  102,15,56,0,13,151,32,0,0           // pshufb        0x2097(%rip),%xmm1        # 4ff0 <_sk_callback_sse41+0xd60>
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  15,40,211                           // movaps        %xmm3,%xmm2
-  .byte  102,15,56,0,21,168,32,0,0           // pshufb        0x20a8(%rip),%xmm2        # 4ff0 <_sk_callback_sse41+0xd84>
+  .byte  102,15,56,0,21,148,32,0,0           // pshufb        0x2094(%rip),%xmm2        # 5000 <_sk_callback_sse41+0xd70>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  102,15,114,211,24                   // psrld         $0x18,%xmm3
@@ -22775,17 +22820,17 @@ _sk_gather_8888_sse41:
   .byte  102,65,15,58,34,28,129,1            // pinsrd        $0x1,(%r9,%rax,4),%xmm3
   .byte  102,67,15,58,34,28,145,2            // pinsrd        $0x2,(%r9,%r10,4),%xmm3
   .byte  102,65,15,58,34,28,137,3            // pinsrd        $0x3,(%r9,%rcx,4),%xmm3
-  .byte  102,15,111,5,65,32,0,0              // movdqa        0x2041(%rip),%xmm0        # 5000 <_sk_callback_sse41+0xd94>
+  .byte  102,15,111,5,45,32,0,0              // movdqa        0x202d(%rip),%xmm0        # 5010 <_sk_callback_sse41+0xd80>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,66,32,0,0                // movaps        0x2042(%rip),%xmm8        # 5010 <_sk_callback_sse41+0xda4>
+  .byte  68,15,40,5,46,32,0,0                // movaps        0x202e(%rip),%xmm8        # 5020 <_sk_callback_sse41+0xd90>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
-  .byte  102,15,56,0,13,65,32,0,0            // pshufb        0x2041(%rip),%xmm1        # 5020 <_sk_callback_sse41+0xdb4>
+  .byte  102,15,56,0,13,45,32,0,0            // pshufb        0x202d(%rip),%xmm1        # 5030 <_sk_callback_sse41+0xda0>
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,111,211                      // movdqa        %xmm3,%xmm2
-  .byte  102,15,56,0,21,61,32,0,0            // pshufb        0x203d(%rip),%xmm2        # 5030 <_sk_callback_sse41+0xdc4>
+  .byte  102,15,56,0,21,41,32,0,0            // pshufb        0x2029(%rip),%xmm2        # 5040 <_sk_callback_sse41+0xdb0>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  102,15,114,211,24                   // psrld         $0x18,%xmm3
@@ -22800,7 +22845,7 @@ FUNCTION(_sk_store_8888_sse41)
 _sk_store_8888_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,41,32,0,0                // movaps        0x2029(%rip),%xmm8        # 5040 <_sk_callback_sse41+0xdd4>
+  .byte  68,15,40,5,21,32,0,0                // movaps        0x2015(%rip),%xmm8        # 5050 <_sk_callback_sse41+0xdc0>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -22837,18 +22882,18 @@ _sk_load_f16_sse41:
   .byte  102,68,15,97,216                    // punpcklwd     %xmm0,%xmm11
   .byte  102,68,15,105,200                   // punpckhwd     %xmm0,%xmm9
   .byte  102,65,15,56,51,203                 // pmovzxwd      %xmm11,%xmm1
-  .byte  102,68,15,111,5,162,31,0,0          // movdqa        0x1fa2(%rip),%xmm8        # 5050 <_sk_callback_sse41+0xde4>
+  .byte  102,68,15,111,5,142,31,0,0          // movdqa        0x1f8e(%rip),%xmm8        # 5060 <_sk_callback_sse41+0xdd0>
   .byte  102,15,111,209                      // movdqa        %xmm1,%xmm2
   .byte  102,65,15,219,208                   // pand          %xmm8,%xmm2
   .byte  102,15,239,202                      // pxor          %xmm2,%xmm1
-  .byte  102,15,111,29,157,31,0,0            // movdqa        0x1f9d(%rip),%xmm3        # 5060 <_sk_callback_sse41+0xdf4>
+  .byte  102,15,111,29,137,31,0,0            // movdqa        0x1f89(%rip),%xmm3        # 5070 <_sk_callback_sse41+0xde0>
   .byte  102,15,114,242,16                   // pslld         $0x10,%xmm2
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,15,56,63,195                    // pmaxud        %xmm3,%xmm0
   .byte  102,15,118,193                      // pcmpeqd       %xmm1,%xmm0
   .byte  102,15,114,241,13                   // pslld         $0xd,%xmm1
   .byte  102,15,235,202                      // por           %xmm2,%xmm1
-  .byte  102,68,15,111,21,137,31,0,0         // movdqa        0x1f89(%rip),%xmm10        # 5070 <_sk_callback_sse41+0xe04>
+  .byte  102,68,15,111,21,117,31,0,0         // movdqa        0x1f75(%rip),%xmm10        # 5080 <_sk_callback_sse41+0xdf0>
   .byte  102,65,15,254,202                   // paddd         %xmm10,%xmm1
   .byte  102,15,219,193                      // pand          %xmm1,%xmm0
   .byte  102,65,15,115,219,8                 // psrldq        $0x8,%xmm11
@@ -22921,18 +22966,18 @@ _sk_gather_f16_sse41:
   .byte  102,68,15,97,218                    // punpcklwd     %xmm2,%xmm11
   .byte  102,68,15,105,202                   // punpckhwd     %xmm2,%xmm9
   .byte  102,65,15,56,51,203                 // pmovzxwd      %xmm11,%xmm1
-  .byte  102,68,15,111,5,71,30,0,0           // movdqa        0x1e47(%rip),%xmm8        # 5080 <_sk_callback_sse41+0xe14>
+  .byte  102,68,15,111,5,51,30,0,0           // movdqa        0x1e33(%rip),%xmm8        # 5090 <_sk_callback_sse41+0xe00>
   .byte  102,15,111,209                      // movdqa        %xmm1,%xmm2
   .byte  102,65,15,219,208                   // pand          %xmm8,%xmm2
   .byte  102,15,239,202                      // pxor          %xmm2,%xmm1
-  .byte  102,15,111,29,66,30,0,0             // movdqa        0x1e42(%rip),%xmm3        # 5090 <_sk_callback_sse41+0xe24>
+  .byte  102,15,111,29,46,30,0,0             // movdqa        0x1e2e(%rip),%xmm3        # 50a0 <_sk_callback_sse41+0xe10>
   .byte  102,15,114,242,16                   // pslld         $0x10,%xmm2
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,15,56,63,195                    // pmaxud        %xmm3,%xmm0
   .byte  102,15,118,193                      // pcmpeqd       %xmm1,%xmm0
   .byte  102,15,114,241,13                   // pslld         $0xd,%xmm1
   .byte  102,15,235,202                      // por           %xmm2,%xmm1
-  .byte  102,68,15,111,21,46,30,0,0          // movdqa        0x1e2e(%rip),%xmm10        # 50a0 <_sk_callback_sse41+0xe34>
+  .byte  102,68,15,111,21,26,30,0,0          // movdqa        0x1e1a(%rip),%xmm10        # 50b0 <_sk_callback_sse41+0xe20>
   .byte  102,65,15,254,202                   // paddd         %xmm10,%xmm1
   .byte  102,15,219,193                      // pand          %xmm1,%xmm0
   .byte  102,65,15,115,219,8                 // psrldq        $0x8,%xmm11
@@ -22980,17 +23025,17 @@ FUNCTION(_sk_store_f16_sse41)
 _sk_store_f16_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  102,68,15,111,21,100,29,0,0         // movdqa        0x1d64(%rip),%xmm10        # 50b0 <_sk_callback_sse41+0xe44>
+  .byte  102,68,15,111,21,80,29,0,0          // movdqa        0x1d50(%rip),%xmm10        # 50c0 <_sk_callback_sse41+0xe30>
   .byte  102,68,15,111,224                   // movdqa        %xmm0,%xmm12
   .byte  102,68,15,111,232                   // movdqa        %xmm0,%xmm13
   .byte  102,69,15,219,234                   // pand          %xmm10,%xmm13
   .byte  102,69,15,239,229                   // pxor          %xmm13,%xmm12
-  .byte  102,68,15,111,13,87,29,0,0          // movdqa        0x1d57(%rip),%xmm9        # 50c0 <_sk_callback_sse41+0xe54>
+  .byte  102,68,15,111,13,67,29,0,0          // movdqa        0x1d43(%rip),%xmm9        # 50d0 <_sk_callback_sse41+0xe40>
   .byte  102,65,15,114,213,16                // psrld         $0x10,%xmm13
   .byte  102,69,15,111,193                   // movdqa        %xmm9,%xmm8
   .byte  102,69,15,102,196                   // pcmpgtd       %xmm12,%xmm8
   .byte  102,65,15,114,212,13                // psrld         $0xd,%xmm12
-  .byte  102,68,15,111,29,72,29,0,0          // movdqa        0x1d48(%rip),%xmm11        # 50d0 <_sk_callback_sse41+0xe64>
+  .byte  102,68,15,111,29,52,29,0,0          // movdqa        0x1d34(%rip),%xmm11        # 50e0 <_sk_callback_sse41+0xe50>
   .byte  102,69,15,235,235                   // por           %xmm11,%xmm13
   .byte  102,69,15,254,236                   // paddd         %xmm12,%xmm13
   .byte  102,69,15,223,197                   // pandn         %xmm13,%xmm8
@@ -23060,7 +23105,7 @@ _sk_load_u16_be_sse41:
   .byte  102,15,235,200                      // por           %xmm0,%xmm1
   .byte  102,15,56,51,193                    // pmovzxwd      %xmm1,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,23,28,0,0                // movaps        0x1c17(%rip),%xmm8        # 50e0 <_sk_callback_sse41+0xe74>
+  .byte  68,15,40,5,3,28,0,0                 // movaps        0x1c03(%rip),%xmm8        # 50f0 <_sk_callback_sse41+0xe60>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -23112,7 +23157,7 @@ _sk_load_rgb_u16_be_sse41:
   .byte  102,15,235,193                      // por           %xmm1,%xmm0
   .byte  102,15,56,51,192                    // pmovzxwd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,88,27,0,0                // movaps        0x1b58(%rip),%xmm8        # 50f0 <_sk_callback_sse41+0xe84>
+  .byte  68,15,40,5,68,27,0,0                // movaps        0x1b44(%rip),%xmm8        # 5100 <_sk_callback_sse41+0xe70>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -23129,7 +23174,7 @@ _sk_load_rgb_u16_be_sse41:
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,31,27,0,0                  // movaps        0x1b1f(%rip),%xmm3        # 5100 <_sk_callback_sse41+0xe94>
+  .byte  15,40,29,11,27,0,0                  // movaps        0x1b0b(%rip),%xmm3        # 5110 <_sk_callback_sse41+0xe80>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_u16_be_sse41
@@ -23138,7 +23183,7 @@ FUNCTION(_sk_store_u16_be_sse41)
 _sk_store_u16_be_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,13,32,27,0,0               // movaps        0x1b20(%rip),%xmm9        # 5110 <_sk_callback_sse41+0xea4>
+  .byte  68,15,40,13,12,27,0,0               // movaps        0x1b0c(%rip),%xmm9        # 5120 <_sk_callback_sse41+0xe90>
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
@@ -23361,10 +23406,10 @@ HIDDEN _sk_luminance_to_alpha_sse41
 FUNCTION(_sk_luminance_to_alpha_sse41)
 _sk_luminance_to_alpha_sse41:
   .byte  15,40,218                           // movaps        %xmm2,%xmm3
-  .byte  15,89,5,62,24,0,0                   // mulps         0x183e(%rip),%xmm0        # 5120 <_sk_callback_sse41+0xeb4>
-  .byte  15,89,13,71,24,0,0                  // mulps         0x1847(%rip),%xmm1        # 5130 <_sk_callback_sse41+0xec4>
+  .byte  15,89,5,42,24,0,0                   // mulps         0x182a(%rip),%xmm0        # 5130 <_sk_callback_sse41+0xea0>
+  .byte  15,89,13,51,24,0,0                  // mulps         0x1833(%rip),%xmm1        # 5140 <_sk_callback_sse41+0xeb0>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  15,89,29,77,24,0,0                  // mulps         0x184d(%rip),%xmm3        # 5140 <_sk_callback_sse41+0xed4>
+  .byte  15,89,29,57,24,0,0                  // mulps         0x1839(%rip),%xmm3        # 5150 <_sk_callback_sse41+0xec0>
   .byte  15,88,217                           // addps         %xmm1,%xmm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
@@ -23597,7 +23642,7 @@ _sk_linear_gradient_sse41:
   .byte  69,15,198,237,0                     // shufps        $0x0,%xmm13,%xmm13
   .byte  72,139,8                            // mov           (%rax),%rcx
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,132,254,0,0,0                    // je            3d95 <_sk_linear_gradient_sse41+0x138>
+  .byte  15,132,254,0,0,0                    // je            3db9 <_sk_linear_gradient_sse41+0x138>
   .byte  15,41,100,36,168                    // movaps        %xmm4,-0x58(%rsp)
   .byte  15,41,108,36,184                    // movaps        %xmm5,-0x48(%rsp)
   .byte  15,41,116,36,200                    // movaps        %xmm6,-0x38(%rsp)
@@ -23647,12 +23692,12 @@ _sk_linear_gradient_sse41:
   .byte  15,40,196                           // movaps        %xmm4,%xmm0
   .byte  72,131,192,36                       // add           $0x24,%rax
   .byte  72,255,201                          // dec           %rcx
-  .byte  15,133,65,255,255,255               // jne           3cc0 <_sk_linear_gradient_sse41+0x63>
+  .byte  15,133,65,255,255,255               // jne           3ce4 <_sk_linear_gradient_sse41+0x63>
   .byte  15,40,124,36,216                    // movaps        -0x28(%rsp),%xmm7
   .byte  15,40,116,36,200                    // movaps        -0x38(%rsp),%xmm6
   .byte  15,40,108,36,184                    // movaps        -0x48(%rsp),%xmm5
   .byte  15,40,100,36,168                    // movaps        -0x58(%rsp),%xmm4
-  .byte  235,13                              // jmp           3da2 <_sk_linear_gradient_sse41+0x145>
+  .byte  235,13                              // jmp           3dc6 <_sk_linear_gradient_sse41+0x145>
   .byte  15,87,201                           // xorps         %xmm1,%xmm1
   .byte  15,87,210                           // xorps         %xmm2,%xmm2
   .byte  15,87,219                           // xorps         %xmm3,%xmm3
@@ -23721,26 +23766,26 @@ _sk_xy_to_polar_unit_sse41:
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,40,236                        // movaps        %xmm12,%xmm13
   .byte  69,15,89,237                        // mulps         %xmm13,%xmm13
-  .byte  68,15,40,21,214,18,0,0              // movaps        0x12d6(%rip),%xmm10        # 5150 <_sk_callback_sse41+0xee4>
+  .byte  68,15,40,21,194,18,0,0              // movaps        0x12c2(%rip),%xmm10        # 5160 <_sk_callback_sse41+0xed0>
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,88,21,218,18,0,0              // addps         0x12da(%rip),%xmm10        # 5160 <_sk_callback_sse41+0xef4>
+  .byte  68,15,88,21,198,18,0,0              // addps         0x12c6(%rip),%xmm10        # 5170 <_sk_callback_sse41+0xee0>
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,88,21,222,18,0,0              // addps         0x12de(%rip),%xmm10        # 5170 <_sk_callback_sse41+0xf04>
+  .byte  68,15,88,21,202,18,0,0              // addps         0x12ca(%rip),%xmm10        # 5180 <_sk_callback_sse41+0xef0>
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,88,21,226,18,0,0              // addps         0x12e2(%rip),%xmm10        # 5180 <_sk_callback_sse41+0xf14>
+  .byte  68,15,88,21,206,18,0,0              // addps         0x12ce(%rip),%xmm10        # 5190 <_sk_callback_sse41+0xf00>
   .byte  69,15,89,212                        // mulps         %xmm12,%xmm10
   .byte  65,15,194,195,1                     // cmpltps       %xmm11,%xmm0
-  .byte  68,15,40,29,225,18,0,0              // movaps        0x12e1(%rip),%xmm11        # 5190 <_sk_callback_sse41+0xf24>
+  .byte  68,15,40,29,205,18,0,0              // movaps        0x12cd(%rip),%xmm11        # 51a0 <_sk_callback_sse41+0xf10>
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  102,69,15,56,20,211                 // blendvps      %xmm0,%xmm11,%xmm10
   .byte  69,15,194,200,1                     // cmpltps       %xmm8,%xmm9
-  .byte  68,15,40,29,218,18,0,0              // movaps        0x12da(%rip),%xmm11        # 51a0 <_sk_callback_sse41+0xf34>
+  .byte  68,15,40,29,198,18,0,0              // movaps        0x12c6(%rip),%xmm11        # 51b0 <_sk_callback_sse41+0xf20>
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  102,69,15,56,20,211                 // blendvps      %xmm0,%xmm11,%xmm10
   .byte  15,40,193                           // movaps        %xmm1,%xmm0
   .byte  65,15,194,192,1                     // cmpltps       %xmm8,%xmm0
-  .byte  68,15,40,13,204,18,0,0              // movaps        0x12cc(%rip),%xmm9        # 51b0 <_sk_callback_sse41+0xf44>
+  .byte  68,15,40,13,184,18,0,0              // movaps        0x12b8(%rip),%xmm9        # 51c0 <_sk_callback_sse41+0xf30>
   .byte  69,15,92,202                        // subps         %xmm10,%xmm9
   .byte  102,69,15,56,20,209                 // blendvps      %xmm0,%xmm9,%xmm10
   .byte  69,15,194,194,7                     // cmpordps      %xmm10,%xmm8
@@ -23767,7 +23812,7 @@ HIDDEN _sk_save_xy_sse41
 FUNCTION(_sk_save_xy_sse41)
 _sk_save_xy_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,157,18,0,0               // movaps        0x129d(%rip),%xmm8        # 51c0 <_sk_callback_sse41+0xf54>
+  .byte  68,15,40,5,137,18,0,0               // movaps        0x1289(%rip),%xmm8        # 51d0 <_sk_callback_sse41+0xf40>
   .byte  15,17,0                             // movups        %xmm0,(%rax)
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,88,200                        // addps         %xmm8,%xmm9
@@ -23811,8 +23856,8 @@ _sk_bilinear_nx_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,31,18,0,0                   // addps         0x121f(%rip),%xmm0        # 51d0 <_sk_callback_sse41+0xf64>
-  .byte  68,15,40,13,39,18,0,0               // movaps        0x1227(%rip),%xmm9        # 51e0 <_sk_callback_sse41+0xf74>
+  .byte  15,88,5,11,18,0,0                   // addps         0x120b(%rip),%xmm0        # 51e0 <_sk_callback_sse41+0xf50>
+  .byte  68,15,40,13,19,18,0,0               // movaps        0x1213(%rip),%xmm9        # 51f0 <_sk_callback_sse41+0xf60>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -23825,7 +23870,7 @@ _sk_bilinear_px_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,22,18,0,0                   // addps         0x1216(%rip),%xmm0        # 51f0 <_sk_callback_sse41+0xf84>
+  .byte  15,88,5,2,18,0,0                    // addps         0x1202(%rip),%xmm0        # 5200 <_sk_callback_sse41+0xf70>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -23837,8 +23882,8 @@ _sk_bilinear_ny_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,8,18,0,0                   // addps         0x1208(%rip),%xmm1        # 5200 <_sk_callback_sse41+0xf94>
-  .byte  68,15,40,13,16,18,0,0               // movaps        0x1210(%rip),%xmm9        # 5210 <_sk_callback_sse41+0xfa4>
+  .byte  15,88,13,244,17,0,0                 // addps         0x11f4(%rip),%xmm1        # 5210 <_sk_callback_sse41+0xf80>
+  .byte  68,15,40,13,252,17,0,0              // movaps        0x11fc(%rip),%xmm9        # 5220 <_sk_callback_sse41+0xf90>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -23851,7 +23896,7 @@ _sk_bilinear_py_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,254,17,0,0                 // addps         0x11fe(%rip),%xmm1        # 5220 <_sk_callback_sse41+0xfb4>
+  .byte  15,88,13,234,17,0,0                 // addps         0x11ea(%rip),%xmm1        # 5230 <_sk_callback_sse41+0xfa0>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -23863,13 +23908,13 @@ _sk_bicubic_n3x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,241,17,0,0                  // addps         0x11f1(%rip),%xmm0        # 5230 <_sk_callback_sse41+0xfc4>
-  .byte  68,15,40,13,249,17,0,0              // movaps        0x11f9(%rip),%xmm9        # 5240 <_sk_callback_sse41+0xfd4>
+  .byte  15,88,5,221,17,0,0                  // addps         0x11dd(%rip),%xmm0        # 5240 <_sk_callback_sse41+0xfb0>
+  .byte  68,15,40,13,229,17,0,0              // movaps        0x11e5(%rip),%xmm9        # 5250 <_sk_callback_sse41+0xfc0>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,245,17,0,0              // mulps         0x11f5(%rip),%xmm9        # 5250 <_sk_callback_sse41+0xfe4>
-  .byte  68,15,88,13,253,17,0,0              // addps         0x11fd(%rip),%xmm9        # 5260 <_sk_callback_sse41+0xff4>
+  .byte  68,15,89,13,225,17,0,0              // mulps         0x11e1(%rip),%xmm9        # 5260 <_sk_callback_sse41+0xfd0>
+  .byte  68,15,88,13,233,17,0,0              // addps         0x11e9(%rip),%xmm9        # 5270 <_sk_callback_sse41+0xfe0>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -23882,16 +23927,16 @@ _sk_bicubic_n1x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,236,17,0,0                  // addps         0x11ec(%rip),%xmm0        # 5270 <_sk_callback_sse41+0x1004>
-  .byte  68,15,40,13,244,17,0,0              // movaps        0x11f4(%rip),%xmm9        # 5280 <_sk_callback_sse41+0x1014>
+  .byte  15,88,5,216,17,0,0                  // addps         0x11d8(%rip),%xmm0        # 5280 <_sk_callback_sse41+0xff0>
+  .byte  68,15,40,13,224,17,0,0              // movaps        0x11e0(%rip),%xmm9        # 5290 <_sk_callback_sse41+0x1000>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,248,17,0,0               // movaps        0x11f8(%rip),%xmm8        # 5290 <_sk_callback_sse41+0x1024>
+  .byte  68,15,40,5,228,17,0,0               // movaps        0x11e4(%rip),%xmm8        # 52a0 <_sk_callback_sse41+0x1010>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,252,17,0,0               // addps         0x11fc(%rip),%xmm8        # 52a0 <_sk_callback_sse41+0x1034>
+  .byte  68,15,88,5,232,17,0,0               // addps         0x11e8(%rip),%xmm8        # 52b0 <_sk_callback_sse41+0x1020>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,0,18,0,0                 // addps         0x1200(%rip),%xmm8        # 52b0 <_sk_callback_sse41+0x1044>
+  .byte  68,15,88,5,236,17,0,0               // addps         0x11ec(%rip),%xmm8        # 52c0 <_sk_callback_sse41+0x1030>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,4,18,0,0                 // addps         0x1204(%rip),%xmm8        # 52c0 <_sk_callback_sse41+0x1054>
+  .byte  68,15,88,5,240,17,0,0               // addps         0x11f0(%rip),%xmm8        # 52d0 <_sk_callback_sse41+0x1040>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -23901,17 +23946,17 @@ HIDDEN _sk_bicubic_p1x_sse41
 FUNCTION(_sk_bicubic_p1x_sse41)
 _sk_bicubic_p1x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,254,17,0,0               // movaps        0x11fe(%rip),%xmm8        # 52d0 <_sk_callback_sse41+0x1064>
+  .byte  68,15,40,5,234,17,0,0               // movaps        0x11ea(%rip),%xmm8        # 52e0 <_sk_callback_sse41+0x1050>
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,72,64                      // movups        0x40(%rax),%xmm9
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
-  .byte  68,15,40,21,250,17,0,0              // movaps        0x11fa(%rip),%xmm10        # 52e0 <_sk_callback_sse41+0x1074>
+  .byte  68,15,40,21,230,17,0,0              // movaps        0x11e6(%rip),%xmm10        # 52f0 <_sk_callback_sse41+0x1060>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,254,17,0,0              // addps         0x11fe(%rip),%xmm10        # 52f0 <_sk_callback_sse41+0x1084>
+  .byte  68,15,88,21,234,17,0,0              // addps         0x11ea(%rip),%xmm10        # 5300 <_sk_callback_sse41+0x1070>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,250,17,0,0              // addps         0x11fa(%rip),%xmm10        # 5300 <_sk_callback_sse41+0x1094>
+  .byte  68,15,88,21,230,17,0,0              // addps         0x11e6(%rip),%xmm10        # 5310 <_sk_callback_sse41+0x1080>
   .byte  68,15,17,144,128,0,0,0              // movups        %xmm10,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -23923,11 +23968,11 @@ _sk_bicubic_p3x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,237,17,0,0                  // addps         0x11ed(%rip),%xmm0        # 5310 <_sk_callback_sse41+0x10a4>
+  .byte  15,88,5,217,17,0,0                  // addps         0x11d9(%rip),%xmm0        # 5320 <_sk_callback_sse41+0x1090>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,237,17,0,0               // mulps         0x11ed(%rip),%xmm8        # 5320 <_sk_callback_sse41+0x10b4>
-  .byte  68,15,88,5,245,17,0,0               // addps         0x11f5(%rip),%xmm8        # 5330 <_sk_callback_sse41+0x10c4>
+  .byte  68,15,89,5,217,17,0,0               // mulps         0x11d9(%rip),%xmm8        # 5330 <_sk_callback_sse41+0x10a0>
+  .byte  68,15,88,5,225,17,0,0               // addps         0x11e1(%rip),%xmm8        # 5340 <_sk_callback_sse41+0x10b0>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -23940,13 +23985,13 @@ _sk_bicubic_n3y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,227,17,0,0                 // addps         0x11e3(%rip),%xmm1        # 5340 <_sk_callback_sse41+0x10d4>
-  .byte  68,15,40,13,235,17,0,0              // movaps        0x11eb(%rip),%xmm9        # 5350 <_sk_callback_sse41+0x10e4>
+  .byte  15,88,13,207,17,0,0                 // addps         0x11cf(%rip),%xmm1        # 5350 <_sk_callback_sse41+0x10c0>
+  .byte  68,15,40,13,215,17,0,0              // movaps        0x11d7(%rip),%xmm9        # 5360 <_sk_callback_sse41+0x10d0>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,231,17,0,0              // mulps         0x11e7(%rip),%xmm9        # 5360 <_sk_callback_sse41+0x10f4>
-  .byte  68,15,88,13,239,17,0,0              // addps         0x11ef(%rip),%xmm9        # 5370 <_sk_callback_sse41+0x1104>
+  .byte  68,15,89,13,211,17,0,0              // mulps         0x11d3(%rip),%xmm9        # 5370 <_sk_callback_sse41+0x10e0>
+  .byte  68,15,88,13,219,17,0,0              // addps         0x11db(%rip),%xmm9        # 5380 <_sk_callback_sse41+0x10f0>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -23959,16 +24004,16 @@ _sk_bicubic_n1y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,221,17,0,0                 // addps         0x11dd(%rip),%xmm1        # 5380 <_sk_callback_sse41+0x1114>
-  .byte  68,15,40,13,229,17,0,0              // movaps        0x11e5(%rip),%xmm9        # 5390 <_sk_callback_sse41+0x1124>
+  .byte  15,88,13,201,17,0,0                 // addps         0x11c9(%rip),%xmm1        # 5390 <_sk_callback_sse41+0x1100>
+  .byte  68,15,40,13,209,17,0,0              // movaps        0x11d1(%rip),%xmm9        # 53a0 <_sk_callback_sse41+0x1110>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,233,17,0,0               // movaps        0x11e9(%rip),%xmm8        # 53a0 <_sk_callback_sse41+0x1134>
+  .byte  68,15,40,5,213,17,0,0               // movaps        0x11d5(%rip),%xmm8        # 53b0 <_sk_callback_sse41+0x1120>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,237,17,0,0               // addps         0x11ed(%rip),%xmm8        # 53b0 <_sk_callback_sse41+0x1144>
+  .byte  68,15,88,5,217,17,0,0               // addps         0x11d9(%rip),%xmm8        # 53c0 <_sk_callback_sse41+0x1130>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,241,17,0,0               // addps         0x11f1(%rip),%xmm8        # 53c0 <_sk_callback_sse41+0x1154>
+  .byte  68,15,88,5,221,17,0,0               // addps         0x11dd(%rip),%xmm8        # 53d0 <_sk_callback_sse41+0x1140>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,245,17,0,0               // addps         0x11f5(%rip),%xmm8        # 53d0 <_sk_callback_sse41+0x1164>
+  .byte  68,15,88,5,225,17,0,0               // addps         0x11e1(%rip),%xmm8        # 53e0 <_sk_callback_sse41+0x1150>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -23978,17 +24023,17 @@ HIDDEN _sk_bicubic_p1y_sse41
 FUNCTION(_sk_bicubic_p1y_sse41)
 _sk_bicubic_p1y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,239,17,0,0               // movaps        0x11ef(%rip),%xmm8        # 53e0 <_sk_callback_sse41+0x1174>
+  .byte  68,15,40,5,219,17,0,0               // movaps        0x11db(%rip),%xmm8        # 53f0 <_sk_callback_sse41+0x1160>
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,72,96                      // movups        0x60(%rax),%xmm9
   .byte  65,15,88,200                        // addps         %xmm8,%xmm1
-  .byte  68,15,40,21,234,17,0,0              // movaps        0x11ea(%rip),%xmm10        # 53f0 <_sk_callback_sse41+0x1184>
+  .byte  68,15,40,21,214,17,0,0              // movaps        0x11d6(%rip),%xmm10        # 5400 <_sk_callback_sse41+0x1170>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,238,17,0,0              // addps         0x11ee(%rip),%xmm10        # 5400 <_sk_callback_sse41+0x1194>
+  .byte  68,15,88,21,218,17,0,0              // addps         0x11da(%rip),%xmm10        # 5410 <_sk_callback_sse41+0x1180>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,234,17,0,0              // addps         0x11ea(%rip),%xmm10        # 5410 <_sk_callback_sse41+0x11a4>
+  .byte  68,15,88,21,214,17,0,0              // addps         0x11d6(%rip),%xmm10        # 5420 <_sk_callback_sse41+0x1190>
   .byte  68,15,17,144,160,0,0,0              // movups        %xmm10,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24000,11 +24045,11 @@ _sk_bicubic_p3y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,220,17,0,0                 // addps         0x11dc(%rip),%xmm1        # 5420 <_sk_callback_sse41+0x11b4>
+  .byte  15,88,13,200,17,0,0                 // addps         0x11c8(%rip),%xmm1        # 5430 <_sk_callback_sse41+0x11a0>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,220,17,0,0               // mulps         0x11dc(%rip),%xmm8        # 5430 <_sk_callback_sse41+0x11c4>
-  .byte  68,15,88,5,228,17,0,0               // addps         0x11e4(%rip),%xmm8        # 5440 <_sk_callback_sse41+0x11d4>
+  .byte  68,15,89,5,200,17,0,0               // mulps         0x11c8(%rip),%xmm8        # 5440 <_sk_callback_sse41+0x11b0>
+  .byte  68,15,88,5,208,17,0,0               // addps         0x11d0(%rip),%xmm8        # 5450 <_sk_callback_sse41+0x11c0>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -24223,11 +24268,11 @@ BALIGN16
   .byte  128,191,0,0,128,191,0               // cmpb          $0x0,-0x40800000(%rdi)
   .byte  0,224                               // add           %ah,%al
   .byte  64,0,0                              // add           %al,(%rax)
-  .byte  224,64                              // loopne        4528 <.literal16+0x1d8>
+  .byte  224,64                              // loopne        4548 <.literal16+0x1d8>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        452c <.literal16+0x1dc>
+  .byte  224,64                              // loopne        454c <.literal16+0x1dc>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        4530 <.literal16+0x1e0>
+  .byte  224,64                              // loopne        4550 <.literal16+0x1e0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -24252,13 +24297,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4561 <.literal16+0x211>
+  .byte  71,225,61                           // rex.RXB       loope 4581 <.literal16+0x211>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4565 <.literal16+0x215>
+  .byte  71,225,61                           // rex.RXB       loope 4585 <.literal16+0x215>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4569 <.literal16+0x219>
+  .byte  71,225,61                           // rex.RXB       loope 4589 <.literal16+0x219>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 456d <.literal16+0x21d>
+  .byte  71,225,61                           // rex.RXB       loope 458d <.literal16+0x21d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -24283,13 +24328,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45a1 <.literal16+0x251>
+  .byte  71,225,61                           // rex.RXB       loope 45c1 <.literal16+0x251>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45a5 <.literal16+0x255>
+  .byte  71,225,61                           // rex.RXB       loope 45c5 <.literal16+0x255>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45a9 <.literal16+0x259>
+  .byte  71,225,61                           // rex.RXB       loope 45c9 <.literal16+0x259>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45ad <.literal16+0x25d>
+  .byte  71,225,61                           // rex.RXB       loope 45cd <.literal16+0x25d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -24314,13 +24359,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45e1 <.literal16+0x291>
+  .byte  71,225,61                           // rex.RXB       loope 4601 <.literal16+0x291>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45e5 <.literal16+0x295>
+  .byte  71,225,61                           // rex.RXB       loope 4605 <.literal16+0x295>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45e9 <.literal16+0x299>
+  .byte  71,225,61                           // rex.RXB       loope 4609 <.literal16+0x299>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 45ed <.literal16+0x29d>
+  .byte  71,225,61                           // rex.RXB       loope 460d <.literal16+0x29d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -24345,13 +24390,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4621 <.literal16+0x2d1>
+  .byte  71,225,61                           // rex.RXB       loope 4641 <.literal16+0x2d1>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4625 <.literal16+0x2d5>
+  .byte  71,225,61                           // rex.RXB       loope 4645 <.literal16+0x2d5>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4629 <.literal16+0x2d9>
+  .byte  71,225,61                           // rex.RXB       loope 4649 <.literal16+0x2d9>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 462d <.literal16+0x2dd>
+  .byte  71,225,61                           // rex.RXB       loope 464d <.literal16+0x2dd>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -24570,13 +24615,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        47e9 <.literal16+0x499>
+  .byte  224,7                               // loopne        4809 <.literal16+0x499>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        47ed <.literal16+0x49d>
+  .byte  224,7                               // loopne        480d <.literal16+0x49d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        47f1 <.literal16+0x4a1>
+  .byte  224,7                               // loopne        4811 <.literal16+0x4a1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        47f5 <.literal16+0x4a5>
+  .byte  224,7                               // loopne        4815 <.literal16+0x4a5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -24600,26 +24645,20 @@ BALIGN16
   .byte  4,61                                // add           $0x3d,%al
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  4,61                                // add           $0x3d,%al
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  128,63,0                            // cmpb          $0x0,(%rdi)
-  .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
-  .byte  63                                  // (bad)
-  .byte  0,0                                 // add           %al,(%rax)
-  .byte  128,63,255                          // cmpb          $0xff,(%rdi)
-  .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,255                               // add           %bh,%bh
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,255                               // add           %bh,%bh
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,255                               // add           %bh,%bh
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,1                                 // add           %al,(%rcx)
-  .byte  255                                 // (bad)
+  .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004848 <_sk_callback_sse41+0xa0005dc>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004858 <_sk_callback_sse41+0xa0005c8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004850 <_sk_callback_sse41+0x30005e4>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004860 <_sk_callback_sse41+0x30005d0>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -24674,11 +24713,11 @@ BALIGN16
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            491b <.literal16+0x5cb>
+  .byte  127,67                              // jg            492b <.literal16+0x5bb>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            491f <.literal16+0x5cf>
+  .byte  127,67                              // jg            492f <.literal16+0x5bf>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4923 <.literal16+0x5d3>
+  .byte  127,67                              // jg            4933 <.literal16+0x5c3>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,129,128,128,59           // addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -24693,16 +24732,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4914 <.literal16+0x5c4>
+  .byte  127,0                               // jg            4924 <.literal16+0x5b4>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4918 <.literal16+0x5c8>
+  .byte  127,0                               // jg            4928 <.literal16+0x5b8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            491c <.literal16+0x5cc>
+  .byte  127,0                               // jg            492c <.literal16+0x5bc>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4920 <.literal16+0x5d0>
+  .byte  127,0                               // jg            4930 <.literal16+0x5c0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -24711,7 +24750,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            49a5 <.literal16+0x655>
+  .byte  119,115                             // ja            49b5 <.literal16+0x645>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -24722,7 +24761,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4909 <.literal16+0x5b9>
+  .byte  117,191                             // jne           4919 <.literal16+0x5a9>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -24734,7 +24773,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3894a <_sk_callback_sse41+0xffffffffe9a346de>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3895a <_sk_callback_sse41+0xffffffffe9a346ca>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -24789,16 +24828,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            49e4 <.literal16+0x694>
+  .byte  127,0                               // jg            49f4 <.literal16+0x684>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            49e8 <.literal16+0x698>
+  .byte  127,0                               // jg            49f8 <.literal16+0x688>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            49ec <.literal16+0x69c>
+  .byte  127,0                               // jg            49fc <.literal16+0x68c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            49f0 <.literal16+0x6a0>
+  .byte  127,0                               // jg            4a00 <.literal16+0x690>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -24807,7 +24846,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4a75 <.literal16+0x725>
+  .byte  119,115                             // ja            4a85 <.literal16+0x715>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -24818,7 +24857,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           49d9 <.literal16+0x689>
+  .byte  117,191                             // jne           49e9 <.literal16+0x679>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -24830,7 +24869,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38a1a <_sk_callback_sse41+0xffffffffe9a347ae>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38a2a <_sk_callback_sse41+0xffffffffe9a3479a>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -24885,16 +24924,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4ab4 <.literal16+0x764>
+  .byte  127,0                               // jg            4ac4 <.literal16+0x754>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4ab8 <.literal16+0x768>
+  .byte  127,0                               // jg            4ac8 <.literal16+0x758>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4abc <.literal16+0x76c>
+  .byte  127,0                               // jg            4acc <.literal16+0x75c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4ac0 <.literal16+0x770>
+  .byte  127,0                               // jg            4ad0 <.literal16+0x760>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -24903,7 +24942,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4b45 <.literal16+0x7f5>
+  .byte  119,115                             // ja            4b55 <.literal16+0x7e5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -24914,7 +24953,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4aa9 <.literal16+0x759>
+  .byte  117,191                             // jne           4ab9 <.literal16+0x749>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -24926,7 +24965,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38aea <_sk_callback_sse41+0xffffffffe9a3487e>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38afa <_sk_callback_sse41+0xffffffffe9a3486a>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -24981,16 +25020,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4b84 <.literal16+0x834>
+  .byte  127,0                               // jg            4b94 <.literal16+0x824>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4b88 <.literal16+0x838>
+  .byte  127,0                               // jg            4b98 <.literal16+0x828>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4b8c <.literal16+0x83c>
+  .byte  127,0                               // jg            4b9c <.literal16+0x82c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4b90 <.literal16+0x840>
+  .byte  127,0                               // jg            4ba0 <.literal16+0x830>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -24999,7 +25038,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4c15 <.literal16+0x8c5>
+  .byte  119,115                             // ja            4c25 <.literal16+0x8b5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -25010,7 +25049,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4b79 <.literal16+0x829>
+  .byte  117,191                             // jne           4b89 <.literal16+0x819>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -25022,7 +25061,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38bba <_sk_callback_sse41+0xffffffffe9a3494e>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38bca <_sk_callback_sse41+0xffffffffe9a3493a>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -25073,13 +25112,13 @@ BALIGN16
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
-  .byte  127,67                              // jg            4c97 <.literal16+0x947>
+  .byte  127,67                              // jg            4ca7 <.literal16+0x937>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4c9b <.literal16+0x94b>
+  .byte  127,67                              // jg            4cab <.literal16+0x93b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4c9f <.literal16+0x94f>
+  .byte  127,67                              // jg            4caf <.literal16+0x93f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4ca3 <.literal16+0x953>
+  .byte  127,67                              // jg            4cb3 <.literal16+0x943>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -25126,16 +25165,16 @@ BALIGN16
   .byte  128,3,62                            // addb          $0x3e,(%rbx)
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           4d23 <.literal16+0x9d3>
+  .byte  118,63                              // jbe           4d33 <.literal16+0x9c3>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           4d27 <.literal16+0x9d7>
+  .byte  118,63                              // jbe           4d37 <.literal16+0x9c7>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           4d2b <.literal16+0x9db>
+  .byte  118,63                              // jbe           4d3b <.literal16+0x9cb>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           4d2f <.literal16+0x9df>
+  .byte  118,63                              // jbe           4d3f <.literal16+0x9cf>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
@@ -25147,11 +25186,11 @@ BALIGN16
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4d6b <.literal16+0xa1b>
+  .byte  127,67                              // jg            4d7b <.literal16+0xa0b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4d6f <.literal16+0xa1f>
+  .byte  127,67                              // jg            4d7f <.literal16+0xa0f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4d73 <.literal16+0xa23>
+  .byte  127,67                              // jg            4d83 <.literal16+0xa13>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,0,0,128,63               // addb          $0x3f,-0x7fffffc5(%rax)
@@ -25180,7 +25219,7 @@ BALIGN16
   .byte  5,255,255,255,9                     // add           $0x9ffffff,%eax
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004da0 <_sk_callback_sse41+0x3000b34>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004db0 <_sk_callback_sse41+0x3000b20>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -25209,13 +25248,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4dd9 <.literal16+0xa89>
+  .byte  224,7                               // loopne        4de9 <.literal16+0xa79>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4ddd <.literal16+0xa8d>
+  .byte  224,7                               // loopne        4ded <.literal16+0xa7d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4de1 <.literal16+0xa91>
+  .byte  224,7                               // loopne        4df1 <.literal16+0xa81>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4de5 <.literal16+0xa95>
+  .byte  224,7                               // loopne        4df5 <.literal16+0xa85>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -25261,13 +25300,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4e49 <.literal16+0xaf9>
+  .byte  224,7                               // loopne        4e59 <.literal16+0xae9>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4e4d <.literal16+0xafd>
+  .byte  224,7                               // loopne        4e5d <.literal16+0xaed>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4e51 <.literal16+0xb01>
+  .byte  224,7                               // loopne        4e61 <.literal16+0xaf1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4e55 <.literal16+0xb05>
+  .byte  224,7                               // loopne        4e65 <.literal16+0xaf5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -25305,13 +25344,13 @@ BALIGN16
   .byte  65,0,0                              // add           %al,(%r8)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            4ee6 <.literal16+0xb96>
+  .byte  124,66                              // jl            4ef6 <.literal16+0xb86>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            4eea <.literal16+0xb9a>
+  .byte  124,66                              // jl            4efa <.literal16+0xb8a>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            4eee <.literal16+0xb9e>
+  .byte  124,66                              // jl            4efe <.literal16+0xb8e>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            4ef2 <.literal16+0xba2>
+  .byte  124,66                              // jl            4f02 <.literal16+0xb92>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,240                               // add           %dh,%al
@@ -25401,13 +25440,13 @@ BALIGN16
   .byte  136,136,61,137,136,136              // mov           %cl,-0x777776c3(%rax)
   .byte  61,137,136,136,61                   // cmp           $0x3d888889,%eax
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            4ff5 <.literal16+0xca5>
+  .byte  112,65                              // jo            5005 <.literal16+0xc95>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            4ff9 <.literal16+0xca9>
+  .byte  112,65                              // jo            5009 <.literal16+0xc99>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            4ffd <.literal16+0xcad>
+  .byte  112,65                              // jo            500d <.literal16+0xc9d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5001 <.literal16+0xcb1>
+  .byte  112,65                              // jo            5011 <.literal16+0xca1>
   .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  255,0                               // incl          (%rax)
@@ -25422,7 +25461,7 @@ BALIGN16
   .byte  5,255,255,255,9                     // add           $0x9ffffff,%eax
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004ff0 <_sk_callback_sse41+0x3000d84>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3005000 <_sk_callback_sse41+0x3000d70>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -25449,7 +25488,7 @@ BALIGN16
   .byte  5,255,255,255,9                     // add           $0x9ffffff,%eax
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3005030 <_sk_callback_sse41+0x3000dc4>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3005040 <_sk_callback_sse41+0x3000db0>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -25464,11 +25503,11 @@ BALIGN16
   .byte  255,0                               // incl          (%rax)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            508b <.literal16+0xd3b>
+  .byte  127,67                              // jg            509b <.literal16+0xd2b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            508f <.literal16+0xd3f>
+  .byte  127,67                              // jg            509f <.literal16+0xd2f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5093 <.literal16+0xd43>
+  .byte  127,67                              // jg            50a3 <.literal16+0xd33>
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
@@ -25544,13 +25583,13 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  255                                 // (bad)
-  .byte  127,71                              // jg            515b <.literal16+0xe0b>
+  .byte  127,71                              // jg            516b <.literal16+0xdfb>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            515f <.literal16+0xe0f>
+  .byte  127,71                              // jg            516f <.literal16+0xdff>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            5163 <.literal16+0xe13>
+  .byte  127,71                              // jg            5173 <.literal16+0xe03>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            5167 <.literal16+0xe17>
+  .byte  127,71                              // jg            5177 <.literal16+0xe07>
   .byte  208                                 // (bad)
   .byte  179,89                              // mov           $0x59,%bl
   .byte  62,208                              // ds            (bad)
@@ -25676,11 +25715,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5272 <.literal16+0xf22>
+  .byte  62,114,28                           // jb,pt         5282 <.literal16+0xf12>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5276 <.literal16+0xf26>
+  .byte  62,114,28                           // jb,pt         5286 <.literal16+0xf16>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         527a <.literal16+0xf2a>
+  .byte  62,114,28                           // jb,pt         528a <.literal16+0xf1a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -25724,7 +25763,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e105 <_sk_callback_sse41+0x3d639e99>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e115 <_sk_callback_sse41+0x3d639e85>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -25750,7 +25789,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e145 <_sk_callback_sse41+0x3d639ed9>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e155 <_sk_callback_sse41+0x3d639ec5>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -25759,13 +25798,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            533e <.literal16+0xfee>
+  .byte  114,28                              // jb            534e <.literal16+0xfde>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5342 <.literal16+0xff2>
+  .byte  62,114,28                           // jb,pt         5352 <.literal16+0xfe2>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5346 <.literal16+0xff6>
+  .byte  62,114,28                           // jb,pt         5356 <.literal16+0xfe6>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         534a <.literal16+0xffa>
+  .byte  62,114,28                           // jb,pt         535a <.literal16+0xfea>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -25786,11 +25825,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5382 <.literal16+0x1032>
+  .byte  62,114,28                           // jb,pt         5392 <.literal16+0x1022>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5386 <.literal16+0x1036>
+  .byte  62,114,28                           // jb,pt         5396 <.literal16+0x1026>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         538a <.literal16+0x103a>
+  .byte  62,114,28                           // jb,pt         539a <.literal16+0x102a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -25834,7 +25873,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e215 <_sk_callback_sse41+0x3d639fa9>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e225 <_sk_callback_sse41+0x3d639f95>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -25860,7 +25899,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e255 <_sk_callback_sse41+0x3d639fe9>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e265 <_sk_callback_sse41+0x3d639fd5>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -25869,13 +25908,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            544e <.literal16+0x10fe>
+  .byte  114,28                              // jb            545e <.literal16+0x10ee>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5452 <_sk_callback_sse41+0x11e6>
+  .byte  62,114,28                           // jb,pt         5462 <_sk_callback_sse41+0x11d2>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5456 <_sk_callback_sse41+0x11ea>
+  .byte  62,114,28                           // jb,pt         5466 <_sk_callback_sse41+0x11d6>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         545a <_sk_callback_sse41+0x11ee>
+  .byte  62,114,28                           // jb,pt         546a <_sk_callback_sse41+0x11da>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -25945,7 +25984,7 @@ _sk_seed_shader_sse2:
   .byte  102,15,110,199                      // movd          %edi,%xmm0
   .byte  102,15,112,192,0                    // pshufd        $0x0,%xmm0,%xmm0
   .byte  15,91,200                           // cvtdq2ps      %xmm0,%xmm1
-  .byte  15,40,21,180,71,0,0                 // movaps        0x47b4(%rip),%xmm2        # 4830 <_sk_callback_sse2+0xe4>
+  .byte  15,40,21,212,71,0,0                 // movaps        0x47d4(%rip),%xmm2        # 4850 <_sk_callback_sse2+0xdf>
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  15,16,2                             // movups        (%rdx),%xmm0
   .byte  15,88,193                           // addps         %xmm1,%xmm0
@@ -25954,7 +25993,7 @@ _sk_seed_shader_sse2:
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,21,163,71,0,0                 // movaps        0x47a3(%rip),%xmm2        # 4840 <_sk_callback_sse2+0xf4>
+  .byte  15,40,21,195,71,0,0                 // movaps        0x47c3(%rip),%xmm2        # 4860 <_sk_callback_sse2+0xef>
   .byte  15,87,219                           // xorps         %xmm3,%xmm3
   .byte  15,87,228                           // xorps         %xmm4,%xmm4
   .byte  15,87,237                           // xorps         %xmm5,%xmm5
@@ -25977,14 +26016,14 @@ _sk_dither_sse2:
   .byte  102,68,15,110,1                     // movd          (%rcx),%xmm8
   .byte  102,69,15,112,192,0                 // pshufd        $0x0,%xmm8,%xmm8
   .byte  102,69,15,239,193                   // pxor          %xmm9,%xmm8
-  .byte  102,68,15,111,21,104,71,0,0         // movdqa        0x4768(%rip),%xmm10        # 4850 <_sk_callback_sse2+0x104>
+  .byte  102,68,15,111,21,136,71,0,0         // movdqa        0x4788(%rip),%xmm10        # 4870 <_sk_callback_sse2+0xff>
   .byte  102,69,15,111,216                   // movdqa        %xmm8,%xmm11
   .byte  102,69,15,219,218                   // pand          %xmm10,%xmm11
   .byte  102,65,15,114,243,5                 // pslld         $0x5,%xmm11
   .byte  102,69,15,219,209                   // pand          %xmm9,%xmm10
   .byte  102,65,15,114,242,4                 // pslld         $0x4,%xmm10
-  .byte  102,68,15,111,37,84,71,0,0          // movdqa        0x4754(%rip),%xmm12        # 4860 <_sk_callback_sse2+0x114>
-  .byte  102,68,15,111,45,91,71,0,0          // movdqa        0x475b(%rip),%xmm13        # 4870 <_sk_callback_sse2+0x124>
+  .byte  102,68,15,111,37,116,71,0,0         // movdqa        0x4774(%rip),%xmm12        # 4880 <_sk_callback_sse2+0x10f>
+  .byte  102,68,15,111,45,123,71,0,0         // movdqa        0x477b(%rip),%xmm13        # 4890 <_sk_callback_sse2+0x11f>
   .byte  102,69,15,111,240                   // movdqa        %xmm8,%xmm14
   .byte  102,69,15,219,245                   // pand          %xmm13,%xmm14
   .byte  102,65,15,114,246,2                 // pslld         $0x2,%xmm14
@@ -26000,8 +26039,8 @@ _sk_dither_sse2:
   .byte  102,69,15,235,245                   // por           %xmm13,%xmm14
   .byte  102,69,15,235,240                   // por           %xmm8,%xmm14
   .byte  69,15,91,198                        // cvtdq2ps      %xmm14,%xmm8
-  .byte  68,15,89,5,22,71,0,0                // mulps         0x4716(%rip),%xmm8        # 4880 <_sk_callback_sse2+0x134>
-  .byte  68,15,88,5,30,71,0,0                // addps         0x471e(%rip),%xmm8        # 4890 <_sk_callback_sse2+0x144>
+  .byte  68,15,89,5,54,71,0,0                // mulps         0x4736(%rip),%xmm8        # 48a0 <_sk_callback_sse2+0x12f>
+  .byte  68,15,88,5,62,71,0,0                // addps         0x473e(%rip),%xmm8        # 48b0 <_sk_callback_sse2+0x13f>
   .byte  243,68,15,16,72,8                   // movss         0x8(%rax),%xmm9
   .byte  69,15,198,201,0                     // shufps        $0x0,%xmm9,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
@@ -26043,7 +26082,7 @@ HIDDEN _sk_srcatop_sse2
 FUNCTION(_sk_srcatop_sse2)
 _sk_srcatop_sse2:
   .byte  15,89,199                           // mulps         %xmm7,%xmm0
-  .byte  68,15,40,5,203,70,0,0               // movaps        0x46cb(%rip),%xmm8        # 48a0 <_sk_callback_sse2+0x154>
+  .byte  68,15,40,5,235,70,0,0               // movaps        0x46eb(%rip),%xmm8        # 48c0 <_sk_callback_sse2+0x14f>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -26068,7 +26107,7 @@ FUNCTION(_sk_dstatop_sse2)
 _sk_dstatop_sse2:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
   .byte  68,15,89,196                        // mulps         %xmm4,%xmm8
-  .byte  68,15,40,13,142,70,0,0              // movaps        0x468e(%rip),%xmm9        # 48b0 <_sk_callback_sse2+0x164>
+  .byte  68,15,40,13,174,70,0,0              // movaps        0x46ae(%rip),%xmm9        # 48d0 <_sk_callback_sse2+0x15f>
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
@@ -26115,7 +26154,7 @@ HIDDEN _sk_srcout_sse2
 .globl _sk_srcout_sse2
 FUNCTION(_sk_srcout_sse2)
 _sk_srcout_sse2:
-  .byte  68,15,40,5,50,70,0,0                // movaps        0x4632(%rip),%xmm8        # 48c0 <_sk_callback_sse2+0x174>
+  .byte  68,15,40,5,82,70,0,0                // movaps        0x4652(%rip),%xmm8        # 48e0 <_sk_callback_sse2+0x16f>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
@@ -26128,7 +26167,7 @@ HIDDEN _sk_dstout_sse2
 .globl _sk_dstout_sse2
 FUNCTION(_sk_dstout_sse2)
 _sk_dstout_sse2:
-  .byte  68,15,40,5,34,70,0,0                // movaps        0x4622(%rip),%xmm8        # 48d0 <_sk_callback_sse2+0x184>
+  .byte  68,15,40,5,66,70,0,0                // movaps        0x4642(%rip),%xmm8        # 48f0 <_sk_callback_sse2+0x17f>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  15,89,196                           // mulps         %xmm4,%xmm0
@@ -26145,7 +26184,7 @@ HIDDEN _sk_srcover_sse2
 .globl _sk_srcover_sse2
 FUNCTION(_sk_srcover_sse2)
 _sk_srcover_sse2:
-  .byte  68,15,40,5,5,70,0,0                 // movaps        0x4605(%rip),%xmm8        # 48e0 <_sk_callback_sse2+0x194>
+  .byte  68,15,40,5,37,70,0,0                // movaps        0x4625(%rip),%xmm8        # 4900 <_sk_callback_sse2+0x18f>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -26165,7 +26204,7 @@ HIDDEN _sk_dstover_sse2
 .globl _sk_dstover_sse2
 FUNCTION(_sk_dstover_sse2)
 _sk_dstover_sse2:
-  .byte  68,15,40,5,217,69,0,0               // movaps        0x45d9(%rip),%xmm8        # 48f0 <_sk_callback_sse2+0x1a4>
+  .byte  68,15,40,5,249,69,0,0               // movaps        0x45f9(%rip),%xmm8        # 4910 <_sk_callback_sse2+0x19f>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -26193,7 +26232,7 @@ HIDDEN _sk_multiply_sse2
 .globl _sk_multiply_sse2
 FUNCTION(_sk_multiply_sse2)
 _sk_multiply_sse2:
-  .byte  68,15,40,5,173,69,0,0               // movaps        0x45ad(%rip),%xmm8        # 4900 <_sk_callback_sse2+0x1b4>
+  .byte  68,15,40,5,205,69,0,0               // movaps        0x45cd(%rip),%xmm8        # 4920 <_sk_callback_sse2+0x1af>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
@@ -26269,7 +26308,7 @@ HIDDEN _sk_xor__sse2
 FUNCTION(_sk_xor__sse2)
 _sk_xor__sse2:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
-  .byte  15,40,29,222,68,0,0                 // movaps        0x44de(%rip),%xmm3        # 4910 <_sk_callback_sse2+0x1c4>
+  .byte  15,40,29,254,68,0,0                 // movaps        0x44fe(%rip),%xmm3        # 4930 <_sk_callback_sse2+0x1bf>
   .byte  68,15,40,203                        // movaps        %xmm3,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
@@ -26317,7 +26356,7 @@ _sk_darken_sse2:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,95,209                        // maxps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,73,68,0,0                  // movaps        0x4449(%rip),%xmm2        # 4920 <_sk_callback_sse2+0x1d4>
+  .byte  15,40,21,105,68,0,0                 // movaps        0x4469(%rip),%xmm2        # 4940 <_sk_callback_sse2+0x1cf>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -26351,7 +26390,7 @@ _sk_lighten_sse2:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,238,67,0,0                 // movaps        0x43ee(%rip),%xmm2        # 4930 <_sk_callback_sse2+0x1e4>
+  .byte  15,40,21,14,68,0,0                  // movaps        0x440e(%rip),%xmm2        # 4950 <_sk_callback_sse2+0x1df>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -26388,7 +26427,7 @@ _sk_difference_sse2:
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,136,67,0,0                 // movaps        0x4388(%rip),%xmm2        # 4940 <_sk_callback_sse2+0x1f4>
+  .byte  15,40,21,168,67,0,0                 // movaps        0x43a8(%rip),%xmm2        # 4960 <_sk_callback_sse2+0x1ef>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -26415,7 +26454,7 @@ _sk_exclusion_sse2:
   .byte  15,89,214                           // mulps         %xmm6,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,202                        // subps         %xmm2,%xmm9
-  .byte  15,40,13,73,67,0,0                  // movaps        0x4349(%rip),%xmm1        # 4950 <_sk_callback_sse2+0x204>
+  .byte  15,40,13,105,67,0,0                 // movaps        0x4369(%rip),%xmm1        # 4970 <_sk_callback_sse2+0x1ff>
   .byte  15,92,203                           // subps         %xmm3,%xmm1
   .byte  15,89,207                           // mulps         %xmm7,%xmm1
   .byte  15,88,217                           // addps         %xmm1,%xmm3
@@ -26429,7 +26468,7 @@ HIDDEN _sk_colorburn_sse2
 FUNCTION(_sk_colorburn_sse2)
 _sk_colorburn_sse2:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,40,21,56,67,0,0               // movaps        0x4338(%rip),%xmm10        # 4960 <_sk_callback_sse2+0x214>
+  .byte  68,15,40,21,88,67,0,0               // movaps        0x4358(%rip),%xmm10        # 4980 <_sk_callback_sse2+0x20f>
   .byte  69,15,40,202                        // movaps        %xmm10,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  69,15,40,217                        // movaps        %xmm9,%xmm11
@@ -26523,7 +26562,7 @@ HIDDEN _sk_colordodge_sse2
 FUNCTION(_sk_colordodge_sse2)
 _sk_colordodge_sse2:
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
-  .byte  68,15,40,21,238,65,0,0              // movaps        0x41ee(%rip),%xmm10        # 4970 <_sk_callback_sse2+0x224>
+  .byte  68,15,40,21,14,66,0,0               // movaps        0x420e(%rip),%xmm10        # 4990 <_sk_callback_sse2+0x21f>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  69,15,40,227                        // movaps        %xmm11,%xmm12
@@ -26617,7 +26656,7 @@ _sk_hardlight_sse2:
   .byte  15,41,116,36,232                    // movaps        %xmm6,-0x18(%rsp)
   .byte  15,40,245                           // movaps        %xmm5,%xmm6
   .byte  15,40,236                           // movaps        %xmm4,%xmm5
-  .byte  68,15,40,29,163,64,0,0              // movaps        0x40a3(%rip),%xmm11        # 4980 <_sk_callback_sse2+0x234>
+  .byte  68,15,40,29,195,64,0,0              // movaps        0x40c3(%rip),%xmm11        # 49a0 <_sk_callback_sse2+0x22f>
   .byte  69,15,40,211                        // movaps        %xmm11,%xmm10
   .byte  68,15,92,215                        // subps         %xmm7,%xmm10
   .byte  69,15,40,194                        // movaps        %xmm10,%xmm8
@@ -26705,7 +26744,7 @@ FUNCTION(_sk_overlay_sse2)
 _sk_overlay_sse2:
   .byte  68,15,40,193                        // movaps        %xmm1,%xmm8
   .byte  68,15,40,232                        // movaps        %xmm0,%xmm13
-  .byte  68,15,40,13,113,63,0,0              // movaps        0x3f71(%rip),%xmm9        # 4990 <_sk_callback_sse2+0x244>
+  .byte  68,15,40,13,145,63,0,0              // movaps        0x3f91(%rip),%xmm9        # 49b0 <_sk_callback_sse2+0x23f>
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
   .byte  68,15,92,215                        // subps         %xmm7,%xmm10
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
@@ -26796,7 +26835,7 @@ _sk_softlight_sse2:
   .byte  68,15,40,213                        // movaps        %xmm5,%xmm10
   .byte  68,15,94,215                        // divps         %xmm7,%xmm10
   .byte  69,15,84,212                        // andps         %xmm12,%xmm10
-  .byte  68,15,40,13,46,62,0,0               // movaps        0x3e2e(%rip),%xmm9        # 49a0 <_sk_callback_sse2+0x254>
+  .byte  68,15,40,13,78,62,0,0               // movaps        0x3e4e(%rip),%xmm9        # 49c0 <_sk_callback_sse2+0x24f>
   .byte  69,15,40,249                        // movaps        %xmm9,%xmm15
   .byte  69,15,92,250                        // subps         %xmm10,%xmm15
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
@@ -26809,10 +26848,10 @@ _sk_softlight_sse2:
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  15,89,192                           // mulps         %xmm0,%xmm0
   .byte  65,15,88,194                        // addps         %xmm10,%xmm0
-  .byte  68,15,40,53,8,62,0,0                // movaps        0x3e08(%rip),%xmm14        # 49b0 <_sk_callback_sse2+0x264>
+  .byte  68,15,40,53,40,62,0,0               // movaps        0x3e28(%rip),%xmm14        # 49d0 <_sk_callback_sse2+0x25f>
   .byte  69,15,88,222                        // addps         %xmm14,%xmm11
   .byte  68,15,89,216                        // mulps         %xmm0,%xmm11
-  .byte  68,15,40,21,8,62,0,0                // movaps        0x3e08(%rip),%xmm10        # 49c0 <_sk_callback_sse2+0x274>
+  .byte  68,15,40,21,40,62,0,0               // movaps        0x3e28(%rip),%xmm10        # 49e0 <_sk_callback_sse2+0x26f>
   .byte  69,15,89,234                        // mulps         %xmm10,%xmm13
   .byte  69,15,88,235                        // addps         %xmm11,%xmm13
   .byte  15,88,228                           // addps         %xmm4,%xmm4
@@ -26958,7 +26997,7 @@ _sk_hue_sse2:
   .byte  15,40,236                           // movaps        %xmm4,%xmm5
   .byte  15,40,227                           // movaps        %xmm3,%xmm4
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
-  .byte  68,15,40,13,26,60,0,0               // movaps        0x3c1a(%rip),%xmm9        # 49d0 <_sk_callback_sse2+0x284>
+  .byte  68,15,40,13,58,60,0,0               // movaps        0x3c3a(%rip),%xmm9        # 49f0 <_sk_callback_sse2+0x27f>
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
   .byte  68,15,94,212                        // divps         %xmm4,%xmm10
   .byte  68,15,40,228                        // movaps        %xmm4,%xmm12
@@ -27000,12 +27039,12 @@ _sk_hue_sse2:
   .byte  65,15,84,199                        // andps         %xmm15,%xmm0
   .byte  65,15,84,207                        // andps         %xmm15,%xmm1
   .byte  69,15,84,231                        // andps         %xmm15,%xmm12
-  .byte  68,15,40,61,127,59,0,0              // movaps        0x3b7f(%rip),%xmm15        # 49e0 <_sk_callback_sse2+0x294>
+  .byte  68,15,40,61,159,59,0,0              // movaps        0x3b9f(%rip),%xmm15        # 4a00 <_sk_callback_sse2+0x28f>
   .byte  69,15,89,247                        // mulps         %xmm15,%xmm14
-  .byte  15,40,29,132,59,0,0                 // movaps        0x3b84(%rip),%xmm3        # 49f0 <_sk_callback_sse2+0x2a4>
+  .byte  15,40,29,164,59,0,0                 // movaps        0x3ba4(%rip),%xmm3        # 4a10 <_sk_callback_sse2+0x29f>
   .byte  68,15,89,235                        // mulps         %xmm3,%xmm13
   .byte  69,15,88,238                        // addps         %xmm14,%xmm13
-  .byte  68,15,40,21,132,59,0,0              // movaps        0x3b84(%rip),%xmm10        # 4a00 <_sk_callback_sse2+0x2b4>
+  .byte  68,15,40,21,164,59,0,0              // movaps        0x3ba4(%rip),%xmm10        # 4a20 <_sk_callback_sse2+0x2af>
   .byte  68,15,40,223                        // movaps        %xmm7,%xmm11
   .byte  69,15,89,218                        // mulps         %xmm10,%xmm11
   .byte  69,15,88,221                        // addps         %xmm13,%xmm11
@@ -27122,7 +27161,7 @@ _sk_saturation_sse2:
   .byte  68,15,40,193                        // movaps        %xmm1,%xmm8
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  15,87,201                           // xorps         %xmm1,%xmm1
-  .byte  68,15,40,29,230,57,0,0              // movaps        0x39e6(%rip),%xmm11        # 4a10 <_sk_callback_sse2+0x2c4>
+  .byte  68,15,40,29,6,58,0,0                // movaps        0x3a06(%rip),%xmm11        # 4a30 <_sk_callback_sse2+0x2bf>
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  15,94,199                           // divps         %xmm7,%xmm0
   .byte  68,15,40,231                        // movaps        %xmm7,%xmm12
@@ -27162,14 +27201,14 @@ _sk_saturation_sse2:
   .byte  15,84,194                           // andps         %xmm2,%xmm0
   .byte  68,15,84,250                        // andps         %xmm2,%xmm15
   .byte  68,15,84,226                        // andps         %xmm2,%xmm12
-  .byte  68,15,40,45,86,57,0,0               // movaps        0x3956(%rip),%xmm13        # 4a20 <_sk_callback_sse2+0x2d4>
+  .byte  68,15,40,45,118,57,0,0              // movaps        0x3976(%rip),%xmm13        # 4a40 <_sk_callback_sse2+0x2cf>
   .byte  68,15,40,197                        // movaps        %xmm5,%xmm8
   .byte  69,15,89,197                        // mulps         %xmm13,%xmm8
-  .byte  68,15,40,53,86,57,0,0               // movaps        0x3956(%rip),%xmm14        # 4a30 <_sk_callback_sse2+0x2e4>
+  .byte  68,15,40,53,118,57,0,0              // movaps        0x3976(%rip),%xmm14        # 4a50 <_sk_callback_sse2+0x2df>
   .byte  15,40,214                           // movaps        %xmm6,%xmm2
   .byte  65,15,89,214                        // mulps         %xmm14,%xmm2
   .byte  65,15,88,208                        // addps         %xmm8,%xmm2
-  .byte  68,15,40,5,83,57,0,0                // movaps        0x3953(%rip),%xmm8        # 4a40 <_sk_callback_sse2+0x2f4>
+  .byte  68,15,40,5,115,57,0,0               // movaps        0x3973(%rip),%xmm8        # 4a60 <_sk_callback_sse2+0x2ef>
   .byte  69,15,40,202                        // movaps        %xmm10,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,88,202                        // addps         %xmm2,%xmm9
@@ -27285,7 +27324,7 @@ _sk_color_sse2:
   .byte  15,40,227                           // movaps        %xmm3,%xmm4
   .byte  68,15,40,249                        // movaps        %xmm1,%xmm15
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
-  .byte  68,15,40,13,184,55,0,0              // movaps        0x37b8(%rip),%xmm9        # 4a50 <_sk_callback_sse2+0x304>
+  .byte  68,15,40,13,216,55,0,0              // movaps        0x37d8(%rip),%xmm9        # 4a70 <_sk_callback_sse2+0x2ff>
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
   .byte  68,15,94,212                        // divps         %xmm4,%xmm10
   .byte  68,15,40,228                        // movaps        %xmm4,%xmm12
@@ -27294,14 +27333,14 @@ _sk_color_sse2:
   .byte  65,15,89,196                        // mulps         %xmm12,%xmm0
   .byte  69,15,89,252                        // mulps         %xmm12,%xmm15
   .byte  68,15,89,226                        // mulps         %xmm2,%xmm12
-  .byte  68,15,40,45,159,55,0,0              // movaps        0x379f(%rip),%xmm13        # 4a60 <_sk_callback_sse2+0x314>
+  .byte  68,15,40,45,191,55,0,0              // movaps        0x37bf(%rip),%xmm13        # 4a80 <_sk_callback_sse2+0x30f>
   .byte  68,15,40,213                        // movaps        %xmm5,%xmm10
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,40,53,159,55,0,0              // movaps        0x379f(%rip),%xmm14        # 4a70 <_sk_callback_sse2+0x324>
+  .byte  68,15,40,53,191,55,0,0              // movaps        0x37bf(%rip),%xmm14        # 4a90 <_sk_callback_sse2+0x31f>
   .byte  65,15,40,211                        // movaps        %xmm11,%xmm2
   .byte  65,15,89,214                        // mulps         %xmm14,%xmm2
   .byte  65,15,88,210                        // addps         %xmm10,%xmm2
-  .byte  68,15,40,21,155,55,0,0              // movaps        0x379b(%rip),%xmm10        # 4a80 <_sk_callback_sse2+0x334>
+  .byte  68,15,40,21,187,55,0,0              // movaps        0x37bb(%rip),%xmm10        # 4aa0 <_sk_callback_sse2+0x32f>
   .byte  68,15,40,222                        // movaps        %xmm6,%xmm11
   .byte  69,15,89,218                        // mulps         %xmm10,%xmm11
   .byte  68,15,88,218                        // addps         %xmm2,%xmm11
@@ -27418,7 +27457,7 @@ _sk_luminosity_sse2:
   .byte  68,15,40,193                        // movaps        %xmm1,%xmm8
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,87,210                        // xorps         %xmm10,%xmm10
-  .byte  68,15,40,29,249,53,0,0              // movaps        0x35f9(%rip),%xmm11        # 4a90 <_sk_callback_sse2+0x344>
+  .byte  68,15,40,29,25,54,0,0               // movaps        0x3619(%rip),%xmm11        # 4ab0 <_sk_callback_sse2+0x33f>
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  15,94,199                           // divps         %xmm7,%xmm0
   .byte  68,15,40,231                        // movaps        %xmm7,%xmm12
@@ -27429,12 +27468,12 @@ _sk_luminosity_sse2:
   .byte  65,15,40,204                        // movaps        %xmm12,%xmm1
   .byte  15,89,206                           // mulps         %xmm6,%xmm1
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
-  .byte  68,15,40,53,219,53,0,0              // movaps        0x35db(%rip),%xmm14        # 4aa0 <_sk_callback_sse2+0x354>
+  .byte  68,15,40,53,251,53,0,0              // movaps        0x35fb(%rip),%xmm14        # 4ac0 <_sk_callback_sse2+0x34f>
   .byte  69,15,89,206                        // mulps         %xmm14,%xmm9
-  .byte  68,15,40,45,223,53,0,0              // movaps        0x35df(%rip),%xmm13        # 4ab0 <_sk_callback_sse2+0x364>
+  .byte  68,15,40,45,255,53,0,0              // movaps        0x35ff(%rip),%xmm13        # 4ad0 <_sk_callback_sse2+0x35f>
   .byte  69,15,89,197                        // mulps         %xmm13,%xmm8
   .byte  69,15,88,193                        // addps         %xmm9,%xmm8
-  .byte  68,15,40,13,223,53,0,0              // movaps        0x35df(%rip),%xmm9        # 4ac0 <_sk_callback_sse2+0x374>
+  .byte  68,15,40,13,255,53,0,0              // movaps        0x35ff(%rip),%xmm9        # 4ae0 <_sk_callback_sse2+0x36f>
   .byte  65,15,89,217                        // mulps         %xmm9,%xmm3
   .byte  65,15,88,216                        // addps         %xmm8,%xmm3
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
@@ -27554,7 +27593,7 @@ HIDDEN _sk_clamp_1_sse2
 .globl _sk_clamp_1_sse2
 FUNCTION(_sk_clamp_1_sse2)
 _sk_clamp_1_sse2:
-  .byte  68,15,40,5,65,52,0,0                // movaps        0x3441(%rip),%xmm8        # 4ad0 <_sk_callback_sse2+0x384>
+  .byte  68,15,40,5,97,52,0,0                // movaps        0x3461(%rip),%xmm8        # 4af0 <_sk_callback_sse2+0x37f>
   .byte  65,15,93,192                        // minps         %xmm8,%xmm0
   .byte  65,15,93,200                        // minps         %xmm8,%xmm1
   .byte  65,15,93,208                        // minps         %xmm8,%xmm2
@@ -27566,7 +27605,7 @@ HIDDEN _sk_clamp_a_sse2
 .globl _sk_clamp_a_sse2
 FUNCTION(_sk_clamp_a_sse2)
 _sk_clamp_a_sse2:
-  .byte  15,93,29,54,52,0,0                  // minps         0x3436(%rip),%xmm3        # 4ae0 <_sk_callback_sse2+0x394>
+  .byte  15,93,29,86,52,0,0                  // minps         0x3456(%rip),%xmm3        # 4b00 <_sk_callback_sse2+0x38f>
   .byte  15,93,195                           // minps         %xmm3,%xmm0
   .byte  15,93,203                           // minps         %xmm3,%xmm1
   .byte  15,93,211                           // minps         %xmm3,%xmm2
@@ -27653,7 +27692,7 @@ HIDDEN _sk_unpremul_sse2
 FUNCTION(_sk_unpremul_sse2)
 _sk_unpremul_sse2:
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
-  .byte  68,15,40,13,161,51,0,0              // movaps        0x33a1(%rip),%xmm9        # 4af0 <_sk_callback_sse2+0x3a4>
+  .byte  68,15,40,13,193,51,0,0              // movaps        0x33c1(%rip),%xmm9        # 4b10 <_sk_callback_sse2+0x39f>
   .byte  68,15,94,203                        // divps         %xmm3,%xmm9
   .byte  68,15,194,195,4                     // cmpneqps      %xmm3,%xmm8
   .byte  69,15,84,193                        // andps         %xmm9,%xmm8
@@ -27667,20 +27706,20 @@ HIDDEN _sk_from_srgb_sse2
 .globl _sk_from_srgb_sse2
 FUNCTION(_sk_from_srgb_sse2)
 _sk_from_srgb_sse2:
-  .byte  68,15,40,5,140,51,0,0               // movaps        0x338c(%rip),%xmm8        # 4b00 <_sk_callback_sse2+0x3b4>
+  .byte  68,15,40,5,172,51,0,0               // movaps        0x33ac(%rip),%xmm8        # 4b20 <_sk_callback_sse2+0x3af>
   .byte  68,15,40,232                        // movaps        %xmm0,%xmm13
   .byte  69,15,89,232                        // mulps         %xmm8,%xmm13
   .byte  68,15,40,216                        // movaps        %xmm0,%xmm11
   .byte  69,15,89,219                        // mulps         %xmm11,%xmm11
-  .byte  68,15,40,13,132,51,0,0              // movaps        0x3384(%rip),%xmm9        # 4b10 <_sk_callback_sse2+0x3c4>
+  .byte  68,15,40,13,164,51,0,0              // movaps        0x33a4(%rip),%xmm9        # 4b30 <_sk_callback_sse2+0x3bf>
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
   .byte  69,15,89,241                        // mulps         %xmm9,%xmm14
-  .byte  68,15,40,21,132,51,0,0              // movaps        0x3384(%rip),%xmm10        # 4b20 <_sk_callback_sse2+0x3d4>
+  .byte  68,15,40,21,164,51,0,0              // movaps        0x33a4(%rip),%xmm10        # 4b40 <_sk_callback_sse2+0x3cf>
   .byte  69,15,88,242                        // addps         %xmm10,%xmm14
   .byte  69,15,89,243                        // mulps         %xmm11,%xmm14
-  .byte  68,15,40,29,132,51,0,0              // movaps        0x3384(%rip),%xmm11        # 4b30 <_sk_callback_sse2+0x3e4>
+  .byte  68,15,40,29,164,51,0,0              // movaps        0x33a4(%rip),%xmm11        # 4b50 <_sk_callback_sse2+0x3df>
   .byte  69,15,88,243                        // addps         %xmm11,%xmm14
-  .byte  68,15,40,37,136,51,0,0              // movaps        0x3388(%rip),%xmm12        # 4b40 <_sk_callback_sse2+0x3f4>
+  .byte  68,15,40,37,168,51,0,0              // movaps        0x33a8(%rip),%xmm12        # 4b60 <_sk_callback_sse2+0x3ef>
   .byte  65,15,194,196,1                     // cmpltps       %xmm12,%xmm0
   .byte  68,15,84,232                        // andps         %xmm0,%xmm13
   .byte  65,15,85,198                        // andnps        %xmm14,%xmm0
@@ -27719,20 +27758,20 @@ _sk_to_srgb_sse2:
   .byte  68,15,82,192                        // rsqrtps       %xmm0,%xmm8
   .byte  69,15,83,200                        // rcpps         %xmm8,%xmm9
   .byte  69,15,82,232                        // rsqrtps       %xmm8,%xmm13
-  .byte  68,15,40,5,13,51,0,0                // movaps        0x330d(%rip),%xmm8        # 4b50 <_sk_callback_sse2+0x404>
+  .byte  68,15,40,5,45,51,0,0                // movaps        0x332d(%rip),%xmm8        # 4b70 <_sk_callback_sse2+0x3ff>
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
   .byte  69,15,89,240                        // mulps         %xmm8,%xmm14
-  .byte  68,15,40,21,13,51,0,0               // movaps        0x330d(%rip),%xmm10        # 4b60 <_sk_callback_sse2+0x414>
+  .byte  68,15,40,21,45,51,0,0               // movaps        0x332d(%rip),%xmm10        # 4b80 <_sk_callback_sse2+0x40f>
   .byte  69,15,89,202                        // mulps         %xmm10,%xmm9
-  .byte  68,15,40,29,17,51,0,0               // movaps        0x3311(%rip),%xmm11        # 4b70 <_sk_callback_sse2+0x424>
+  .byte  68,15,40,29,49,51,0,0               // movaps        0x3331(%rip),%xmm11        # 4b90 <_sk_callback_sse2+0x41f>
   .byte  69,15,88,203                        // addps         %xmm11,%xmm9
-  .byte  68,15,40,37,21,51,0,0               // movaps        0x3315(%rip),%xmm12        # 4b80 <_sk_callback_sse2+0x434>
+  .byte  68,15,40,37,53,51,0,0               // movaps        0x3335(%rip),%xmm12        # 4ba0 <_sk_callback_sse2+0x42f>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,40,13,21,51,0,0               // movaps        0x3315(%rip),%xmm9        # 4b90 <_sk_callback_sse2+0x444>
+  .byte  68,15,40,13,53,51,0,0               // movaps        0x3335(%rip),%xmm9        # 4bb0 <_sk_callback_sse2+0x43f>
   .byte  69,15,40,249                        // movaps        %xmm9,%xmm15
   .byte  69,15,93,253                        // minps         %xmm13,%xmm15
-  .byte  68,15,40,45,21,51,0,0               // movaps        0x3315(%rip),%xmm13        # 4ba0 <_sk_callback_sse2+0x454>
+  .byte  68,15,40,45,53,51,0,0               // movaps        0x3335(%rip),%xmm13        # 4bc0 <_sk_callback_sse2+0x44f>
   .byte  65,15,194,197,1                     // cmpltps       %xmm13,%xmm0
   .byte  68,15,84,240                        // andps         %xmm0,%xmm14
   .byte  65,15,85,199                        // andnps        %xmm15,%xmm0
@@ -27782,7 +27821,7 @@ _sk_rgb_to_hsl_sse2:
   .byte  68,15,93,218                        // minps         %xmm2,%xmm11
   .byte  65,15,40,202                        // movaps        %xmm10,%xmm1
   .byte  65,15,92,203                        // subps         %xmm11,%xmm1
-  .byte  68,15,40,45,110,50,0,0              // movaps        0x326e(%rip),%xmm13        # 4bb0 <_sk_callback_sse2+0x464>
+  .byte  68,15,40,45,142,50,0,0              // movaps        0x328e(%rip),%xmm13        # 4bd0 <_sk_callback_sse2+0x45f>
   .byte  68,15,94,233                        // divps         %xmm1,%xmm13
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  65,15,194,192,0                     // cmpeqps       %xmm8,%xmm0
@@ -27791,30 +27830,30 @@ _sk_rgb_to_hsl_sse2:
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,40,241                        // movaps        %xmm9,%xmm14
   .byte  68,15,194,242,1                     // cmpltps       %xmm2,%xmm14
-  .byte  68,15,84,53,84,50,0,0               // andps         0x3254(%rip),%xmm14        # 4bc0 <_sk_callback_sse2+0x474>
+  .byte  68,15,84,53,116,50,0,0              // andps         0x3274(%rip),%xmm14        # 4be0 <_sk_callback_sse2+0x46f>
   .byte  69,15,88,244                        // addps         %xmm12,%xmm14
   .byte  69,15,40,250                        // movaps        %xmm10,%xmm15
   .byte  69,15,194,249,0                     // cmpeqps       %xmm9,%xmm15
   .byte  65,15,92,208                        // subps         %xmm8,%xmm2
   .byte  65,15,89,213                        // mulps         %xmm13,%xmm2
-  .byte  68,15,40,37,71,50,0,0               // movaps        0x3247(%rip),%xmm12        # 4bd0 <_sk_callback_sse2+0x484>
+  .byte  68,15,40,37,103,50,0,0              // movaps        0x3267(%rip),%xmm12        # 4bf0 <_sk_callback_sse2+0x47f>
   .byte  65,15,88,212                        // addps         %xmm12,%xmm2
   .byte  69,15,92,193                        // subps         %xmm9,%xmm8
   .byte  69,15,89,197                        // mulps         %xmm13,%xmm8
-  .byte  68,15,88,5,67,50,0,0                // addps         0x3243(%rip),%xmm8        # 4be0 <_sk_callback_sse2+0x494>
+  .byte  68,15,88,5,99,50,0,0                // addps         0x3263(%rip),%xmm8        # 4c00 <_sk_callback_sse2+0x48f>
   .byte  65,15,84,215                        // andps         %xmm15,%xmm2
   .byte  69,15,85,248                        // andnps        %xmm8,%xmm15
   .byte  68,15,86,250                        // orps          %xmm2,%xmm15
   .byte  68,15,84,240                        // andps         %xmm0,%xmm14
   .byte  65,15,85,199                        // andnps        %xmm15,%xmm0
   .byte  65,15,86,198                        // orps          %xmm14,%xmm0
-  .byte  15,89,5,52,50,0,0                   // mulps         0x3234(%rip),%xmm0        # 4bf0 <_sk_callback_sse2+0x4a4>
+  .byte  15,89,5,84,50,0,0                   // mulps         0x3254(%rip),%xmm0        # 4c10 <_sk_callback_sse2+0x49f>
   .byte  69,15,40,194                        // movaps        %xmm10,%xmm8
   .byte  69,15,194,195,4                     // cmpneqps      %xmm11,%xmm8
   .byte  65,15,84,192                        // andps         %xmm8,%xmm0
   .byte  69,15,92,226                        // subps         %xmm10,%xmm12
   .byte  69,15,88,211                        // addps         %xmm11,%xmm10
-  .byte  68,15,40,13,39,50,0,0               // movaps        0x3227(%rip),%xmm9        # 4c00 <_sk_callback_sse2+0x4b4>
+  .byte  68,15,40,13,71,50,0,0               // movaps        0x3247(%rip),%xmm9        # 4c20 <_sk_callback_sse2+0x4af>
   .byte  65,15,40,210                        // movaps        %xmm10,%xmm2
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
   .byte  68,15,194,202,1                     // cmpltps       %xmm2,%xmm9
@@ -27838,7 +27877,7 @@ _sk_hsl_to_rgb_sse2:
   .byte  15,41,92,36,168                     // movaps        %xmm3,-0x58(%rsp)
   .byte  68,15,40,218                        // movaps        %xmm2,%xmm11
   .byte  15,40,240                           // movaps        %xmm0,%xmm6
-  .byte  68,15,40,13,230,49,0,0              // movaps        0x31e6(%rip),%xmm9        # 4c10 <_sk_callback_sse2+0x4c4>
+  .byte  68,15,40,13,6,50,0,0                // movaps        0x3206(%rip),%xmm9        # 4c30 <_sk_callback_sse2+0x4bf>
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
   .byte  69,15,194,211,2                     // cmpleps       %xmm11,%xmm10
   .byte  15,40,193                           // movaps        %xmm1,%xmm0
@@ -27855,28 +27894,28 @@ _sk_hsl_to_rgb_sse2:
   .byte  69,15,88,211                        // addps         %xmm11,%xmm10
   .byte  69,15,88,219                        // addps         %xmm11,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  15,40,5,175,49,0,0                  // movaps        0x31af(%rip),%xmm0        # 4c20 <_sk_callback_sse2+0x4d4>
+  .byte  15,40,5,207,49,0,0                  // movaps        0x31cf(%rip),%xmm0        # 4c40 <_sk_callback_sse2+0x4cf>
   .byte  15,88,198                           // addps         %xmm6,%xmm0
   .byte  243,15,91,200                       // cvttps2dq     %xmm0,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  15,40,216                           // movaps        %xmm0,%xmm3
   .byte  15,194,217,1                        // cmpltps       %xmm1,%xmm3
-  .byte  15,84,29,167,49,0,0                 // andps         0x31a7(%rip),%xmm3        # 4c30 <_sk_callback_sse2+0x4e4>
+  .byte  15,84,29,199,49,0,0                 // andps         0x31c7(%rip),%xmm3        # 4c50 <_sk_callback_sse2+0x4df>
   .byte  15,92,203                           // subps         %xmm3,%xmm1
   .byte  15,92,193                           // subps         %xmm1,%xmm0
-  .byte  68,15,40,45,169,49,0,0              // movaps        0x31a9(%rip),%xmm13        # 4c40 <_sk_callback_sse2+0x4f4>
+  .byte  68,15,40,45,201,49,0,0              // movaps        0x31c9(%rip),%xmm13        # 4c60 <_sk_callback_sse2+0x4ef>
   .byte  69,15,40,197                        // movaps        %xmm13,%xmm8
   .byte  68,15,194,192,2                     // cmpleps       %xmm0,%xmm8
   .byte  69,15,40,242                        // movaps        %xmm10,%xmm14
   .byte  69,15,92,243                        // subps         %xmm11,%xmm14
   .byte  65,15,40,217                        // movaps        %xmm9,%xmm3
   .byte  15,194,216,2                        // cmpleps       %xmm0,%xmm3
-  .byte  15,40,21,185,49,0,0                 // movaps        0x31b9(%rip),%xmm2        # 4c70 <_sk_callback_sse2+0x524>
+  .byte  15,40,21,217,49,0,0                 // movaps        0x31d9(%rip),%xmm2        # 4c90 <_sk_callback_sse2+0x51f>
   .byte  68,15,40,250                        // movaps        %xmm2,%xmm15
   .byte  68,15,194,248,2                     // cmpleps       %xmm0,%xmm15
-  .byte  15,40,13,137,49,0,0                 // movaps        0x3189(%rip),%xmm1        # 4c50 <_sk_callback_sse2+0x504>
+  .byte  15,40,13,169,49,0,0                 // movaps        0x31a9(%rip),%xmm1        # 4c70 <_sk_callback_sse2+0x4ff>
   .byte  15,89,193                           // mulps         %xmm1,%xmm0
-  .byte  15,40,45,143,49,0,0                 // movaps        0x318f(%rip),%xmm5        # 4c60 <_sk_callback_sse2+0x514>
+  .byte  15,40,45,175,49,0,0                 // movaps        0x31af(%rip),%xmm5        # 4c80 <_sk_callback_sse2+0x50f>
   .byte  15,40,229                           // movaps        %xmm5,%xmm4
   .byte  15,92,224                           // subps         %xmm0,%xmm4
   .byte  65,15,89,230                        // mulps         %xmm14,%xmm4
@@ -27899,7 +27938,7 @@ _sk_hsl_to_rgb_sse2:
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
   .byte  15,40,222                           // movaps        %xmm6,%xmm3
   .byte  15,194,216,1                        // cmpltps       %xmm0,%xmm3
-  .byte  15,84,29,4,49,0,0                   // andps         0x3104(%rip),%xmm3        # 4c30 <_sk_callback_sse2+0x4e4>
+  .byte  15,84,29,36,49,0,0                  // andps         0x3124(%rip),%xmm3        # 4c50 <_sk_callback_sse2+0x4df>
   .byte  15,92,195                           // subps         %xmm3,%xmm0
   .byte  68,15,40,230                        // movaps        %xmm6,%xmm12
   .byte  68,15,92,224                        // subps         %xmm0,%xmm12
@@ -27929,12 +27968,12 @@ _sk_hsl_to_rgb_sse2:
   .byte  15,40,124,36,136                    // movaps        -0x78(%rsp),%xmm7
   .byte  15,40,231                           // movaps        %xmm7,%xmm4
   .byte  15,85,227                           // andnps        %xmm3,%xmm4
-  .byte  15,88,53,220,48,0,0                 // addps         0x30dc(%rip),%xmm6        # 4c80 <_sk_callback_sse2+0x534>
+  .byte  15,88,53,252,48,0,0                 // addps         0x30fc(%rip),%xmm6        # 4ca0 <_sk_callback_sse2+0x52f>
   .byte  243,15,91,198                       // cvttps2dq     %xmm6,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
   .byte  15,40,222                           // movaps        %xmm6,%xmm3
   .byte  15,194,216,1                        // cmpltps       %xmm0,%xmm3
-  .byte  15,84,29,119,48,0,0                 // andps         0x3077(%rip),%xmm3        # 4c30 <_sk_callback_sse2+0x4e4>
+  .byte  15,84,29,151,48,0,0                 // andps         0x3097(%rip),%xmm3        # 4c50 <_sk_callback_sse2+0x4df>
   .byte  15,92,195                           // subps         %xmm3,%xmm0
   .byte  15,92,240                           // subps         %xmm0,%xmm6
   .byte  15,89,206                           // mulps         %xmm6,%xmm1
@@ -27998,7 +28037,7 @@ _sk_scale_u8_sse2:
   .byte  102,69,15,96,193                    // punpcklbw     %xmm9,%xmm8
   .byte  102,69,15,97,193                    // punpcklwd     %xmm9,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,5,48,0,0                 // mulps         0x3005(%rip),%xmm8        # 4c90 <_sk_callback_sse2+0x544>
+  .byte  68,15,89,5,37,48,0,0                // mulps         0x3025(%rip),%xmm8        # 4cb0 <_sk_callback_sse2+0x53f>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
@@ -28039,7 +28078,7 @@ _sk_lerp_u8_sse2:
   .byte  102,69,15,96,193                    // punpcklbw     %xmm9,%xmm8
   .byte  102,69,15,97,193                    // punpcklwd     %xmm9,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,163,47,0,0               // mulps         0x2fa3(%rip),%xmm8        # 4ca0 <_sk_callback_sse2+0x554>
+  .byte  68,15,89,5,195,47,0,0               // mulps         0x2fc3(%rip),%xmm8        # 4cc0 <_sk_callback_sse2+0x54f>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -28061,31 +28100,40 @@ FUNCTION(_sk_lerp_565_sse2)
 _sk_lerp_565_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  243,68,15,126,4,120                 // movq          (%rax,%rdi,2),%xmm8
-  .byte  102,15,239,219                      // pxor          %xmm3,%xmm3
-  .byte  102,68,15,97,195                    // punpcklwd     %xmm3,%xmm8
-  .byte  102,15,111,29,107,47,0,0            // movdqa        0x2f6b(%rip),%xmm3        # 4cb0 <_sk_callback_sse2+0x564>
-  .byte  102,65,15,219,216                   // pand          %xmm8,%xmm3
-  .byte  68,15,91,203                        // cvtdq2ps      %xmm3,%xmm9
-  .byte  68,15,89,13,106,47,0,0              // mulps         0x2f6a(%rip),%xmm9        # 4cc0 <_sk_callback_sse2+0x574>
-  .byte  102,15,111,29,114,47,0,0            // movdqa        0x2f72(%rip),%xmm3        # 4cd0 <_sk_callback_sse2+0x584>
-  .byte  102,65,15,219,216                   // pand          %xmm8,%xmm3
-  .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,115,47,0,0                 // mulps         0x2f73(%rip),%xmm3        # 4ce0 <_sk_callback_sse2+0x594>
-  .byte  102,68,15,219,5,122,47,0,0          // pand          0x2f7a(%rip),%xmm8        # 4cf0 <_sk_callback_sse2+0x5a4>
+  .byte  243,68,15,126,20,120                // movq          (%rax,%rdi,2),%xmm10
+  .byte  102,69,15,239,192                   // pxor          %xmm8,%xmm8
+  .byte  102,69,15,97,208                    // punpcklwd     %xmm8,%xmm10
+  .byte  102,68,15,111,5,137,47,0,0          // movdqa        0x2f89(%rip),%xmm8        # 4cd0 <_sk_callback_sse2+0x55f>
+  .byte  102,69,15,219,194                   // pand          %xmm10,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,126,47,0,0               // mulps         0x2f7e(%rip),%xmm8        # 4d00 <_sk_callback_sse2+0x5b4>
+  .byte  68,15,89,5,136,47,0,0               // mulps         0x2f88(%rip),%xmm8        # 4ce0 <_sk_callback_sse2+0x56f>
+  .byte  102,68,15,111,13,143,47,0,0         // movdqa        0x2f8f(%rip),%xmm9        # 4cf0 <_sk_callback_sse2+0x57f>
+  .byte  102,69,15,219,202                   // pand          %xmm10,%xmm9
+  .byte  69,15,91,201                        // cvtdq2ps      %xmm9,%xmm9
+  .byte  68,15,89,13,142,47,0,0              // mulps         0x2f8e(%rip),%xmm9        # 4d00 <_sk_callback_sse2+0x58f>
+  .byte  102,68,15,219,21,149,47,0,0         // pand          0x2f95(%rip),%xmm10        # 4d10 <_sk_callback_sse2+0x59f>
+  .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
+  .byte  68,15,89,21,153,47,0,0              // mulps         0x2f99(%rip),%xmm10        # 4d20 <_sk_callback_sse2+0x5af>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
-  .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
+  .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
   .byte  15,92,205                           // subps         %xmm5,%xmm1
-  .byte  15,89,203                           // mulps         %xmm3,%xmm1
+  .byte  65,15,89,201                        // mulps         %xmm9,%xmm1
   .byte  15,88,205                           // addps         %xmm5,%xmm1
   .byte  15,92,214                           // subps         %xmm6,%xmm2
-  .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
+  .byte  65,15,89,210                        // mulps         %xmm10,%xmm2
   .byte  15,88,214                           // addps         %xmm6,%xmm2
+  .byte  15,92,223                           // subps         %xmm7,%xmm3
+  .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
+  .byte  68,15,88,199                        // addps         %xmm7,%xmm8
+  .byte  68,15,89,203                        // mulps         %xmm3,%xmm9
+  .byte  68,15,88,207                        // addps         %xmm7,%xmm9
+  .byte  65,15,89,218                        // mulps         %xmm10,%xmm3
+  .byte  15,88,223                           // addps         %xmm7,%xmm3
+  .byte  68,15,95,203                        // maxps         %xmm3,%xmm9
+  .byte  69,15,95,193                        // maxps         %xmm9,%xmm8
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,104,47,0,0                 // movaps        0x2f68(%rip),%xmm3        # 4d10 <_sk_callback_sse2+0x5c4>
+  .byte  65,15,40,216                        // movaps        %xmm8,%xmm3
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_load_tables_sse2
@@ -28096,7 +28144,7 @@ _sk_load_tables_sse2:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,139,72,8                         // mov           0x8(%rax),%r9
   .byte  243,69,15,111,12,184                // movdqu        (%r8,%rdi,4),%xmm9
-  .byte  102,68,15,111,5,94,47,0,0           // movdqa        0x2f5e(%rip),%xmm8        # 4d20 <_sk_callback_sse2+0x5d4>
+  .byte  102,68,15,111,5,73,47,0,0           // movdqa        0x2f49(%rip),%xmm8        # 4d30 <_sk_callback_sse2+0x5bf>
   .byte  102,65,15,111,193                   // movdqa        %xmm9,%xmm0
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,15,112,200,78                   // pshufd        $0x4e,%xmm0,%xmm1
@@ -28151,7 +28199,7 @@ _sk_load_tables_sse2:
   .byte  65,15,20,208                        // unpcklps      %xmm8,%xmm2
   .byte  102,65,15,114,209,24                // psrld         $0x18,%xmm9
   .byte  65,15,91,217                        // cvtdq2ps      %xmm9,%xmm3
-  .byte  15,89,29,107,46,0,0                 // mulps         0x2e6b(%rip),%xmm3        # 4d30 <_sk_callback_sse2+0x5e4>
+  .byte  15,89,29,86,46,0,0                  // mulps         0x2e56(%rip),%xmm3        # 4d40 <_sk_callback_sse2+0x5cf>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -28170,7 +28218,7 @@ _sk_load_tables_u16_be_sse2:
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,97,200                       // punpcklwd     %xmm0,%xmm1
   .byte  102,68,15,105,200                   // punpckhwd     %xmm0,%xmm9
-  .byte  102,68,15,111,21,62,46,0,0          // movdqa        0x2e3e(%rip),%xmm10        # 4d40 <_sk_callback_sse2+0x5f4>
+  .byte  102,68,15,111,21,41,46,0,0          // movdqa        0x2e29(%rip),%xmm10        # 4d50 <_sk_callback_sse2+0x5df>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,194                   // pand          %xmm10,%xmm0
   .byte  102,69,15,239,192                   // pxor          %xmm8,%xmm8
@@ -28231,7 +28279,7 @@ _sk_load_tables_u16_be_sse2:
   .byte  102,65,15,235,217                   // por           %xmm9,%xmm3
   .byte  102,65,15,97,216                    // punpcklwd     %xmm8,%xmm3
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,45,45,0,0                  // mulps         0x2d2d(%rip),%xmm3        # 4d50 <_sk_callback_sse2+0x604>
+  .byte  15,89,29,24,45,0,0                  // mulps         0x2d18(%rip),%xmm3        # 4d60 <_sk_callback_sse2+0x5ef>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -28253,7 +28301,7 @@ _sk_load_tables_rgb_u16_be_sse2:
   .byte  102,68,15,97,208                    // punpcklwd     %xmm0,%xmm10
   .byte  102,65,15,111,195                   // movdqa        %xmm11,%xmm0
   .byte  102,65,15,97,194                    // punpcklwd     %xmm10,%xmm0
-  .byte  102,68,15,111,5,237,44,0,0          // movdqa        0x2ced(%rip),%xmm8        # 4d60 <_sk_callback_sse2+0x614>
+  .byte  102,68,15,111,5,216,44,0,0          // movdqa        0x2cd8(%rip),%xmm8        # 4d70 <_sk_callback_sse2+0x5ff>
   .byte  102,15,112,200,78                   // pshufd        $0x4e,%xmm0,%xmm1
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,69,15,239,201                   // pxor          %xmm9,%xmm9
@@ -28308,7 +28356,7 @@ _sk_load_tables_rgb_u16_be_sse2:
   .byte  15,20,211                           // unpcklps      %xmm3,%xmm2
   .byte  65,15,20,208                        // unpcklps      %xmm8,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,252,43,0,0                 // movaps        0x2bfc(%rip),%xmm3        # 4d70 <_sk_callback_sse2+0x624>
+  .byte  15,40,29,231,43,0,0                 // movaps        0x2be7(%rip),%xmm3        # 4d80 <_sk_callback_sse2+0x60f>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_byte_tables_sse2
@@ -28318,7 +28366,7 @@ _sk_byte_tables_sse2:
   .byte  65,86                               // push          %r14
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,253,43,0,0               // movaps        0x2bfd(%rip),%xmm8        # 4d80 <_sk_callback_sse2+0x634>
+  .byte  68,15,40,5,232,43,0,0               // movaps        0x2be8(%rip),%xmm8        # 4d90 <_sk_callback_sse2+0x61f>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,91,192                       // cvtps2dq      %xmm0,%xmm0
   .byte  102,72,15,126,193                   // movq          %xmm0,%rcx
@@ -28345,7 +28393,7 @@ _sk_byte_tables_sse2:
   .byte  102,65,15,96,193                    // punpcklbw     %xmm9,%xmm0
   .byte  102,65,15,97,193                    // punpcklwd     %xmm9,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,21,154,43,0,0              // movaps        0x2b9a(%rip),%xmm10        # 4d90 <_sk_callback_sse2+0x644>
+  .byte  68,15,40,21,133,43,0,0              // movaps        0x2b85(%rip),%xmm10        # 4da0 <_sk_callback_sse2+0x62f>
   .byte  65,15,89,194                        // mulps         %xmm10,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -28461,7 +28509,7 @@ _sk_byte_tables_rgb_sse2:
   .byte  102,65,15,96,193                    // punpcklbw     %xmm9,%xmm0
   .byte  102,65,15,97,193                    // punpcklwd     %xmm9,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,21,237,41,0,0              // movaps        0x29ed(%rip),%xmm10        # 4da0 <_sk_callback_sse2+0x654>
+  .byte  68,15,40,21,216,41,0,0              // movaps        0x29d8(%rip),%xmm10        # 4db0 <_sk_callback_sse2+0x63f>
   .byte  65,15,89,194                        // mulps         %xmm10,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -28658,15 +28706,15 @@ _sk_parametric_r_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,44,39,0,0               // mulps         0x272c(%rip),%xmm9        # 4db0 <_sk_callback_sse2+0x664>
-  .byte  68,15,84,21,52,39,0,0               // andps         0x2734(%rip),%xmm10        # 4dc0 <_sk_callback_sse2+0x674>
-  .byte  68,15,86,21,60,39,0,0               // orps          0x273c(%rip),%xmm10        # 4dd0 <_sk_callback_sse2+0x684>
-  .byte  68,15,88,13,68,39,0,0               // addps         0x2744(%rip),%xmm9        # 4de0 <_sk_callback_sse2+0x694>
-  .byte  68,15,40,37,76,39,0,0               // movaps        0x274c(%rip),%xmm12        # 4df0 <_sk_callback_sse2+0x6a4>
+  .byte  68,15,89,13,23,39,0,0               // mulps         0x2717(%rip),%xmm9        # 4dc0 <_sk_callback_sse2+0x64f>
+  .byte  68,15,84,21,31,39,0,0               // andps         0x271f(%rip),%xmm10        # 4dd0 <_sk_callback_sse2+0x65f>
+  .byte  68,15,86,21,39,39,0,0               // orps          0x2727(%rip),%xmm10        # 4de0 <_sk_callback_sse2+0x66f>
+  .byte  68,15,88,13,47,39,0,0               // addps         0x272f(%rip),%xmm9        # 4df0 <_sk_callback_sse2+0x67f>
+  .byte  68,15,40,37,55,39,0,0               // movaps        0x2737(%rip),%xmm12        # 4e00 <_sk_callback_sse2+0x68f>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,76,39,0,0               // addps         0x274c(%rip),%xmm10        # 4e00 <_sk_callback_sse2+0x6b4>
-  .byte  68,15,40,37,84,39,0,0               // movaps        0x2754(%rip),%xmm12        # 4e10 <_sk_callback_sse2+0x6c4>
+  .byte  68,15,88,21,55,39,0,0               // addps         0x2737(%rip),%xmm10        # 4e10 <_sk_callback_sse2+0x69f>
+  .byte  68,15,40,37,63,39,0,0               // movaps        0x273f(%rip),%xmm12        # 4e20 <_sk_callback_sse2+0x6af>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -28674,22 +28722,22 @@ _sk_parametric_r_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,62,39,0,0               // movaps        0x273e(%rip),%xmm10        # 4e20 <_sk_callback_sse2+0x6d4>
+  .byte  68,15,40,21,41,39,0,0               // movaps        0x2729(%rip),%xmm10        # 4e30 <_sk_callback_sse2+0x6bf>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,50,39,0,0               // addps         0x2732(%rip),%xmm9        # 4e30 <_sk_callback_sse2+0x6e4>
-  .byte  68,15,40,37,58,39,0,0               // movaps        0x273a(%rip),%xmm12        # 4e40 <_sk_callback_sse2+0x6f4>
+  .byte  68,15,88,13,29,39,0,0               // addps         0x271d(%rip),%xmm9        # 4e40 <_sk_callback_sse2+0x6cf>
+  .byte  68,15,40,37,37,39,0,0               // movaps        0x2725(%rip),%xmm12        # 4e50 <_sk_callback_sse2+0x6df>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,58,39,0,0               // movaps        0x273a(%rip),%xmm12        # 4e50 <_sk_callback_sse2+0x704>
+  .byte  68,15,40,37,37,39,0,0               // movaps        0x2725(%rip),%xmm12        # 4e60 <_sk_callback_sse2+0x6ef>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,62,39,0,0               // movaps        0x273e(%rip),%xmm13        # 4e60 <_sk_callback_sse2+0x714>
+  .byte  68,15,40,45,41,39,0,0               // movaps        0x2729(%rip),%xmm13        # 4e70 <_sk_callback_sse2+0x6ff>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,62,39,0,0               // mulps         0x273e(%rip),%xmm13        # 4e70 <_sk_callback_sse2+0x724>
+  .byte  68,15,89,45,41,39,0,0               // mulps         0x2729(%rip),%xmm13        # 4e80 <_sk_callback_sse2+0x70f>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -28725,15 +28773,15 @@ _sk_parametric_g_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,190,38,0,0              // mulps         0x26be(%rip),%xmm9        # 4e80 <_sk_callback_sse2+0x734>
-  .byte  68,15,84,21,198,38,0,0              // andps         0x26c6(%rip),%xmm10        # 4e90 <_sk_callback_sse2+0x744>
-  .byte  68,15,86,21,206,38,0,0              // orps          0x26ce(%rip),%xmm10        # 4ea0 <_sk_callback_sse2+0x754>
-  .byte  68,15,88,13,214,38,0,0              // addps         0x26d6(%rip),%xmm9        # 4eb0 <_sk_callback_sse2+0x764>
-  .byte  68,15,40,37,222,38,0,0              // movaps        0x26de(%rip),%xmm12        # 4ec0 <_sk_callback_sse2+0x774>
+  .byte  68,15,89,13,169,38,0,0              // mulps         0x26a9(%rip),%xmm9        # 4e90 <_sk_callback_sse2+0x71f>
+  .byte  68,15,84,21,177,38,0,0              // andps         0x26b1(%rip),%xmm10        # 4ea0 <_sk_callback_sse2+0x72f>
+  .byte  68,15,86,21,185,38,0,0              // orps          0x26b9(%rip),%xmm10        # 4eb0 <_sk_callback_sse2+0x73f>
+  .byte  68,15,88,13,193,38,0,0              // addps         0x26c1(%rip),%xmm9        # 4ec0 <_sk_callback_sse2+0x74f>
+  .byte  68,15,40,37,201,38,0,0              // movaps        0x26c9(%rip),%xmm12        # 4ed0 <_sk_callback_sse2+0x75f>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,222,38,0,0              // addps         0x26de(%rip),%xmm10        # 4ed0 <_sk_callback_sse2+0x784>
-  .byte  68,15,40,37,230,38,0,0              // movaps        0x26e6(%rip),%xmm12        # 4ee0 <_sk_callback_sse2+0x794>
+  .byte  68,15,88,21,201,38,0,0              // addps         0x26c9(%rip),%xmm10        # 4ee0 <_sk_callback_sse2+0x76f>
+  .byte  68,15,40,37,209,38,0,0              // movaps        0x26d1(%rip),%xmm12        # 4ef0 <_sk_callback_sse2+0x77f>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -28741,22 +28789,22 @@ _sk_parametric_g_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,208,38,0,0              // movaps        0x26d0(%rip),%xmm10        # 4ef0 <_sk_callback_sse2+0x7a4>
+  .byte  68,15,40,21,187,38,0,0              // movaps        0x26bb(%rip),%xmm10        # 4f00 <_sk_callback_sse2+0x78f>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,196,38,0,0              // addps         0x26c4(%rip),%xmm9        # 4f00 <_sk_callback_sse2+0x7b4>
-  .byte  68,15,40,37,204,38,0,0              // movaps        0x26cc(%rip),%xmm12        # 4f10 <_sk_callback_sse2+0x7c4>
+  .byte  68,15,88,13,175,38,0,0              // addps         0x26af(%rip),%xmm9        # 4f10 <_sk_callback_sse2+0x79f>
+  .byte  68,15,40,37,183,38,0,0              // movaps        0x26b7(%rip),%xmm12        # 4f20 <_sk_callback_sse2+0x7af>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,204,38,0,0              // movaps        0x26cc(%rip),%xmm12        # 4f20 <_sk_callback_sse2+0x7d4>
+  .byte  68,15,40,37,183,38,0,0              // movaps        0x26b7(%rip),%xmm12        # 4f30 <_sk_callback_sse2+0x7bf>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,208,38,0,0              // movaps        0x26d0(%rip),%xmm13        # 4f30 <_sk_callback_sse2+0x7e4>
+  .byte  68,15,40,45,187,38,0,0              // movaps        0x26bb(%rip),%xmm13        # 4f40 <_sk_callback_sse2+0x7cf>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,208,38,0,0              // mulps         0x26d0(%rip),%xmm13        # 4f40 <_sk_callback_sse2+0x7f4>
+  .byte  68,15,89,45,187,38,0,0              // mulps         0x26bb(%rip),%xmm13        # 4f50 <_sk_callback_sse2+0x7df>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -28792,15 +28840,15 @@ _sk_parametric_b_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,80,38,0,0               // mulps         0x2650(%rip),%xmm9        # 4f50 <_sk_callback_sse2+0x804>
-  .byte  68,15,84,21,88,38,0,0               // andps         0x2658(%rip),%xmm10        # 4f60 <_sk_callback_sse2+0x814>
-  .byte  68,15,86,21,96,38,0,0               // orps          0x2660(%rip),%xmm10        # 4f70 <_sk_callback_sse2+0x824>
-  .byte  68,15,88,13,104,38,0,0              // addps         0x2668(%rip),%xmm9        # 4f80 <_sk_callback_sse2+0x834>
-  .byte  68,15,40,37,112,38,0,0              // movaps        0x2670(%rip),%xmm12        # 4f90 <_sk_callback_sse2+0x844>
+  .byte  68,15,89,13,59,38,0,0               // mulps         0x263b(%rip),%xmm9        # 4f60 <_sk_callback_sse2+0x7ef>
+  .byte  68,15,84,21,67,38,0,0               // andps         0x2643(%rip),%xmm10        # 4f70 <_sk_callback_sse2+0x7ff>
+  .byte  68,15,86,21,75,38,0,0               // orps          0x264b(%rip),%xmm10        # 4f80 <_sk_callback_sse2+0x80f>
+  .byte  68,15,88,13,83,38,0,0               // addps         0x2653(%rip),%xmm9        # 4f90 <_sk_callback_sse2+0x81f>
+  .byte  68,15,40,37,91,38,0,0               // movaps        0x265b(%rip),%xmm12        # 4fa0 <_sk_callback_sse2+0x82f>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,112,38,0,0              // addps         0x2670(%rip),%xmm10        # 4fa0 <_sk_callback_sse2+0x854>
-  .byte  68,15,40,37,120,38,0,0              // movaps        0x2678(%rip),%xmm12        # 4fb0 <_sk_callback_sse2+0x864>
+  .byte  68,15,88,21,91,38,0,0               // addps         0x265b(%rip),%xmm10        # 4fb0 <_sk_callback_sse2+0x83f>
+  .byte  68,15,40,37,99,38,0,0               // movaps        0x2663(%rip),%xmm12        # 4fc0 <_sk_callback_sse2+0x84f>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -28808,22 +28856,22 @@ _sk_parametric_b_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,98,38,0,0               // movaps        0x2662(%rip),%xmm10        # 4fc0 <_sk_callback_sse2+0x874>
+  .byte  68,15,40,21,77,38,0,0               // movaps        0x264d(%rip),%xmm10        # 4fd0 <_sk_callback_sse2+0x85f>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,86,38,0,0               // addps         0x2656(%rip),%xmm9        # 4fd0 <_sk_callback_sse2+0x884>
-  .byte  68,15,40,37,94,38,0,0               // movaps        0x265e(%rip),%xmm12        # 4fe0 <_sk_callback_sse2+0x894>
+  .byte  68,15,88,13,65,38,0,0               // addps         0x2641(%rip),%xmm9        # 4fe0 <_sk_callback_sse2+0x86f>
+  .byte  68,15,40,37,73,38,0,0               // movaps        0x2649(%rip),%xmm12        # 4ff0 <_sk_callback_sse2+0x87f>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,94,38,0,0               // movaps        0x265e(%rip),%xmm12        # 4ff0 <_sk_callback_sse2+0x8a4>
+  .byte  68,15,40,37,73,38,0,0               // movaps        0x2649(%rip),%xmm12        # 5000 <_sk_callback_sse2+0x88f>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,98,38,0,0               // movaps        0x2662(%rip),%xmm13        # 5000 <_sk_callback_sse2+0x8b4>
+  .byte  68,15,40,45,77,38,0,0               // movaps        0x264d(%rip),%xmm13        # 5010 <_sk_callback_sse2+0x89f>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,98,38,0,0               // mulps         0x2662(%rip),%xmm13        # 5010 <_sk_callback_sse2+0x8c4>
+  .byte  68,15,89,45,77,38,0,0               // mulps         0x264d(%rip),%xmm13        # 5020 <_sk_callback_sse2+0x8af>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -28859,15 +28907,15 @@ _sk_parametric_a_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,226,37,0,0              // mulps         0x25e2(%rip),%xmm9        # 5020 <_sk_callback_sse2+0x8d4>
-  .byte  68,15,84,21,234,37,0,0              // andps         0x25ea(%rip),%xmm10        # 5030 <_sk_callback_sse2+0x8e4>
-  .byte  68,15,86,21,242,37,0,0              // orps          0x25f2(%rip),%xmm10        # 5040 <_sk_callback_sse2+0x8f4>
-  .byte  68,15,88,13,250,37,0,0              // addps         0x25fa(%rip),%xmm9        # 5050 <_sk_callback_sse2+0x904>
-  .byte  68,15,40,37,2,38,0,0                // movaps        0x2602(%rip),%xmm12        # 5060 <_sk_callback_sse2+0x914>
+  .byte  68,15,89,13,205,37,0,0              // mulps         0x25cd(%rip),%xmm9        # 5030 <_sk_callback_sse2+0x8bf>
+  .byte  68,15,84,21,213,37,0,0              // andps         0x25d5(%rip),%xmm10        # 5040 <_sk_callback_sse2+0x8cf>
+  .byte  68,15,86,21,221,37,0,0              // orps          0x25dd(%rip),%xmm10        # 5050 <_sk_callback_sse2+0x8df>
+  .byte  68,15,88,13,229,37,0,0              // addps         0x25e5(%rip),%xmm9        # 5060 <_sk_callback_sse2+0x8ef>
+  .byte  68,15,40,37,237,37,0,0              // movaps        0x25ed(%rip),%xmm12        # 5070 <_sk_callback_sse2+0x8ff>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,2,38,0,0                // addps         0x2602(%rip),%xmm10        # 5070 <_sk_callback_sse2+0x924>
-  .byte  68,15,40,37,10,38,0,0               // movaps        0x260a(%rip),%xmm12        # 5080 <_sk_callback_sse2+0x934>
+  .byte  68,15,88,21,237,37,0,0              // addps         0x25ed(%rip),%xmm10        # 5080 <_sk_callback_sse2+0x90f>
+  .byte  68,15,40,37,245,37,0,0              // movaps        0x25f5(%rip),%xmm12        # 5090 <_sk_callback_sse2+0x91f>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -28875,22 +28923,22 @@ _sk_parametric_a_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,244,37,0,0              // movaps        0x25f4(%rip),%xmm10        # 5090 <_sk_callback_sse2+0x944>
+  .byte  68,15,40,21,223,37,0,0              // movaps        0x25df(%rip),%xmm10        # 50a0 <_sk_callback_sse2+0x92f>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,232,37,0,0              // addps         0x25e8(%rip),%xmm9        # 50a0 <_sk_callback_sse2+0x954>
-  .byte  68,15,40,37,240,37,0,0              // movaps        0x25f0(%rip),%xmm12        # 50b0 <_sk_callback_sse2+0x964>
+  .byte  68,15,88,13,211,37,0,0              // addps         0x25d3(%rip),%xmm9        # 50b0 <_sk_callback_sse2+0x93f>
+  .byte  68,15,40,37,219,37,0,0              // movaps        0x25db(%rip),%xmm12        # 50c0 <_sk_callback_sse2+0x94f>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,240,37,0,0              // movaps        0x25f0(%rip),%xmm12        # 50c0 <_sk_callback_sse2+0x974>
+  .byte  68,15,40,37,219,37,0,0              // movaps        0x25db(%rip),%xmm12        # 50d0 <_sk_callback_sse2+0x95f>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,244,37,0,0              // movaps        0x25f4(%rip),%xmm13        # 50d0 <_sk_callback_sse2+0x984>
+  .byte  68,15,40,45,223,37,0,0              // movaps        0x25df(%rip),%xmm13        # 50e0 <_sk_callback_sse2+0x96f>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,244,37,0,0              // mulps         0x25f4(%rip),%xmm13        # 50e0 <_sk_callback_sse2+0x994>
+  .byte  68,15,89,45,223,37,0,0              // mulps         0x25df(%rip),%xmm13        # 50f0 <_sk_callback_sse2+0x97f>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -28907,29 +28955,29 @@ HIDDEN _sk_lab_to_xyz_sse2
 .globl _sk_lab_to_xyz_sse2
 FUNCTION(_sk_lab_to_xyz_sse2)
 _sk_lab_to_xyz_sse2:
-  .byte  15,89,5,209,37,0,0                  // mulps         0x25d1(%rip),%xmm0        # 50f0 <_sk_callback_sse2+0x9a4>
-  .byte  68,15,40,5,217,37,0,0               // movaps        0x25d9(%rip),%xmm8        # 5100 <_sk_callback_sse2+0x9b4>
+  .byte  15,89,5,188,37,0,0                  // mulps         0x25bc(%rip),%xmm0        # 5100 <_sk_callback_sse2+0x98f>
+  .byte  68,15,40,5,196,37,0,0               // movaps        0x25c4(%rip),%xmm8        # 5110 <_sk_callback_sse2+0x99f>
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
-  .byte  68,15,40,13,221,37,0,0              // movaps        0x25dd(%rip),%xmm9        # 5110 <_sk_callback_sse2+0x9c4>
+  .byte  68,15,40,13,200,37,0,0              // movaps        0x25c8(%rip),%xmm9        # 5120 <_sk_callback_sse2+0x9af>
   .byte  65,15,88,201                        // addps         %xmm9,%xmm1
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  65,15,88,209                        // addps         %xmm9,%xmm2
-  .byte  15,88,5,218,37,0,0                  // addps         0x25da(%rip),%xmm0        # 5120 <_sk_callback_sse2+0x9d4>
-  .byte  15,89,5,227,37,0,0                  // mulps         0x25e3(%rip),%xmm0        # 5130 <_sk_callback_sse2+0x9e4>
-  .byte  15,89,13,236,37,0,0                 // mulps         0x25ec(%rip),%xmm1        # 5140 <_sk_callback_sse2+0x9f4>
+  .byte  15,88,5,197,37,0,0                  // addps         0x25c5(%rip),%xmm0        # 5130 <_sk_callback_sse2+0x9bf>
+  .byte  15,89,5,206,37,0,0                  // mulps         0x25ce(%rip),%xmm0        # 5140 <_sk_callback_sse2+0x9cf>
+  .byte  15,89,13,215,37,0,0                 // mulps         0x25d7(%rip),%xmm1        # 5150 <_sk_callback_sse2+0x9df>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  15,89,21,242,37,0,0                 // mulps         0x25f2(%rip),%xmm2        # 5150 <_sk_callback_sse2+0xa04>
+  .byte  15,89,21,221,37,0,0                 // mulps         0x25dd(%rip),%xmm2        # 5160 <_sk_callback_sse2+0x9ef>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  68,15,92,202                        // subps         %xmm2,%xmm9
   .byte  68,15,40,225                        // movaps        %xmm1,%xmm12
   .byte  69,15,89,228                        // mulps         %xmm12,%xmm12
   .byte  68,15,89,225                        // mulps         %xmm1,%xmm12
-  .byte  15,40,21,231,37,0,0                 // movaps        0x25e7(%rip),%xmm2        # 5160 <_sk_callback_sse2+0xa14>
+  .byte  15,40,21,210,37,0,0                 // movaps        0x25d2(%rip),%xmm2        # 5170 <_sk_callback_sse2+0x9ff>
   .byte  68,15,40,194                        // movaps        %xmm2,%xmm8
   .byte  69,15,194,196,1                     // cmpltps       %xmm12,%xmm8
-  .byte  68,15,40,21,230,37,0,0              // movaps        0x25e6(%rip),%xmm10        # 5170 <_sk_callback_sse2+0xa24>
+  .byte  68,15,40,21,209,37,0,0              // movaps        0x25d1(%rip),%xmm10        # 5180 <_sk_callback_sse2+0xa0f>
   .byte  65,15,88,202                        // addps         %xmm10,%xmm1
-  .byte  68,15,40,29,234,37,0,0              // movaps        0x25ea(%rip),%xmm11        # 5180 <_sk_callback_sse2+0xa34>
+  .byte  68,15,40,29,213,37,0,0              // movaps        0x25d5(%rip),%xmm11        # 5190 <_sk_callback_sse2+0xa1f>
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
   .byte  69,15,84,224                        // andps         %xmm8,%xmm12
   .byte  68,15,85,193                        // andnps        %xmm1,%xmm8
@@ -28953,8 +29001,8 @@ _sk_lab_to_xyz_sse2:
   .byte  15,84,194                           // andps         %xmm2,%xmm0
   .byte  65,15,85,209                        // andnps        %xmm9,%xmm2
   .byte  15,86,208                           // orps          %xmm0,%xmm2
-  .byte  68,15,89,5,154,37,0,0               // mulps         0x259a(%rip),%xmm8        # 5190 <_sk_callback_sse2+0xa44>
-  .byte  15,89,21,163,37,0,0                 // mulps         0x25a3(%rip),%xmm2        # 51a0 <_sk_callback_sse2+0xa54>
+  .byte  68,15,89,5,133,37,0,0               // mulps         0x2585(%rip),%xmm8        # 51a0 <_sk_callback_sse2+0xa2f>
+  .byte  15,89,21,142,37,0,0                 // mulps         0x258e(%rip),%xmm2        # 51b0 <_sk_callback_sse2+0xa3f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -28970,7 +29018,7 @@ _sk_load_a8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,139,37,0,0                 // mulps         0x258b(%rip),%xmm3        # 51b0 <_sk_callback_sse2+0xa64>
+  .byte  15,89,29,118,37,0,0                 // mulps         0x2576(%rip),%xmm3        # 51c0 <_sk_callback_sse2+0xa4f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
@@ -29015,7 +29063,7 @@ _sk_gather_a8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,250,36,0,0                 // mulps         0x24fa(%rip),%xmm3        # 51c0 <_sk_callback_sse2+0xa74>
+  .byte  15,89,29,229,36,0,0                 // mulps         0x24e5(%rip),%xmm3        # 51d0 <_sk_callback_sse2+0xa5f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
@@ -29028,7 +29076,7 @@ FUNCTION(_sk_store_a8_sse2)
 _sk_store_a8_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,238,36,0,0               // movaps        0x24ee(%rip),%xmm8        # 51d0 <_sk_callback_sse2+0xa84>
+  .byte  68,15,40,5,217,36,0,0               // movaps        0x24d9(%rip),%xmm8        # 51e0 <_sk_callback_sse2+0xa6f>
   .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
   .byte  102,65,15,114,240,16                // pslld         $0x10,%xmm8
@@ -29050,9 +29098,9 @@ _sk_load_g8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,181,36,0,0                  // mulps         0x24b5(%rip),%xmm0        # 51e0 <_sk_callback_sse2+0xa94>
+  .byte  15,89,5,160,36,0,0                  // mulps         0x24a0(%rip),%xmm0        # 51f0 <_sk_callback_sse2+0xa7f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,188,36,0,0                 // movaps        0x24bc(%rip),%xmm3        # 51f0 <_sk_callback_sse2+0xaa4>
+  .byte  15,40,29,167,36,0,0                 // movaps        0x24a7(%rip),%xmm3        # 5200 <_sk_callback_sse2+0xa8f>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -29095,9 +29143,9 @@ _sk_gather_g8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,49,36,0,0                   // mulps         0x2431(%rip),%xmm0        # 5200 <_sk_callback_sse2+0xab4>
+  .byte  15,89,5,28,36,0,0                   // mulps         0x241c(%rip),%xmm0        # 5210 <_sk_callback_sse2+0xa9f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,56,36,0,0                  // movaps        0x2438(%rip),%xmm3        # 5210 <_sk_callback_sse2+0xac4>
+  .byte  15,40,29,35,36,0,0                  // movaps        0x2423(%rip),%xmm3        # 5220 <_sk_callback_sse2+0xaaf>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -29109,9 +29157,9 @@ _sk_gather_i8_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            2def <_sk_gather_i8_sse2+0xf>
+  .byte  116,5                               // je            2e14 <_sk_gather_i8_sse2+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           2df1 <_sk_gather_i8_sse2+0x11>
+  .byte  235,2                               // jmp           2e16 <_sk_gather_i8_sse2+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  243,15,91,201                       // cvttps2dq     %xmm1,%xmm1
@@ -29160,11 +29208,11 @@ _sk_gather_i8_sse2:
   .byte  102,67,15,110,12,136                // movd          (%r8,%r9,4),%xmm1
   .byte  102,68,15,98,201                    // punpckldq     %xmm1,%xmm9
   .byte  102,68,15,98,200                    // punpckldq     %xmm0,%xmm9
-  .byte  102,15,111,21,87,35,0,0             // movdqa        0x2357(%rip),%xmm2        # 5220 <_sk_callback_sse2+0xad4>
+  .byte  102,15,111,21,66,35,0,0             // movdqa        0x2342(%rip),%xmm2        # 5230 <_sk_callback_sse2+0xabf>
   .byte  102,65,15,111,193                   // movdqa        %xmm9,%xmm0
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,83,35,0,0                // movaps        0x2353(%rip),%xmm8        # 5230 <_sk_callback_sse2+0xae4>
+  .byte  68,15,40,5,62,35,0,0                // movaps        0x233e(%rip),%xmm8        # 5240 <_sk_callback_sse2+0xacf>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,114,209,8                    // psrld         $0x8,%xmm1
@@ -29191,19 +29239,19 @@ _sk_load_565_sse2:
   .byte  243,15,126,20,120                   // movq          (%rax,%rdi,2),%xmm2
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,208                       // punpcklwd     %xmm0,%xmm2
-  .byte  102,15,111,5,9,35,0,0               // movdqa        0x2309(%rip),%xmm0        # 5240 <_sk_callback_sse2+0xaf4>
+  .byte  102,15,111,5,244,34,0,0             // movdqa        0x22f4(%rip),%xmm0        # 5250 <_sk_callback_sse2+0xadf>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,11,35,0,0                   // mulps         0x230b(%rip),%xmm0        # 5250 <_sk_callback_sse2+0xb04>
-  .byte  102,15,111,13,19,35,0,0             // movdqa        0x2313(%rip),%xmm1        # 5260 <_sk_callback_sse2+0xb14>
+  .byte  15,89,5,246,34,0,0                  // mulps         0x22f6(%rip),%xmm0        # 5260 <_sk_callback_sse2+0xaef>
+  .byte  102,15,111,13,254,34,0,0            // movdqa        0x22fe(%rip),%xmm1        # 5270 <_sk_callback_sse2+0xaff>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,21,35,0,0                  // mulps         0x2315(%rip),%xmm1        # 5270 <_sk_callback_sse2+0xb24>
-  .byte  102,15,219,21,29,35,0,0             // pand          0x231d(%rip),%xmm2        # 5280 <_sk_callback_sse2+0xb34>
+  .byte  15,89,13,0,35,0,0                   // mulps         0x2300(%rip),%xmm1        # 5280 <_sk_callback_sse2+0xb0f>
+  .byte  102,15,219,21,8,35,0,0              // pand          0x2308(%rip),%xmm2        # 5290 <_sk_callback_sse2+0xb1f>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,35,35,0,0                  // mulps         0x2323(%rip),%xmm2        # 5290 <_sk_callback_sse2+0xb44>
+  .byte  15,89,21,14,35,0,0                  // mulps         0x230e(%rip),%xmm2        # 52a0 <_sk_callback_sse2+0xb2f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,42,35,0,0                  // movaps        0x232a(%rip),%xmm3        # 52a0 <_sk_callback_sse2+0xb54>
+  .byte  15,40,29,21,35,0,0                  // movaps        0x2315(%rip),%xmm3        # 52b0 <_sk_callback_sse2+0xb3f>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_gather_565_sse2
@@ -29238,19 +29286,19 @@ _sk_gather_565_sse2:
   .byte  102,15,196,208,3                    // pinsrw        $0x3,%eax,%xmm2
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,208                       // punpcklwd     %xmm0,%xmm2
-  .byte  102,15,111,5,179,34,0,0             // movdqa        0x22b3(%rip),%xmm0        # 52b0 <_sk_callback_sse2+0xb64>
+  .byte  102,15,111,5,158,34,0,0             // movdqa        0x229e(%rip),%xmm0        # 52c0 <_sk_callback_sse2+0xb4f>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,181,34,0,0                  // mulps         0x22b5(%rip),%xmm0        # 52c0 <_sk_callback_sse2+0xb74>
-  .byte  102,15,111,13,189,34,0,0            // movdqa        0x22bd(%rip),%xmm1        # 52d0 <_sk_callback_sse2+0xb84>
+  .byte  15,89,5,160,34,0,0                  // mulps         0x22a0(%rip),%xmm0        # 52d0 <_sk_callback_sse2+0xb5f>
+  .byte  102,15,111,13,168,34,0,0            // movdqa        0x22a8(%rip),%xmm1        # 52e0 <_sk_callback_sse2+0xb6f>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,191,34,0,0                 // mulps         0x22bf(%rip),%xmm1        # 52e0 <_sk_callback_sse2+0xb94>
-  .byte  102,15,219,21,199,34,0,0            // pand          0x22c7(%rip),%xmm2        # 52f0 <_sk_callback_sse2+0xba4>
+  .byte  15,89,13,170,34,0,0                 // mulps         0x22aa(%rip),%xmm1        # 52f0 <_sk_callback_sse2+0xb7f>
+  .byte  102,15,219,21,178,34,0,0            // pand          0x22b2(%rip),%xmm2        # 5300 <_sk_callback_sse2+0xb8f>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,205,34,0,0                 // mulps         0x22cd(%rip),%xmm2        # 5300 <_sk_callback_sse2+0xbb4>
+  .byte  15,89,21,184,34,0,0                 // mulps         0x22b8(%rip),%xmm2        # 5310 <_sk_callback_sse2+0xb9f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,212,34,0,0                 // movaps        0x22d4(%rip),%xmm3        # 5310 <_sk_callback_sse2+0xbc4>
+  .byte  15,40,29,191,34,0,0                 // movaps        0x22bf(%rip),%xmm3        # 5320 <_sk_callback_sse2+0xbaf>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_565_sse2
@@ -29259,12 +29307,12 @@ FUNCTION(_sk_store_565_sse2)
 _sk_store_565_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,213,34,0,0               // movaps        0x22d5(%rip),%xmm8        # 5320 <_sk_callback_sse2+0xbd4>
+  .byte  68,15,40,5,192,34,0,0               // movaps        0x22c0(%rip),%xmm8        # 5330 <_sk_callback_sse2+0xbbf>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
   .byte  102,65,15,114,241,11                // pslld         $0xb,%xmm9
-  .byte  68,15,40,21,202,34,0,0              // movaps        0x22ca(%rip),%xmm10        # 5330 <_sk_callback_sse2+0xbe4>
+  .byte  68,15,40,21,181,34,0,0              // movaps        0x22b5(%rip),%xmm10        # 5340 <_sk_callback_sse2+0xbcf>
   .byte  68,15,89,209                        // mulps         %xmm1,%xmm10
   .byte  102,69,15,91,210                    // cvtps2dq      %xmm10,%xmm10
   .byte  102,65,15,114,242,5                 // pslld         $0x5,%xmm10
@@ -29288,21 +29336,21 @@ _sk_load_4444_sse2:
   .byte  243,15,126,28,120                   // movq          (%rax,%rdi,2),%xmm3
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,216                       // punpcklwd     %xmm0,%xmm3
-  .byte  102,15,111,5,131,34,0,0             // movdqa        0x2283(%rip),%xmm0        # 5340 <_sk_callback_sse2+0xbf4>
+  .byte  102,15,111,5,110,34,0,0             // movdqa        0x226e(%rip),%xmm0        # 5350 <_sk_callback_sse2+0xbdf>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,133,34,0,0                  // mulps         0x2285(%rip),%xmm0        # 5350 <_sk_callback_sse2+0xc04>
-  .byte  102,15,111,13,141,34,0,0            // movdqa        0x228d(%rip),%xmm1        # 5360 <_sk_callback_sse2+0xc14>
+  .byte  15,89,5,112,34,0,0                  // mulps         0x2270(%rip),%xmm0        # 5360 <_sk_callback_sse2+0xbef>
+  .byte  102,15,111,13,120,34,0,0            // movdqa        0x2278(%rip),%xmm1        # 5370 <_sk_callback_sse2+0xbff>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,143,34,0,0                 // mulps         0x228f(%rip),%xmm1        # 5370 <_sk_callback_sse2+0xc24>
-  .byte  102,15,111,21,151,34,0,0            // movdqa        0x2297(%rip),%xmm2        # 5380 <_sk_callback_sse2+0xc34>
+  .byte  15,89,13,122,34,0,0                 // mulps         0x227a(%rip),%xmm1        # 5380 <_sk_callback_sse2+0xc0f>
+  .byte  102,15,111,21,130,34,0,0            // movdqa        0x2282(%rip),%xmm2        # 5390 <_sk_callback_sse2+0xc1f>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,153,34,0,0                 // mulps         0x2299(%rip),%xmm2        # 5390 <_sk_callback_sse2+0xc44>
-  .byte  102,15,219,29,161,34,0,0            // pand          0x22a1(%rip),%xmm3        # 53a0 <_sk_callback_sse2+0xc54>
+  .byte  15,89,21,132,34,0,0                 // mulps         0x2284(%rip),%xmm2        # 53a0 <_sk_callback_sse2+0xc2f>
+  .byte  102,15,219,29,140,34,0,0            // pand          0x228c(%rip),%xmm3        # 53b0 <_sk_callback_sse2+0xc3f>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,167,34,0,0                 // mulps         0x22a7(%rip),%xmm3        # 53b0 <_sk_callback_sse2+0xc64>
+  .byte  15,89,29,146,34,0,0                 // mulps         0x2292(%rip),%xmm3        # 53c0 <_sk_callback_sse2+0xc4f>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -29338,21 +29386,21 @@ _sk_gather_4444_sse2:
   .byte  102,15,196,216,3                    // pinsrw        $0x3,%eax,%xmm3
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,216                       // punpcklwd     %xmm0,%xmm3
-  .byte  102,15,111,5,46,34,0,0              // movdqa        0x222e(%rip),%xmm0        # 53c0 <_sk_callback_sse2+0xc74>
+  .byte  102,15,111,5,25,34,0,0              // movdqa        0x2219(%rip),%xmm0        # 53d0 <_sk_callback_sse2+0xc5f>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,48,34,0,0                   // mulps         0x2230(%rip),%xmm0        # 53d0 <_sk_callback_sse2+0xc84>
-  .byte  102,15,111,13,56,34,0,0             // movdqa        0x2238(%rip),%xmm1        # 53e0 <_sk_callback_sse2+0xc94>
+  .byte  15,89,5,27,34,0,0                   // mulps         0x221b(%rip),%xmm0        # 53e0 <_sk_callback_sse2+0xc6f>
+  .byte  102,15,111,13,35,34,0,0             // movdqa        0x2223(%rip),%xmm1        # 53f0 <_sk_callback_sse2+0xc7f>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,58,34,0,0                  // mulps         0x223a(%rip),%xmm1        # 53f0 <_sk_callback_sse2+0xca4>
-  .byte  102,15,111,21,66,34,0,0             // movdqa        0x2242(%rip),%xmm2        # 5400 <_sk_callback_sse2+0xcb4>
+  .byte  15,89,13,37,34,0,0                  // mulps         0x2225(%rip),%xmm1        # 5400 <_sk_callback_sse2+0xc8f>
+  .byte  102,15,111,21,45,34,0,0             // movdqa        0x222d(%rip),%xmm2        # 5410 <_sk_callback_sse2+0xc9f>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,68,34,0,0                  // mulps         0x2244(%rip),%xmm2        # 5410 <_sk_callback_sse2+0xcc4>
-  .byte  102,15,219,29,76,34,0,0             // pand          0x224c(%rip),%xmm3        # 5420 <_sk_callback_sse2+0xcd4>
+  .byte  15,89,21,47,34,0,0                  // mulps         0x222f(%rip),%xmm2        # 5420 <_sk_callback_sse2+0xcaf>
+  .byte  102,15,219,29,55,34,0,0             // pand          0x2237(%rip),%xmm3        # 5430 <_sk_callback_sse2+0xcbf>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,82,34,0,0                  // mulps         0x2252(%rip),%xmm3        # 5430 <_sk_callback_sse2+0xce4>
+  .byte  15,89,29,61,34,0,0                  // mulps         0x223d(%rip),%xmm3        # 5440 <_sk_callback_sse2+0xccf>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -29362,7 +29410,7 @@ FUNCTION(_sk_store_4444_sse2)
 _sk_store_4444_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,81,34,0,0                // movaps        0x2251(%rip),%xmm8        # 5440 <_sk_callback_sse2+0xcf4>
+  .byte  68,15,40,5,60,34,0,0                // movaps        0x223c(%rip),%xmm8        # 5450 <_sk_callback_sse2+0xcdf>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -29394,11 +29442,11 @@ _sk_load_8888_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  68,15,16,12,184                     // movups        (%rax,%rdi,4),%xmm9
-  .byte  15,40,21,228,33,0,0                 // movaps        0x21e4(%rip),%xmm2        # 5450 <_sk_callback_sse2+0xd04>
+  .byte  15,40,21,207,33,0,0                 // movaps        0x21cf(%rip),%xmm2        # 5460 <_sk_callback_sse2+0xcef>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  15,84,194                           // andps         %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,226,33,0,0               // movaps        0x21e2(%rip),%xmm8        # 5460 <_sk_callback_sse2+0xd14>
+  .byte  68,15,40,5,205,33,0,0               // movaps        0x21cd(%rip),%xmm8        # 5470 <_sk_callback_sse2+0xcff>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,40,201                        // movaps        %xmm9,%xmm1
   .byte  102,15,114,209,8                    // psrld         $0x8,%xmm1
@@ -29447,11 +29495,11 @@ _sk_gather_8888_sse2:
   .byte  102,67,15,110,12,129                // movd          (%r9,%r8,4),%xmm1
   .byte  102,68,15,98,201                    // punpckldq     %xmm1,%xmm9
   .byte  102,68,15,98,200                    // punpckldq     %xmm0,%xmm9
-  .byte  102,15,111,21,51,33,0,0             // movdqa        0x2133(%rip),%xmm2        # 5470 <_sk_callback_sse2+0xd24>
+  .byte  102,15,111,21,30,33,0,0             // movdqa        0x211e(%rip),%xmm2        # 5480 <_sk_callback_sse2+0xd0f>
   .byte  102,65,15,111,193                   // movdqa        %xmm9,%xmm0
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,47,33,0,0                // movaps        0x212f(%rip),%xmm8        # 5480 <_sk_callback_sse2+0xd34>
+  .byte  68,15,40,5,26,33,0,0                // movaps        0x211a(%rip),%xmm8        # 5490 <_sk_callback_sse2+0xd1f>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,114,209,8                    // psrld         $0x8,%xmm1
@@ -29475,7 +29523,7 @@ FUNCTION(_sk_store_8888_sse2)
 _sk_store_8888_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,242,32,0,0               // movaps        0x20f2(%rip),%xmm8        # 5490 <_sk_callback_sse2+0xd44>
+  .byte  68,15,40,5,221,32,0,0               // movaps        0x20dd(%rip),%xmm8        # 54a0 <_sk_callback_sse2+0xd2f>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -29514,7 +29562,7 @@ _sk_load_f16_sse2:
   .byte  102,69,15,239,210                   // pxor          %xmm10,%xmm10
   .byte  102,65,15,111,206                   // movdqa        %xmm14,%xmm1
   .byte  102,65,15,97,202                    // punpcklwd     %xmm10,%xmm1
-  .byte  102,68,15,111,13,98,32,0,0          // movdqa        0x2062(%rip),%xmm9        # 54a0 <_sk_callback_sse2+0xd54>
+  .byte  102,68,15,111,13,77,32,0,0          // movdqa        0x204d(%rip),%xmm9        # 54b0 <_sk_callback_sse2+0xd3f>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,193                   // pand          %xmm9,%xmm0
   .byte  102,15,239,200                      // pxor          %xmm0,%xmm1
@@ -29522,11 +29570,11 @@ _sk_load_f16_sse2:
   .byte  102,68,15,111,233                   // movdqa        %xmm1,%xmm13
   .byte  102,65,15,114,245,13                // pslld         $0xd,%xmm13
   .byte  102,68,15,235,232                   // por           %xmm0,%xmm13
-  .byte  102,68,15,111,29,71,32,0,0          // movdqa        0x2047(%rip),%xmm11        # 54b0 <_sk_callback_sse2+0xd64>
+  .byte  102,68,15,111,29,50,32,0,0          // movdqa        0x2032(%rip),%xmm11        # 54c0 <_sk_callback_sse2+0xd4f>
   .byte  102,69,15,254,235                   // paddd         %xmm11,%xmm13
-  .byte  102,68,15,111,37,73,32,0,0          // movdqa        0x2049(%rip),%xmm12        # 54c0 <_sk_callback_sse2+0xd74>
+  .byte  102,68,15,111,37,52,32,0,0          // movdqa        0x2034(%rip),%xmm12        # 54d0 <_sk_callback_sse2+0xd5f>
   .byte  102,65,15,239,204                   // pxor          %xmm12,%xmm1
-  .byte  102,15,111,29,76,32,0,0             // movdqa        0x204c(%rip),%xmm3        # 54d0 <_sk_callback_sse2+0xd84>
+  .byte  102,15,111,29,55,32,0,0             // movdqa        0x2037(%rip),%xmm3        # 54e0 <_sk_callback_sse2+0xd6f>
   .byte  102,15,111,195                      // movdqa        %xmm3,%xmm0
   .byte  102,15,102,193                      // pcmpgtd       %xmm1,%xmm0
   .byte  102,65,15,223,197                   // pandn         %xmm13,%xmm0
@@ -29612,7 +29660,7 @@ _sk_gather_f16_sse2:
   .byte  102,69,15,239,210                   // pxor          %xmm10,%xmm10
   .byte  102,65,15,111,206                   // movdqa        %xmm14,%xmm1
   .byte  102,65,15,97,202                    // punpcklwd     %xmm10,%xmm1
-  .byte  102,68,15,111,13,218,30,0,0         // movdqa        0x1eda(%rip),%xmm9        # 54e0 <_sk_callback_sse2+0xd94>
+  .byte  102,68,15,111,13,197,30,0,0         // movdqa        0x1ec5(%rip),%xmm9        # 54f0 <_sk_callback_sse2+0xd7f>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,193                   // pand          %xmm9,%xmm0
   .byte  102,15,239,200                      // pxor          %xmm0,%xmm1
@@ -29620,11 +29668,11 @@ _sk_gather_f16_sse2:
   .byte  102,68,15,111,233                   // movdqa        %xmm1,%xmm13
   .byte  102,65,15,114,245,13                // pslld         $0xd,%xmm13
   .byte  102,68,15,235,232                   // por           %xmm0,%xmm13
-  .byte  102,68,15,111,29,191,30,0,0         // movdqa        0x1ebf(%rip),%xmm11        # 54f0 <_sk_callback_sse2+0xda4>
+  .byte  102,68,15,111,29,170,30,0,0         // movdqa        0x1eaa(%rip),%xmm11        # 5500 <_sk_callback_sse2+0xd8f>
   .byte  102,69,15,254,235                   // paddd         %xmm11,%xmm13
-  .byte  102,68,15,111,37,193,30,0,0         // movdqa        0x1ec1(%rip),%xmm12        # 5500 <_sk_callback_sse2+0xdb4>
+  .byte  102,68,15,111,37,172,30,0,0         // movdqa        0x1eac(%rip),%xmm12        # 5510 <_sk_callback_sse2+0xd9f>
   .byte  102,65,15,239,204                   // pxor          %xmm12,%xmm1
-  .byte  102,15,111,29,196,30,0,0            // movdqa        0x1ec4(%rip),%xmm3        # 5510 <_sk_callback_sse2+0xdc4>
+  .byte  102,15,111,29,175,30,0,0            // movdqa        0x1eaf(%rip),%xmm3        # 5520 <_sk_callback_sse2+0xdaf>
   .byte  102,15,111,195                      // movdqa        %xmm3,%xmm0
   .byte  102,15,102,193                      // pcmpgtd       %xmm1,%xmm0
   .byte  102,65,15,223,197                   // pandn         %xmm13,%xmm0
@@ -29677,17 +29725,17 @@ FUNCTION(_sk_store_f16_sse2)
 _sk_store_f16_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  102,68,15,111,21,236,29,0,0         // movdqa        0x1dec(%rip),%xmm10        # 5520 <_sk_callback_sse2+0xdd4>
+  .byte  102,68,15,111,21,215,29,0,0         // movdqa        0x1dd7(%rip),%xmm10        # 5530 <_sk_callback_sse2+0xdbf>
   .byte  102,68,15,111,224                   // movdqa        %xmm0,%xmm12
   .byte  102,68,15,111,232                   // movdqa        %xmm0,%xmm13
   .byte  102,69,15,219,234                   // pand          %xmm10,%xmm13
   .byte  102,69,15,239,229                   // pxor          %xmm13,%xmm12
-  .byte  102,68,15,111,13,223,29,0,0         // movdqa        0x1ddf(%rip),%xmm9        # 5530 <_sk_callback_sse2+0xde4>
+  .byte  102,68,15,111,13,202,29,0,0         // movdqa        0x1dca(%rip),%xmm9        # 5540 <_sk_callback_sse2+0xdcf>
   .byte  102,65,15,114,213,16                // psrld         $0x10,%xmm13
   .byte  102,69,15,111,193                   // movdqa        %xmm9,%xmm8
   .byte  102,69,15,102,196                   // pcmpgtd       %xmm12,%xmm8
   .byte  102,65,15,114,212,13                // psrld         $0xd,%xmm12
-  .byte  102,68,15,111,29,208,29,0,0         // movdqa        0x1dd0(%rip),%xmm11        # 5540 <_sk_callback_sse2+0xdf4>
+  .byte  102,68,15,111,29,187,29,0,0         // movdqa        0x1dbb(%rip),%xmm11        # 5550 <_sk_callback_sse2+0xddf>
   .byte  102,69,15,235,235                   // por           %xmm11,%xmm13
   .byte  102,69,15,254,236                   // paddd         %xmm12,%xmm13
   .byte  102,65,15,114,245,16                // pslld         $0x10,%xmm13
@@ -29766,7 +29814,7 @@ _sk_load_u16_be_sse2:
   .byte  102,69,15,239,201                   // pxor          %xmm9,%xmm9
   .byte  102,65,15,97,201                    // punpcklwd     %xmm9,%xmm1
   .byte  15,91,193                           // cvtdq2ps      %xmm1,%xmm0
-  .byte  68,15,40,5,110,28,0,0               // movaps        0x1c6e(%rip),%xmm8        # 5550 <_sk_callback_sse2+0xe04>
+  .byte  68,15,40,5,89,28,0,0                // movaps        0x1c59(%rip),%xmm8        # 5560 <_sk_callback_sse2+0xdef>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -29819,7 +29867,7 @@ _sk_load_rgb_u16_be_sse2:
   .byte  102,69,15,239,192                   // pxor          %xmm8,%xmm8
   .byte  102,65,15,97,192                    // punpcklwd     %xmm8,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,13,170,27,0,0              // movaps        0x1baa(%rip),%xmm9        # 5560 <_sk_callback_sse2+0xe14>
+  .byte  68,15,40,13,149,27,0,0              // movaps        0x1b95(%rip),%xmm9        # 5570 <_sk_callback_sse2+0xdff>
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -29836,7 +29884,7 @@ _sk_load_rgb_u16_be_sse2:
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,113,27,0,0                 // movaps        0x1b71(%rip),%xmm3        # 5570 <_sk_callback_sse2+0xe24>
+  .byte  15,40,29,92,27,0,0                  // movaps        0x1b5c(%rip),%xmm3        # 5580 <_sk_callback_sse2+0xe0f>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_u16_be_sse2
@@ -29845,7 +29893,7 @@ FUNCTION(_sk_store_u16_be_sse2)
 _sk_store_u16_be_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,13,114,27,0,0              // movaps        0x1b72(%rip),%xmm9        # 5580 <_sk_callback_sse2+0xe34>
+  .byte  68,15,40,13,93,27,0,0               // movaps        0x1b5d(%rip),%xmm9        # 5590 <_sk_callback_sse2+0xe1f>
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
@@ -29995,7 +30043,7 @@ _sk_repeat_x_sse2:
   .byte  243,69,15,91,209                    // cvttps2dq     %xmm9,%xmm10
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
   .byte  69,15,194,202,1                     // cmpltps       %xmm10,%xmm9
-  .byte  68,15,84,13,92,25,0,0               // andps         0x195c(%rip),%xmm9        # 5590 <_sk_callback_sse2+0xe44>
+  .byte  68,15,84,13,71,25,0,0               // andps         0x1947(%rip),%xmm9        # 55a0 <_sk_callback_sse2+0xe2f>
   .byte  69,15,92,209                        // subps         %xmm9,%xmm10
   .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
   .byte  65,15,92,194                        // subps         %xmm10,%xmm0
@@ -30017,7 +30065,7 @@ _sk_repeat_y_sse2:
   .byte  243,69,15,91,209                    // cvttps2dq     %xmm9,%xmm10
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
   .byte  69,15,194,202,1                     // cmpltps       %xmm10,%xmm9
-  .byte  68,15,84,13,36,25,0,0               // andps         0x1924(%rip),%xmm9        # 55a0 <_sk_callback_sse2+0xe54>
+  .byte  68,15,84,13,15,25,0,0               // andps         0x190f(%rip),%xmm9        # 55b0 <_sk_callback_sse2+0xe3f>
   .byte  69,15,92,209                        // subps         %xmm9,%xmm10
   .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
   .byte  65,15,92,202                        // subps         %xmm10,%xmm1
@@ -30043,7 +30091,7 @@ _sk_mirror_x_sse2:
   .byte  243,69,15,91,218                    // cvttps2dq     %xmm10,%xmm11
   .byte  69,15,91,219                        // cvtdq2ps      %xmm11,%xmm11
   .byte  69,15,194,211,1                     // cmpltps       %xmm11,%xmm10
-  .byte  68,15,84,21,218,24,0,0              // andps         0x18da(%rip),%xmm10        # 55b0 <_sk_callback_sse2+0xe64>
+  .byte  68,15,84,21,197,24,0,0              // andps         0x18c5(%rip),%xmm10        # 55c0 <_sk_callback_sse2+0xe4f>
   .byte  69,15,87,228                        // xorps         %xmm12,%xmm12
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  69,15,89,216                        // mulps         %xmm8,%xmm11
@@ -30073,7 +30121,7 @@ _sk_mirror_y_sse2:
   .byte  243,69,15,91,218                    // cvttps2dq     %xmm10,%xmm11
   .byte  69,15,91,219                        // cvtdq2ps      %xmm11,%xmm11
   .byte  69,15,194,211,1                     // cmpltps       %xmm11,%xmm10
-  .byte  68,15,84,21,128,24,0,0              // andps         0x1880(%rip),%xmm10        # 55c0 <_sk_callback_sse2+0xe74>
+  .byte  68,15,84,21,107,24,0,0              // andps         0x186b(%rip),%xmm10        # 55d0 <_sk_callback_sse2+0xe5f>
   .byte  69,15,87,228                        // xorps         %xmm12,%xmm12
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  69,15,89,216                        // mulps         %xmm8,%xmm11
@@ -30092,10 +30140,10 @@ HIDDEN _sk_luminance_to_alpha_sse2
 FUNCTION(_sk_luminance_to_alpha_sse2)
 _sk_luminance_to_alpha_sse2:
   .byte  15,40,218                           // movaps        %xmm2,%xmm3
-  .byte  15,89,5,88,24,0,0                   // mulps         0x1858(%rip),%xmm0        # 55d0 <_sk_callback_sse2+0xe84>
-  .byte  15,89,13,97,24,0,0                  // mulps         0x1861(%rip),%xmm1        # 55e0 <_sk_callback_sse2+0xe94>
+  .byte  15,89,5,67,24,0,0                   // mulps         0x1843(%rip),%xmm0        # 55e0 <_sk_callback_sse2+0xe6f>
+  .byte  15,89,13,76,24,0,0                  // mulps         0x184c(%rip),%xmm1        # 55f0 <_sk_callback_sse2+0xe7f>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  15,89,29,103,24,0,0                 // mulps         0x1867(%rip),%xmm3        # 55f0 <_sk_callback_sse2+0xea4>
+  .byte  15,89,29,82,24,0,0                  // mulps         0x1852(%rip),%xmm3        # 5600 <_sk_callback_sse2+0xe8f>
   .byte  15,88,217                           // addps         %xmm1,%xmm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
@@ -30328,7 +30376,7 @@ _sk_linear_gradient_sse2:
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
   .byte  72,139,8                            // mov           (%rax),%rcx
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,132,15,1,0,0                     // je            423c <_sk_linear_gradient_sse2+0x149>
+  .byte  15,132,15,1,0,0                     // je            4261 <_sk_linear_gradient_sse2+0x149>
   .byte  72,139,64,8                         // mov           0x8(%rax),%rax
   .byte  72,131,192,32                       // add           $0x20,%rax
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
@@ -30389,8 +30437,8 @@ _sk_linear_gradient_sse2:
   .byte  69,15,86,231                        // orps          %xmm15,%xmm12
   .byte  72,131,192,36                       // add           $0x24,%rax
   .byte  72,255,201                          // dec           %rcx
-  .byte  15,133,8,255,255,255                // jne           4142 <_sk_linear_gradient_sse2+0x4f>
-  .byte  235,13                              // jmp           4249 <_sk_linear_gradient_sse2+0x156>
+  .byte  15,133,8,255,255,255                // jne           4167 <_sk_linear_gradient_sse2+0x4f>
+  .byte  235,13                              // jmp           426e <_sk_linear_gradient_sse2+0x156>
   .byte  15,87,201                           // xorps         %xmm1,%xmm1
   .byte  15,87,210                           // xorps         %xmm2,%xmm2
   .byte  15,87,219                           // xorps         %xmm3,%xmm3
@@ -30459,29 +30507,29 @@ _sk_xy_to_polar_unit_sse2:
   .byte  69,15,94,220                        // divps         %xmm12,%xmm11
   .byte  69,15,40,227                        // movaps        %xmm11,%xmm12
   .byte  69,15,89,228                        // mulps         %xmm12,%xmm12
-  .byte  68,15,40,45,223,18,0,0              // movaps        0x12df(%rip),%xmm13        # 5600 <_sk_callback_sse2+0xeb4>
+  .byte  68,15,40,45,202,18,0,0              // movaps        0x12ca(%rip),%xmm13        # 5610 <_sk_callback_sse2+0xe9f>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
-  .byte  68,15,88,45,227,18,0,0              // addps         0x12e3(%rip),%xmm13        # 5610 <_sk_callback_sse2+0xec4>
+  .byte  68,15,88,45,206,18,0,0              // addps         0x12ce(%rip),%xmm13        # 5620 <_sk_callback_sse2+0xeaf>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
-  .byte  68,15,88,45,231,18,0,0              // addps         0x12e7(%rip),%xmm13        # 5620 <_sk_callback_sse2+0xed4>
+  .byte  68,15,88,45,210,18,0,0              // addps         0x12d2(%rip),%xmm13        # 5630 <_sk_callback_sse2+0xebf>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
-  .byte  68,15,88,45,235,18,0,0              // addps         0x12eb(%rip),%xmm13        # 5630 <_sk_callback_sse2+0xee4>
+  .byte  68,15,88,45,214,18,0,0              // addps         0x12d6(%rip),%xmm13        # 5640 <_sk_callback_sse2+0xecf>
   .byte  69,15,89,235                        // mulps         %xmm11,%xmm13
   .byte  69,15,194,202,1                     // cmpltps       %xmm10,%xmm9
-  .byte  68,15,40,21,234,18,0,0              // movaps        0x12ea(%rip),%xmm10        # 5640 <_sk_callback_sse2+0xef4>
+  .byte  68,15,40,21,213,18,0,0              // movaps        0x12d5(%rip),%xmm10        # 5650 <_sk_callback_sse2+0xedf>
   .byte  69,15,92,213                        // subps         %xmm13,%xmm10
   .byte  69,15,84,209                        // andps         %xmm9,%xmm10
   .byte  69,15,85,205                        // andnps        %xmm13,%xmm9
   .byte  69,15,86,202                        // orps          %xmm10,%xmm9
   .byte  68,15,194,192,1                     // cmpltps       %xmm0,%xmm8
-  .byte  68,15,40,21,221,18,0,0              // movaps        0x12dd(%rip),%xmm10        # 5650 <_sk_callback_sse2+0xf04>
+  .byte  68,15,40,21,200,18,0,0              // movaps        0x12c8(%rip),%xmm10        # 5660 <_sk_callback_sse2+0xeef>
   .byte  69,15,92,209                        // subps         %xmm9,%xmm10
   .byte  69,15,84,208                        // andps         %xmm8,%xmm10
   .byte  69,15,85,193                        // andnps        %xmm9,%xmm8
   .byte  69,15,86,194                        // orps          %xmm10,%xmm8
   .byte  68,15,40,201                        // movaps        %xmm1,%xmm9
   .byte  68,15,194,200,1                     // cmpltps       %xmm0,%xmm9
-  .byte  68,15,40,21,204,18,0,0              // movaps        0x12cc(%rip),%xmm10        # 5660 <_sk_callback_sse2+0xf14>
+  .byte  68,15,40,21,183,18,0,0              // movaps        0x12b7(%rip),%xmm10        # 5670 <_sk_callback_sse2+0xeff>
   .byte  69,15,92,208                        // subps         %xmm8,%xmm10
   .byte  69,15,84,209                        // andps         %xmm9,%xmm10
   .byte  69,15,85,200                        // andnps        %xmm8,%xmm9
@@ -30509,7 +30557,7 @@ HIDDEN _sk_save_xy_sse2
 FUNCTION(_sk_save_xy_sse2)
 _sk_save_xy_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,155,18,0,0               // movaps        0x129b(%rip),%xmm8        # 5670 <_sk_callback_sse2+0xf24>
+  .byte  68,15,40,5,134,18,0,0               // movaps        0x1286(%rip),%xmm8        # 5680 <_sk_callback_sse2+0xf0f>
   .byte  15,17,0                             // movups        %xmm0,(%rax)
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,88,200                        // addps         %xmm8,%xmm9
@@ -30517,7 +30565,7 @@ _sk_save_xy_sse2:
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
   .byte  69,15,40,217                        // movaps        %xmm9,%xmm11
   .byte  69,15,194,218,1                     // cmpltps       %xmm10,%xmm11
-  .byte  68,15,40,37,134,18,0,0              // movaps        0x1286(%rip),%xmm12        # 5680 <_sk_callback_sse2+0xf34>
+  .byte  68,15,40,37,113,18,0,0              // movaps        0x1271(%rip),%xmm12        # 5690 <_sk_callback_sse2+0xf1f>
   .byte  69,15,84,220                        // andps         %xmm12,%xmm11
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
   .byte  69,15,92,202                        // subps         %xmm10,%xmm9
@@ -30564,8 +30612,8 @@ _sk_bilinear_nx_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,255,17,0,0                  // addps         0x11ff(%rip),%xmm0        # 5690 <_sk_callback_sse2+0xf44>
-  .byte  68,15,40,13,7,18,0,0                // movaps        0x1207(%rip),%xmm9        # 56a0 <_sk_callback_sse2+0xf54>
+  .byte  15,88,5,234,17,0,0                  // addps         0x11ea(%rip),%xmm0        # 56a0 <_sk_callback_sse2+0xf2f>
+  .byte  68,15,40,13,242,17,0,0              // movaps        0x11f2(%rip),%xmm9        # 56b0 <_sk_callback_sse2+0xf3f>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -30578,7 +30626,7 @@ _sk_bilinear_px_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,246,17,0,0                  // addps         0x11f6(%rip),%xmm0        # 56b0 <_sk_callback_sse2+0xf64>
+  .byte  15,88,5,225,17,0,0                  // addps         0x11e1(%rip),%xmm0        # 56c0 <_sk_callback_sse2+0xf4f>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -30590,8 +30638,8 @@ _sk_bilinear_ny_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,232,17,0,0                 // addps         0x11e8(%rip),%xmm1        # 56c0 <_sk_callback_sse2+0xf74>
-  .byte  68,15,40,13,240,17,0,0              // movaps        0x11f0(%rip),%xmm9        # 56d0 <_sk_callback_sse2+0xf84>
+  .byte  15,88,13,211,17,0,0                 // addps         0x11d3(%rip),%xmm1        # 56d0 <_sk_callback_sse2+0xf5f>
+  .byte  68,15,40,13,219,17,0,0              // movaps        0x11db(%rip),%xmm9        # 56e0 <_sk_callback_sse2+0xf6f>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -30604,7 +30652,7 @@ _sk_bilinear_py_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,222,17,0,0                 // addps         0x11de(%rip),%xmm1        # 56e0 <_sk_callback_sse2+0xf94>
+  .byte  15,88,13,201,17,0,0                 // addps         0x11c9(%rip),%xmm1        # 56f0 <_sk_callback_sse2+0xf7f>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -30616,13 +30664,13 @@ _sk_bicubic_n3x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,209,17,0,0                  // addps         0x11d1(%rip),%xmm0        # 56f0 <_sk_callback_sse2+0xfa4>
-  .byte  68,15,40,13,217,17,0,0              // movaps        0x11d9(%rip),%xmm9        # 5700 <_sk_callback_sse2+0xfb4>
+  .byte  15,88,5,188,17,0,0                  // addps         0x11bc(%rip),%xmm0        # 5700 <_sk_callback_sse2+0xf8f>
+  .byte  68,15,40,13,196,17,0,0              // movaps        0x11c4(%rip),%xmm9        # 5710 <_sk_callback_sse2+0xf9f>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,213,17,0,0              // mulps         0x11d5(%rip),%xmm9        # 5710 <_sk_callback_sse2+0xfc4>
-  .byte  68,15,88,13,221,17,0,0              // addps         0x11dd(%rip),%xmm9        # 5720 <_sk_callback_sse2+0xfd4>
+  .byte  68,15,89,13,192,17,0,0              // mulps         0x11c0(%rip),%xmm9        # 5720 <_sk_callback_sse2+0xfaf>
+  .byte  68,15,88,13,200,17,0,0              // addps         0x11c8(%rip),%xmm9        # 5730 <_sk_callback_sse2+0xfbf>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -30635,16 +30683,16 @@ _sk_bicubic_n1x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,204,17,0,0                  // addps         0x11cc(%rip),%xmm0        # 5730 <_sk_callback_sse2+0xfe4>
-  .byte  68,15,40,13,212,17,0,0              // movaps        0x11d4(%rip),%xmm9        # 5740 <_sk_callback_sse2+0xff4>
+  .byte  15,88,5,183,17,0,0                  // addps         0x11b7(%rip),%xmm0        # 5740 <_sk_callback_sse2+0xfcf>
+  .byte  68,15,40,13,191,17,0,0              // movaps        0x11bf(%rip),%xmm9        # 5750 <_sk_callback_sse2+0xfdf>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,216,17,0,0               // movaps        0x11d8(%rip),%xmm8        # 5750 <_sk_callback_sse2+0x1004>
+  .byte  68,15,40,5,195,17,0,0               // movaps        0x11c3(%rip),%xmm8        # 5760 <_sk_callback_sse2+0xfef>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,220,17,0,0               // addps         0x11dc(%rip),%xmm8        # 5760 <_sk_callback_sse2+0x1014>
+  .byte  68,15,88,5,199,17,0,0               // addps         0x11c7(%rip),%xmm8        # 5770 <_sk_callback_sse2+0xfff>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,224,17,0,0               // addps         0x11e0(%rip),%xmm8        # 5770 <_sk_callback_sse2+0x1024>
+  .byte  68,15,88,5,203,17,0,0               // addps         0x11cb(%rip),%xmm8        # 5780 <_sk_callback_sse2+0x100f>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,228,17,0,0               // addps         0x11e4(%rip),%xmm8        # 5780 <_sk_callback_sse2+0x1034>
+  .byte  68,15,88,5,207,17,0,0               // addps         0x11cf(%rip),%xmm8        # 5790 <_sk_callback_sse2+0x101f>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -30654,17 +30702,17 @@ HIDDEN _sk_bicubic_p1x_sse2
 FUNCTION(_sk_bicubic_p1x_sse2)
 _sk_bicubic_p1x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,222,17,0,0               // movaps        0x11de(%rip),%xmm8        # 5790 <_sk_callback_sse2+0x1044>
+  .byte  68,15,40,5,201,17,0,0               // movaps        0x11c9(%rip),%xmm8        # 57a0 <_sk_callback_sse2+0x102f>
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,72,64                      // movups        0x40(%rax),%xmm9
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
-  .byte  68,15,40,21,218,17,0,0              // movaps        0x11da(%rip),%xmm10        # 57a0 <_sk_callback_sse2+0x1054>
+  .byte  68,15,40,21,197,17,0,0              // movaps        0x11c5(%rip),%xmm10        # 57b0 <_sk_callback_sse2+0x103f>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,222,17,0,0              // addps         0x11de(%rip),%xmm10        # 57b0 <_sk_callback_sse2+0x1064>
+  .byte  68,15,88,21,201,17,0,0              // addps         0x11c9(%rip),%xmm10        # 57c0 <_sk_callback_sse2+0x104f>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,218,17,0,0              // addps         0x11da(%rip),%xmm10        # 57c0 <_sk_callback_sse2+0x1074>
+  .byte  68,15,88,21,197,17,0,0              // addps         0x11c5(%rip),%xmm10        # 57d0 <_sk_callback_sse2+0x105f>
   .byte  68,15,17,144,128,0,0,0              // movups        %xmm10,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -30676,11 +30724,11 @@ _sk_bicubic_p3x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,205,17,0,0                  // addps         0x11cd(%rip),%xmm0        # 57d0 <_sk_callback_sse2+0x1084>
+  .byte  15,88,5,184,17,0,0                  // addps         0x11b8(%rip),%xmm0        # 57e0 <_sk_callback_sse2+0x106f>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,205,17,0,0               // mulps         0x11cd(%rip),%xmm8        # 57e0 <_sk_callback_sse2+0x1094>
-  .byte  68,15,88,5,213,17,0,0               // addps         0x11d5(%rip),%xmm8        # 57f0 <_sk_callback_sse2+0x10a4>
+  .byte  68,15,89,5,184,17,0,0               // mulps         0x11b8(%rip),%xmm8        # 57f0 <_sk_callback_sse2+0x107f>
+  .byte  68,15,88,5,192,17,0,0               // addps         0x11c0(%rip),%xmm8        # 5800 <_sk_callback_sse2+0x108f>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -30693,13 +30741,13 @@ _sk_bicubic_n3y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,195,17,0,0                 // addps         0x11c3(%rip),%xmm1        # 5800 <_sk_callback_sse2+0x10b4>
-  .byte  68,15,40,13,203,17,0,0              // movaps        0x11cb(%rip),%xmm9        # 5810 <_sk_callback_sse2+0x10c4>
+  .byte  15,88,13,174,17,0,0                 // addps         0x11ae(%rip),%xmm1        # 5810 <_sk_callback_sse2+0x109f>
+  .byte  68,15,40,13,182,17,0,0              // movaps        0x11b6(%rip),%xmm9        # 5820 <_sk_callback_sse2+0x10af>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,199,17,0,0              // mulps         0x11c7(%rip),%xmm9        # 5820 <_sk_callback_sse2+0x10d4>
-  .byte  68,15,88,13,207,17,0,0              // addps         0x11cf(%rip),%xmm9        # 5830 <_sk_callback_sse2+0x10e4>
+  .byte  68,15,89,13,178,17,0,0              // mulps         0x11b2(%rip),%xmm9        # 5830 <_sk_callback_sse2+0x10bf>
+  .byte  68,15,88,13,186,17,0,0              // addps         0x11ba(%rip),%xmm9        # 5840 <_sk_callback_sse2+0x10cf>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -30712,16 +30760,16 @@ _sk_bicubic_n1y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,189,17,0,0                 // addps         0x11bd(%rip),%xmm1        # 5840 <_sk_callback_sse2+0x10f4>
-  .byte  68,15,40,13,197,17,0,0              // movaps        0x11c5(%rip),%xmm9        # 5850 <_sk_callback_sse2+0x1104>
+  .byte  15,88,13,168,17,0,0                 // addps         0x11a8(%rip),%xmm1        # 5850 <_sk_callback_sse2+0x10df>
+  .byte  68,15,40,13,176,17,0,0              // movaps        0x11b0(%rip),%xmm9        # 5860 <_sk_callback_sse2+0x10ef>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,201,17,0,0               // movaps        0x11c9(%rip),%xmm8        # 5860 <_sk_callback_sse2+0x1114>
+  .byte  68,15,40,5,180,17,0,0               // movaps        0x11b4(%rip),%xmm8        # 5870 <_sk_callback_sse2+0x10ff>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,205,17,0,0               // addps         0x11cd(%rip),%xmm8        # 5870 <_sk_callback_sse2+0x1124>
+  .byte  68,15,88,5,184,17,0,0               // addps         0x11b8(%rip),%xmm8        # 5880 <_sk_callback_sse2+0x110f>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,209,17,0,0               // addps         0x11d1(%rip),%xmm8        # 5880 <_sk_callback_sse2+0x1134>
+  .byte  68,15,88,5,188,17,0,0               // addps         0x11bc(%rip),%xmm8        # 5890 <_sk_callback_sse2+0x111f>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,213,17,0,0               // addps         0x11d5(%rip),%xmm8        # 5890 <_sk_callback_sse2+0x1144>
+  .byte  68,15,88,5,192,17,0,0               // addps         0x11c0(%rip),%xmm8        # 58a0 <_sk_callback_sse2+0x112f>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -30731,17 +30779,17 @@ HIDDEN _sk_bicubic_p1y_sse2
 FUNCTION(_sk_bicubic_p1y_sse2)
 _sk_bicubic_p1y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,207,17,0,0               // movaps        0x11cf(%rip),%xmm8        # 58a0 <_sk_callback_sse2+0x1154>
+  .byte  68,15,40,5,186,17,0,0               // movaps        0x11ba(%rip),%xmm8        # 58b0 <_sk_callback_sse2+0x113f>
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,72,96                      // movups        0x60(%rax),%xmm9
   .byte  65,15,88,200                        // addps         %xmm8,%xmm1
-  .byte  68,15,40,21,202,17,0,0              // movaps        0x11ca(%rip),%xmm10        # 58b0 <_sk_callback_sse2+0x1164>
+  .byte  68,15,40,21,181,17,0,0              // movaps        0x11b5(%rip),%xmm10        # 58c0 <_sk_callback_sse2+0x114f>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,206,17,0,0              // addps         0x11ce(%rip),%xmm10        # 58c0 <_sk_callback_sse2+0x1174>
+  .byte  68,15,88,21,185,17,0,0              // addps         0x11b9(%rip),%xmm10        # 58d0 <_sk_callback_sse2+0x115f>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,202,17,0,0              // addps         0x11ca(%rip),%xmm10        # 58d0 <_sk_callback_sse2+0x1184>
+  .byte  68,15,88,21,181,17,0,0              // addps         0x11b5(%rip),%xmm10        # 58e0 <_sk_callback_sse2+0x116f>
   .byte  68,15,17,144,160,0,0,0              // movups        %xmm10,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -30753,11 +30801,11 @@ _sk_bicubic_p3y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,188,17,0,0                 // addps         0x11bc(%rip),%xmm1        # 58e0 <_sk_callback_sse2+0x1194>
+  .byte  15,88,13,167,17,0,0                 // addps         0x11a7(%rip),%xmm1        # 58f0 <_sk_callback_sse2+0x117f>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,188,17,0,0               // mulps         0x11bc(%rip),%xmm8        # 58f0 <_sk_callback_sse2+0x11a4>
-  .byte  68,15,88,5,196,17,0,0               // addps         0x11c4(%rip),%xmm8        # 5900 <_sk_callback_sse2+0x11b4>
+  .byte  68,15,89,5,167,17,0,0               // mulps         0x11a7(%rip),%xmm8        # 5900 <_sk_callback_sse2+0x118f>
+  .byte  68,15,88,5,175,17,0,0               // addps         0x11af(%rip),%xmm8        # 5910 <_sk_callback_sse2+0x119f>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -30976,11 +31024,11 @@ BALIGN16
   .byte  128,191,0,0,128,191,0               // cmpb          $0x0,-0x40800000(%rdi)
   .byte  0,224                               // add           %ah,%al
   .byte  64,0,0                              // add           %al,(%rax)
-  .byte  224,64                              // loopne        4a08 <.literal16+0x1d8>
+  .byte  224,64                              // loopne        4a28 <.literal16+0x1d8>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        4a0c <.literal16+0x1dc>
+  .byte  224,64                              // loopne        4a2c <.literal16+0x1dc>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        4a10 <.literal16+0x1e0>
+  .byte  224,64                              // loopne        4a30 <.literal16+0x1e0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -31005,13 +31053,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a41 <.literal16+0x211>
+  .byte  71,225,61                           // rex.RXB       loope 4a61 <.literal16+0x211>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a45 <.literal16+0x215>
+  .byte  71,225,61                           // rex.RXB       loope 4a65 <.literal16+0x215>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a49 <.literal16+0x219>
+  .byte  71,225,61                           // rex.RXB       loope 4a69 <.literal16+0x219>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a4d <.literal16+0x21d>
+  .byte  71,225,61                           // rex.RXB       loope 4a6d <.literal16+0x21d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -31036,13 +31084,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a81 <.literal16+0x251>
+  .byte  71,225,61                           // rex.RXB       loope 4aa1 <.literal16+0x251>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a85 <.literal16+0x255>
+  .byte  71,225,61                           // rex.RXB       loope 4aa5 <.literal16+0x255>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a89 <.literal16+0x259>
+  .byte  71,225,61                           // rex.RXB       loope 4aa9 <.literal16+0x259>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4a8d <.literal16+0x25d>
+  .byte  71,225,61                           // rex.RXB       loope 4aad <.literal16+0x25d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -31067,13 +31115,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4ac1 <.literal16+0x291>
+  .byte  71,225,61                           // rex.RXB       loope 4ae1 <.literal16+0x291>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4ac5 <.literal16+0x295>
+  .byte  71,225,61                           // rex.RXB       loope 4ae5 <.literal16+0x295>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4ac9 <.literal16+0x299>
+  .byte  71,225,61                           // rex.RXB       loope 4ae9 <.literal16+0x299>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4acd <.literal16+0x29d>
+  .byte  71,225,61                           // rex.RXB       loope 4aed <.literal16+0x29d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -31098,13 +31146,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4b01 <.literal16+0x2d1>
+  .byte  71,225,61                           // rex.RXB       loope 4b21 <.literal16+0x2d1>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4b05 <.literal16+0x2d5>
+  .byte  71,225,61                           // rex.RXB       loope 4b25 <.literal16+0x2d5>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4b09 <.literal16+0x2d9>
+  .byte  71,225,61                           // rex.RXB       loope 4b29 <.literal16+0x2d9>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4b0d <.literal16+0x2dd>
+  .byte  71,225,61                           // rex.RXB       loope 4b2d <.literal16+0x2dd>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -31328,13 +31376,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4cd9 <.literal16+0x4a9>
+  .byte  224,7                               // loopne        4cf9 <.literal16+0x4a9>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4cdd <.literal16+0x4ad>
+  .byte  224,7                               // loopne        4cfd <.literal16+0x4ad>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4ce1 <.literal16+0x4b1>
+  .byte  224,7                               // loopne        4d01 <.literal16+0x4b1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4ce5 <.literal16+0x4b5>
+  .byte  224,7                               // loopne        4d05 <.literal16+0x4b5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -31358,22 +31406,18 @@ BALIGN16
   .byte  4,61                                // add           $0x3d,%al
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  4,61                                // add           $0x3d,%al
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  128,63,0                            // cmpb          $0x0,(%rdi)
-  .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
-  .byte  63                                  // (bad)
-  .byte  0,0                                 // add           %al,(%rax)
-  .byte  128,63,255                          // cmpb          $0xff,(%rdi)
-  .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,255                               // add           %bh,%bh
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,255                               // add           %bh,%bh
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,255                               // add           %bh,%bh
+  .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  0,129,128,128,59,129                // add           %al,-0x7ec47f80(%rcx)
-  .byte  128,128,59,129,128,128,59           // addb          $0x3b,-0x7f7f7ec5(%rax)
-  .byte  129,128,128,59,255,0,255,0,255,0    // addl          $0xff00ff,0xff3b80(%rax)
+  .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
+  .byte  128,59,129                          // cmpb          $0x81,(%rbx)
+  .byte  128,128,59,255,0,255,0              // addb          $0x0,-0xff00c5(%rax)
+  .byte  255,0                               // incl          (%rax)
   .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,0                                 // add           %al,(%rax)
@@ -31403,11 +31447,11 @@ BALIGN16
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4dcb <.literal16+0x59b>
+  .byte  127,67                              // jg            4ddb <.literal16+0x58b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4dcf <.literal16+0x59f>
+  .byte  127,67                              // jg            4ddf <.literal16+0x58f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4dd3 <.literal16+0x5a3>
+  .byte  127,67                              // jg            4de3 <.literal16+0x593>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,129,128,128,59           // addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -31422,16 +31466,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4dc4 <.literal16+0x594>
+  .byte  127,0                               // jg            4dd4 <.literal16+0x584>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4dc8 <.literal16+0x598>
+  .byte  127,0                               // jg            4dd8 <.literal16+0x588>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4dcc <.literal16+0x59c>
+  .byte  127,0                               // jg            4ddc <.literal16+0x58c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4dd0 <.literal16+0x5a0>
+  .byte  127,0                               // jg            4de0 <.literal16+0x590>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -31440,7 +31484,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4e55 <.literal16+0x625>
+  .byte  119,115                             // ja            4e65 <.literal16+0x615>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -31451,7 +31495,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4db9 <.literal16+0x589>
+  .byte  117,191                             // jne           4dc9 <.literal16+0x579>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -31463,7 +31507,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38dfa <_sk_callback_sse2+0xffffffffe9a346ae>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38e0a <_sk_callback_sse2+0xffffffffe9a34699>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -31517,16 +31561,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e94 <.literal16+0x664>
+  .byte  127,0                               // jg            4ea4 <.literal16+0x654>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e98 <.literal16+0x668>
+  .byte  127,0                               // jg            4ea8 <.literal16+0x658>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e9c <.literal16+0x66c>
+  .byte  127,0                               // jg            4eac <.literal16+0x65c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4ea0 <.literal16+0x670>
+  .byte  127,0                               // jg            4eb0 <.literal16+0x660>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -31535,7 +31579,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4f25 <.literal16+0x6f5>
+  .byte  119,115                             // ja            4f35 <.literal16+0x6e5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -31546,7 +31590,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4e89 <.literal16+0x659>
+  .byte  117,191                             // jne           4e99 <.literal16+0x649>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -31558,7 +31602,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38eca <_sk_callback_sse2+0xffffffffe9a3477e>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38eda <_sk_callback_sse2+0xffffffffe9a34769>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -31612,16 +31656,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f64 <.literal16+0x734>
+  .byte  127,0                               // jg            4f74 <.literal16+0x724>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f68 <.literal16+0x738>
+  .byte  127,0                               // jg            4f78 <.literal16+0x728>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f6c <.literal16+0x73c>
+  .byte  127,0                               // jg            4f7c <.literal16+0x72c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f70 <.literal16+0x740>
+  .byte  127,0                               // jg            4f80 <.literal16+0x730>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -31630,7 +31674,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4ff5 <.literal16+0x7c5>
+  .byte  119,115                             // ja            5005 <.literal16+0x7b5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -31641,7 +31685,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4f59 <.literal16+0x729>
+  .byte  117,191                             // jne           4f69 <.literal16+0x719>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -31653,7 +31697,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38f9a <_sk_callback_sse2+0xffffffffe9a3484e>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38faa <_sk_callback_sse2+0xffffffffe9a34839>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -31707,16 +31751,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5034 <.literal16+0x804>
+  .byte  127,0                               // jg            5044 <.literal16+0x7f4>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5038 <.literal16+0x808>
+  .byte  127,0                               // jg            5048 <.literal16+0x7f8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            503c <.literal16+0x80c>
+  .byte  127,0                               // jg            504c <.literal16+0x7fc>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5040 <.literal16+0x810>
+  .byte  127,0                               // jg            5050 <.literal16+0x800>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -31725,7 +31769,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            50c5 <.literal16+0x895>
+  .byte  119,115                             // ja            50d5 <.literal16+0x885>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -31736,7 +31780,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           5029 <.literal16+0x7f9>
+  .byte  117,191                             // jne           5039 <.literal16+0x7e9>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -31748,7 +31792,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3906a <_sk_callback_sse2+0xffffffffe9a3491e>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3907a <_sk_callback_sse2+0xffffffffe9a34909>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -31798,13 +31842,13 @@ BALIGN16
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
-  .byte  127,67                              // jg            5147 <.literal16+0x917>
+  .byte  127,67                              // jg            5157 <.literal16+0x907>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            514b <.literal16+0x91b>
+  .byte  127,67                              // jg            515b <.literal16+0x90b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            514f <.literal16+0x91f>
+  .byte  127,67                              // jg            515f <.literal16+0x90f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5153 <.literal16+0x923>
+  .byte  127,67                              // jg            5163 <.literal16+0x913>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -31851,16 +31895,16 @@ BALIGN16
   .byte  128,3,62                            // addb          $0x3e,(%rbx)
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           51d3 <.literal16+0x9a3>
+  .byte  118,63                              // jbe           51e3 <.literal16+0x993>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           51d7 <.literal16+0x9a7>
+  .byte  118,63                              // jbe           51e7 <.literal16+0x997>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           51db <.literal16+0x9ab>
+  .byte  118,63                              // jbe           51eb <.literal16+0x99b>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           51df <.literal16+0x9af>
+  .byte  118,63                              // jbe           51ef <.literal16+0x99f>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
@@ -31872,11 +31916,11 @@ BALIGN16
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            521b <.literal16+0x9eb>
+  .byte  127,67                              // jg            522b <.literal16+0x9db>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            521f <.literal16+0x9ef>
+  .byte  127,67                              // jg            522f <.literal16+0x9df>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5223 <.literal16+0x9f3>
+  .byte  127,67                              // jg            5233 <.literal16+0x9e3>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,0,0,128,63               // addb          $0x3f,-0x7fffffc5(%rax)
@@ -31916,13 +31960,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5269 <.literal16+0xa39>
+  .byte  224,7                               // loopne        5279 <.literal16+0xa29>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        526d <.literal16+0xa3d>
+  .byte  224,7                               // loopne        527d <.literal16+0xa2d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5271 <.literal16+0xa41>
+  .byte  224,7                               // loopne        5281 <.literal16+0xa31>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5275 <.literal16+0xa45>
+  .byte  224,7                               // loopne        5285 <.literal16+0xa35>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -31968,13 +32012,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        52d9 <.literal16+0xaa9>
+  .byte  224,7                               // loopne        52e9 <.literal16+0xa99>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        52dd <.literal16+0xaad>
+  .byte  224,7                               // loopne        52ed <.literal16+0xa9d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        52e1 <.literal16+0xab1>
+  .byte  224,7                               // loopne        52f1 <.literal16+0xaa1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        52e5 <.literal16+0xab5>
+  .byte  224,7                               // loopne        52f5 <.literal16+0xaa5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -32012,13 +32056,13 @@ BALIGN16
   .byte  65,0,0                              // add           %al,(%r8)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            5376 <.literal16+0xb46>
+  .byte  124,66                              // jl            5386 <.literal16+0xb36>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            537a <.literal16+0xb4a>
+  .byte  124,66                              // jl            538a <.literal16+0xb3a>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            537e <.literal16+0xb4e>
+  .byte  124,66                              // jl            538e <.literal16+0xb3e>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            5382 <.literal16+0xb52>
+  .byte  124,66                              // jl            5392 <.literal16+0xb42>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,240                               // add           %dh,%al
@@ -32108,13 +32152,13 @@ BALIGN16
   .byte  136,136,61,137,136,136              // mov           %cl,-0x777776c3(%rax)
   .byte  61,137,136,136,61                   // cmp           $0x3d888889,%eax
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5485 <.literal16+0xc55>
+  .byte  112,65                              // jo            5495 <.literal16+0xc45>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5489 <.literal16+0xc59>
+  .byte  112,65                              // jo            5499 <.literal16+0xc49>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            548d <.literal16+0xc5d>
+  .byte  112,65                              // jo            549d <.literal16+0xc4d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5491 <.literal16+0xc61>
+  .byte  112,65                              // jo            54a1 <.literal16+0xc51>
   .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  255,0                               // incl          (%rax)
@@ -32136,11 +32180,11 @@ BALIGN16
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,0,0,127,67               // addb          $0x43,0x7f00003b(%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            54db <.literal16+0xcab>
+  .byte  127,67                              // jg            54eb <.literal16+0xc9b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            54df <.literal16+0xcaf>
+  .byte  127,67                              // jg            54ef <.literal16+0xc9f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            54e3 <.literal16+0xcb3>
+  .byte  127,67                              // jg            54f3 <.literal16+0xca3>
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
@@ -32216,13 +32260,13 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  255                                 // (bad)
-  .byte  127,71                              // jg            55cb <.literal16+0xd9b>
+  .byte  127,71                              // jg            55db <.literal16+0xd8b>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            55cf <.literal16+0xd9f>
+  .byte  127,71                              // jg            55df <.literal16+0xd8f>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            55d3 <.literal16+0xda3>
+  .byte  127,71                              // jg            55e3 <.literal16+0xd93>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            55d7 <.literal16+0xda7>
+  .byte  127,71                              // jg            55e7 <.literal16+0xd97>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -32375,11 +32419,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5732 <.literal16+0xf02>
+  .byte  62,114,28                           // jb,pt         5742 <.literal16+0xef2>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5736 <.literal16+0xf06>
+  .byte  62,114,28                           // jb,pt         5746 <.literal16+0xef6>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         573a <.literal16+0xf0a>
+  .byte  62,114,28                           // jb,pt         574a <.literal16+0xefa>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -32423,7 +32467,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e5c5 <_sk_callback_sse2+0x3d639e79>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e5d5 <_sk_callback_sse2+0x3d639e64>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -32449,7 +32493,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e605 <_sk_callback_sse2+0x3d639eb9>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e615 <_sk_callback_sse2+0x3d639ea4>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -32458,13 +32502,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            57fe <.literal16+0xfce>
+  .byte  114,28                              // jb            580e <.literal16+0xfbe>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5802 <.literal16+0xfd2>
+  .byte  62,114,28                           // jb,pt         5812 <.literal16+0xfc2>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5806 <.literal16+0xfd6>
+  .byte  62,114,28                           // jb,pt         5816 <.literal16+0xfc6>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         580a <.literal16+0xfda>
+  .byte  62,114,28                           // jb,pt         581a <.literal16+0xfca>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -32485,11 +32529,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5842 <.literal16+0x1012>
+  .byte  62,114,28                           // jb,pt         5852 <.literal16+0x1002>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5846 <.literal16+0x1016>
+  .byte  62,114,28                           // jb,pt         5856 <.literal16+0x1006>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         584a <.literal16+0x101a>
+  .byte  62,114,28                           // jb,pt         585a <.literal16+0x100a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -32533,7 +32577,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e6d5 <_sk_callback_sse2+0x3d639f89>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e6e5 <_sk_callback_sse2+0x3d639f74>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -32559,7 +32603,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e715 <_sk_callback_sse2+0x3d639fc9>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e725 <_sk_callback_sse2+0x3d639fb4>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -32568,13 +32612,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            590e <.literal16+0x10de>
+  .byte  114,28                              // jb            591e <.literal16+0x10ce>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5912 <_sk_callback_sse2+0x11c6>
+  .byte  62,114,28                           // jb,pt         5922 <_sk_callback_sse2+0x11b1>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5916 <_sk_callback_sse2+0x11ca>
+  .byte  62,114,28                           // jb,pt         5926 <_sk_callback_sse2+0x11b5>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         591a <_sk_callback_sse2+0x11ce>
+  .byte  62,114,28                           // jb,pt         592a <_sk_callback_sse2+0x11b9>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
index c0e6cf5..23d929b 100644 (file)
@@ -106,14 +106,14 @@ _sk_seed_shader_hsw LABEL PROC
   DB  197,249,110,199                     ; vmovd         %edi,%xmm0
   DB  196,226,125,88,192                  ; vpbroadcastd  %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,66,68,0,0         ; vbroadcastss  0x4442(%rip),%ymm1        # 459c <_sk_callback_hsw+0x119>
+  DB  196,226,125,24,13,86,68,0,0         ; vbroadcastss  0x4456(%rip),%ymm1        # 45b0 <_sk_callback_hsw+0x119>
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
   DB  197,252,88,2                        ; vaddps        (%rdx),%ymm0,%ymm0
   DB  196,226,125,24,16                   ; vbroadcastss  (%rax),%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  197,236,88,201                      ; vaddps        %ymm1,%ymm2,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,21,38,68,0,0         ; vbroadcastss  0x4426(%rip),%ymm2        # 45a0 <_sk_callback_hsw+0x11d>
+  DB  196,226,125,24,21,58,68,0,0         ; vbroadcastss  0x443a(%rip),%ymm2        # 45b4 <_sk_callback_hsw+0x11d>
   DB  197,228,87,219                      ; vxorps        %ymm3,%ymm3,%ymm3
   DB  197,220,87,228                      ; vxorps        %ymm4,%ymm4,%ymm4
   DB  197,212,87,237                      ; vxorps        %ymm5,%ymm5,%ymm5
@@ -132,13 +132,13 @@ _sk_dither_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  196,66,125,88,8                     ; vpbroadcastd  (%r8),%ymm9
   DB  196,65,61,239,201                   ; vpxor         %ymm9,%ymm8,%ymm9
-  DB  196,98,125,88,21,229,67,0,0         ; vpbroadcastd  0x43e5(%rip),%ymm10        # 45a4 <_sk_callback_hsw+0x121>
+  DB  196,98,125,88,21,249,67,0,0         ; vpbroadcastd  0x43f9(%rip),%ymm10        # 45b8 <_sk_callback_hsw+0x121>
   DB  196,65,53,219,218                   ; vpand         %ymm10,%ymm9,%ymm11
   DB  196,193,37,114,243,5                ; vpslld        $0x5,%ymm11,%ymm11
   DB  196,65,61,219,210                   ; vpand         %ymm10,%ymm8,%ymm10
   DB  196,193,45,114,242,4                ; vpslld        $0x4,%ymm10,%ymm10
-  DB  196,98,125,88,37,202,67,0,0         ; vpbroadcastd  0x43ca(%rip),%ymm12        # 45a8 <_sk_callback_hsw+0x125>
-  DB  196,98,125,88,45,197,67,0,0         ; vpbroadcastd  0x43c5(%rip),%ymm13        # 45ac <_sk_callback_hsw+0x129>
+  DB  196,98,125,88,37,222,67,0,0         ; vpbroadcastd  0x43de(%rip),%ymm12        # 45bc <_sk_callback_hsw+0x125>
+  DB  196,98,125,88,45,217,67,0,0         ; vpbroadcastd  0x43d9(%rip),%ymm13        # 45c0 <_sk_callback_hsw+0x129>
   DB  196,65,53,219,245                   ; vpand         %ymm13,%ymm9,%ymm14
   DB  196,193,13,114,246,2                ; vpslld        $0x2,%ymm14,%ymm14
   DB  196,65,61,219,237                   ; vpand         %ymm13,%ymm8,%ymm13
@@ -153,8 +153,8 @@ _sk_dither_hsw LABEL PROC
   DB  196,65,61,235,194                   ; vpor          %ymm10,%ymm8,%ymm8
   DB  196,65,61,235,193                   ; vpor          %ymm9,%ymm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,119,67,0,0         ; vbroadcastss  0x4377(%rip),%ymm9        # 45b0 <_sk_callback_hsw+0x12d>
-  DB  196,98,125,24,21,114,67,0,0         ; vbroadcastss  0x4372(%rip),%ymm10        # 45b4 <_sk_callback_hsw+0x131>
+  DB  196,98,125,24,13,139,67,0,0         ; vbroadcastss  0x438b(%rip),%ymm9        # 45c4 <_sk_callback_hsw+0x12d>
+  DB  196,98,125,24,21,134,67,0,0         ; vbroadcastss  0x4386(%rip),%ymm10        # 45c8 <_sk_callback_hsw+0x131>
   DB  196,66,61,184,209                   ; vfmadd231ps   %ymm9,%ymm8,%ymm10
   DB  196,98,125,24,64,8                  ; vbroadcastss  0x8(%rax),%ymm8
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
@@ -186,7 +186,7 @@ _sk_clear_hsw LABEL PROC
 PUBLIC _sk_srcatop_hsw
 _sk_srcatop_hsw LABEL PROC
   DB  197,252,89,199                      ; vmulps        %ymm7,%ymm0,%ymm0
-  DB  196,98,125,24,5,24,67,0,0           ; vbroadcastss  0x4318(%rip),%ymm8        # 45b8 <_sk_callback_hsw+0x135>
+  DB  196,98,125,24,5,44,67,0,0           ; vbroadcastss  0x432c(%rip),%ymm8        # 45cc <_sk_callback_hsw+0x135>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,226,61,184,196                  ; vfmadd231ps   %ymm4,%ymm8,%ymm0
   DB  197,244,89,207                      ; vmulps        %ymm7,%ymm1,%ymm1
@@ -200,7 +200,7 @@ _sk_srcatop_hsw LABEL PROC
 
 PUBLIC _sk_dstatop_hsw
 _sk_dstatop_hsw LABEL PROC
-  DB  196,98,125,24,5,235,66,0,0          ; vbroadcastss  0x42eb(%rip),%ymm8        # 45bc <_sk_callback_hsw+0x139>
+  DB  196,98,125,24,5,255,66,0,0          ; vbroadcastss  0x42ff(%rip),%ymm8        # 45d0 <_sk_callback_hsw+0x139>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  196,226,101,184,196                 ; vfmadd231ps   %ymm4,%ymm3,%ymm0
@@ -233,7 +233,7 @@ _sk_dstin_hsw LABEL PROC
 
 PUBLIC _sk_srcout_hsw
 _sk_srcout_hsw LABEL PROC
-  DB  196,98,125,24,5,146,66,0,0          ; vbroadcastss  0x4292(%rip),%ymm8        # 45c0 <_sk_callback_hsw+0x13d>
+  DB  196,98,125,24,5,166,66,0,0          ; vbroadcastss  0x42a6(%rip),%ymm8        # 45d4 <_sk_callback_hsw+0x13d>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -244,7 +244,7 @@ _sk_srcout_hsw LABEL PROC
 
 PUBLIC _sk_dstout_hsw
 _sk_dstout_hsw LABEL PROC
-  DB  196,226,125,24,5,117,66,0,0         ; vbroadcastss  0x4275(%rip),%ymm0        # 45c4 <_sk_callback_hsw+0x141>
+  DB  196,226,125,24,5,137,66,0,0         ; vbroadcastss  0x4289(%rip),%ymm0        # 45d8 <_sk_callback_hsw+0x141>
   DB  197,252,92,219                      ; vsubps        %ymm3,%ymm0,%ymm3
   DB  197,228,89,196                      ; vmulps        %ymm4,%ymm3,%ymm0
   DB  197,228,89,205                      ; vmulps        %ymm5,%ymm3,%ymm1
@@ -255,7 +255,7 @@ _sk_dstout_hsw LABEL PROC
 
 PUBLIC _sk_srcover_hsw
 _sk_srcover_hsw LABEL PROC
-  DB  196,98,125,24,5,88,66,0,0           ; vbroadcastss  0x4258(%rip),%ymm8        # 45c8 <_sk_callback_hsw+0x145>
+  DB  196,98,125,24,5,108,66,0,0          ; vbroadcastss  0x426c(%rip),%ymm8        # 45dc <_sk_callback_hsw+0x145>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,93,184,192                  ; vfmadd231ps   %ymm8,%ymm4,%ymm0
   DB  196,194,85,184,200                  ; vfmadd231ps   %ymm8,%ymm5,%ymm1
@@ -266,7 +266,7 @@ _sk_srcover_hsw LABEL PROC
 
 PUBLIC _sk_dstover_hsw
 _sk_dstover_hsw LABEL PROC
-  DB  196,98,125,24,5,55,66,0,0           ; vbroadcastss  0x4237(%rip),%ymm8        # 45cc <_sk_callback_hsw+0x149>
+  DB  196,98,125,24,5,75,66,0,0           ; vbroadcastss  0x424b(%rip),%ymm8        # 45e0 <_sk_callback_hsw+0x149>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  196,226,61,168,196                  ; vfmadd213ps   %ymm4,%ymm8,%ymm0
   DB  196,226,61,168,205                  ; vfmadd213ps   %ymm5,%ymm8,%ymm1
@@ -286,7 +286,7 @@ _sk_modulate_hsw LABEL PROC
 
 PUBLIC _sk_multiply_hsw
 _sk_multiply_hsw LABEL PROC
-  DB  196,98,125,24,5,2,66,0,0            ; vbroadcastss  0x4202(%rip),%ymm8        # 45d0 <_sk_callback_hsw+0x14d>
+  DB  196,98,125,24,5,22,66,0,0           ; vbroadcastss  0x4216(%rip),%ymm8        # 45e4 <_sk_callback_hsw+0x14d>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,208                       ; vmulps        %ymm0,%ymm9,%ymm10
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -328,7 +328,7 @@ _sk_screen_hsw LABEL PROC
 
 PUBLIC _sk_xor__hsw
 _sk_xor__hsw LABEL PROC
-  DB  196,98,125,24,5,125,65,0,0          ; vbroadcastss  0x417d(%rip),%ymm8        # 45d4 <_sk_callback_hsw+0x151>
+  DB  196,98,125,24,5,145,65,0,0          ; vbroadcastss  0x4191(%rip),%ymm8        # 45e8 <_sk_callback_hsw+0x151>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -360,7 +360,7 @@ _sk_darken_hsw LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,95,209                  ; vmaxps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,5,65,0,0            ; vbroadcastss  0x4105(%rip),%ymm8        # 45d8 <_sk_callback_hsw+0x155>
+  DB  196,98,125,24,5,25,65,0,0           ; vbroadcastss  0x4119(%rip),%ymm8        # 45ec <_sk_callback_hsw+0x155>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -383,7 +383,7 @@ _sk_lighten_hsw LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,180,64,0,0          ; vbroadcastss  0x40b4(%rip),%ymm8        # 45dc <_sk_callback_hsw+0x159>
+  DB  196,98,125,24,5,200,64,0,0          ; vbroadcastss  0x40c8(%rip),%ymm8        # 45f0 <_sk_callback_hsw+0x159>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -409,7 +409,7 @@ _sk_difference_hsw LABEL PROC
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,87,64,0,0           ; vbroadcastss  0x4057(%rip),%ymm8        # 45e0 <_sk_callback_hsw+0x15d>
+  DB  196,98,125,24,5,107,64,0,0          ; vbroadcastss  0x406b(%rip),%ymm8        # 45f4 <_sk_callback_hsw+0x15d>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -429,7 +429,7 @@ _sk_exclusion_hsw LABEL PROC
   DB  197,236,89,214                      ; vmulps        %ymm6,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,21,64,0,0           ; vbroadcastss  0x4015(%rip),%ymm8        # 45e4 <_sk_callback_hsw+0x161>
+  DB  196,98,125,24,5,41,64,0,0           ; vbroadcastss  0x4029(%rip),%ymm8        # 45f8 <_sk_callback_hsw+0x161>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -437,7 +437,7 @@ _sk_exclusion_hsw LABEL PROC
 
 PUBLIC _sk_colorburn_hsw
 _sk_colorburn_hsw LABEL PROC
-  DB  196,98,125,24,5,3,64,0,0            ; vbroadcastss  0x4003(%rip),%ymm8        # 45e8 <_sk_callback_hsw+0x165>
+  DB  196,98,125,24,5,23,64,0,0           ; vbroadcastss  0x4017(%rip),%ymm8        # 45fc <_sk_callback_hsw+0x165>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,216                       ; vmulps        %ymm0,%ymm9,%ymm11
   DB  196,65,44,87,210                    ; vxorps        %ymm10,%ymm10,%ymm10
@@ -493,7 +493,7 @@ _sk_colorburn_hsw LABEL PROC
 PUBLIC _sk_colordodge_hsw
 _sk_colordodge_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
-  DB  196,98,125,24,13,14,63,0,0          ; vbroadcastss  0x3f0e(%rip),%ymm9        # 45ec <_sk_callback_hsw+0x169>
+  DB  196,98,125,24,13,34,63,0,0          ; vbroadcastss  0x3f22(%rip),%ymm9        # 4600 <_sk_callback_hsw+0x169>
   DB  197,52,92,215                       ; vsubps        %ymm7,%ymm9,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,52,92,203                       ; vsubps        %ymm3,%ymm9,%ymm9
@@ -544,7 +544,7 @@ _sk_colordodge_hsw LABEL PROC
 
 PUBLIC _sk_hardlight_hsw
 _sk_hardlight_hsw LABEL PROC
-  DB  196,98,125,24,5,47,62,0,0           ; vbroadcastss  0x3e2f(%rip),%ymm8        # 45f0 <_sk_callback_hsw+0x16d>
+  DB  196,98,125,24,5,67,62,0,0           ; vbroadcastss  0x3e43(%rip),%ymm8        # 4604 <_sk_callback_hsw+0x16d>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -593,7 +593,7 @@ _sk_hardlight_hsw LABEL PROC
 
 PUBLIC _sk_overlay_hsw
 _sk_overlay_hsw LABEL PROC
-  DB  196,98,125,24,5,103,61,0,0          ; vbroadcastss  0x3d67(%rip),%ymm8        # 45f4 <_sk_callback_hsw+0x171>
+  DB  196,98,125,24,5,123,61,0,0          ; vbroadcastss  0x3d7b(%rip),%ymm8        # 4608 <_sk_callback_hsw+0x171>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -653,10 +653,10 @@ _sk_softlight_hsw LABEL PROC
   DB  196,65,20,88,197                    ; vaddps        %ymm13,%ymm13,%ymm8
   DB  196,65,60,88,192                    ; vaddps        %ymm8,%ymm8,%ymm8
   DB  196,66,61,168,192                   ; vfmadd213ps   %ymm8,%ymm8,%ymm8
-  DB  196,98,125,24,29,110,60,0,0         ; vbroadcastss  0x3c6e(%rip),%ymm11        # 45fc <_sk_callback_hsw+0x179>
+  DB  196,98,125,24,29,130,60,0,0         ; vbroadcastss  0x3c82(%rip),%ymm11        # 4610 <_sk_callback_hsw+0x179>
   DB  196,65,20,88,227                    ; vaddps        %ymm11,%ymm13,%ymm12
   DB  196,65,28,89,192                    ; vmulps        %ymm8,%ymm12,%ymm8
-  DB  196,98,125,24,37,95,60,0,0          ; vbroadcastss  0x3c5f(%rip),%ymm12        # 4600 <_sk_callback_hsw+0x17d>
+  DB  196,98,125,24,37,115,60,0,0         ; vbroadcastss  0x3c73(%rip),%ymm12        # 4614 <_sk_callback_hsw+0x17d>
   DB  196,66,21,184,196                   ; vfmadd231ps   %ymm12,%ymm13,%ymm8
   DB  196,65,124,82,245                   ; vrsqrtps      %ymm13,%ymm14
   DB  196,65,124,83,246                   ; vrcpps        %ymm14,%ymm14
@@ -666,7 +666,7 @@ _sk_softlight_hsw LABEL PROC
   DB  197,4,194,255,2                     ; vcmpleps      %ymm7,%ymm15,%ymm15
   DB  196,67,13,74,240,240                ; vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   DB  197,116,88,249                      ; vaddps        %ymm1,%ymm1,%ymm15
-  DB  196,98,125,24,5,34,60,0,0           ; vbroadcastss  0x3c22(%rip),%ymm8        # 45f8 <_sk_callback_hsw+0x175>
+  DB  196,98,125,24,5,54,60,0,0           ; vbroadcastss  0x3c36(%rip),%ymm8        # 460c <_sk_callback_hsw+0x175>
   DB  196,65,60,92,237                    ; vsubps        %ymm13,%ymm8,%ymm13
   DB  197,132,92,195                      ; vsubps        %ymm3,%ymm15,%ymm0
   DB  196,98,125,168,235                  ; vfmadd213ps   %ymm3,%ymm0,%ymm13
@@ -748,7 +748,7 @@ PUBLIC _sk_hue_hsw
 _sk_hue_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,208,0                ; vcmpeqps      %ymm8,%ymm3,%ymm10
-  DB  196,98,125,24,13,183,58,0,0         ; vbroadcastss  0x3ab7(%rip),%ymm9        # 4604 <_sk_callback_hsw+0x181>
+  DB  196,98,125,24,13,203,58,0,0         ; vbroadcastss  0x3acb(%rip),%ymm9        # 4618 <_sk_callback_hsw+0x181>
   DB  197,52,94,219                       ; vdivps        %ymm3,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,172,89,192                      ; vmulps        %ymm0,%ymm10,%ymm0
@@ -777,11 +777,11 @@ _sk_hue_hsw LABEL PROC
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  196,193,108,94,212                  ; vdivps        %ymm12,%ymm2,%ymm2
   DB  196,195,109,74,208,208              ; vblendvps     %ymm13,%ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,21,44,58,0,0          ; vbroadcastss  0x3a2c(%rip),%ymm10        # 4608 <_sk_callback_hsw+0x185>
-  DB  196,98,125,24,29,39,58,0,0          ; vbroadcastss  0x3a27(%rip),%ymm11        # 460c <_sk_callback_hsw+0x189>
+  DB  196,98,125,24,21,64,58,0,0          ; vbroadcastss  0x3a40(%rip),%ymm10        # 461c <_sk_callback_hsw+0x185>
+  DB  196,98,125,24,29,59,58,0,0          ; vbroadcastss  0x3a3b(%rip),%ymm11        # 4620 <_sk_callback_hsw+0x189>
   DB  196,65,84,89,227                    ; vmulps        %ymm11,%ymm5,%ymm12
   DB  196,66,93,184,226                   ; vfmadd231ps   %ymm10,%ymm4,%ymm12
-  DB  196,98,125,24,45,24,58,0,0          ; vbroadcastss  0x3a18(%rip),%ymm13        # 4610 <_sk_callback_hsw+0x18d>
+  DB  196,98,125,24,45,44,58,0,0          ; vbroadcastss  0x3a2c(%rip),%ymm13        # 4624 <_sk_callback_hsw+0x18d>
   DB  196,66,77,184,229                   ; vfmadd231ps   %ymm13,%ymm6,%ymm12
   DB  196,65,116,89,243                   ; vmulps        %ymm11,%ymm1,%ymm14
   DB  196,66,125,184,242                  ; vfmadd231ps   %ymm10,%ymm0,%ymm14
@@ -847,7 +847,7 @@ PUBLIC _sk_saturation_hsw
 _sk_saturation_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,68,194,208,0                 ; vcmpeqps      %ymm8,%ymm7,%ymm10
-  DB  196,98,125,24,13,240,56,0,0         ; vbroadcastss  0x38f0(%rip),%ymm9        # 4614 <_sk_callback_hsw+0x191>
+  DB  196,98,125,24,13,4,57,0,0           ; vbroadcastss  0x3904(%rip),%ymm9        # 4628 <_sk_callback_hsw+0x191>
   DB  197,52,94,223                       ; vdivps        %ymm7,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,44,89,220                       ; vmulps        %ymm4,%ymm10,%ymm11
@@ -876,11 +876,11 @@ _sk_saturation_hsw LABEL PROC
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  197,252,94,194                      ; vdivps        %ymm2,%ymm0,%ymm0
   DB  196,195,125,74,192,208              ; vblendvps     %ymm13,%ymm8,%ymm0,%ymm0
-  DB  196,226,125,24,21,108,56,0,0        ; vbroadcastss  0x386c(%rip),%ymm2        # 4618 <_sk_callback_hsw+0x195>
-  DB  196,226,125,24,13,103,56,0,0        ; vbroadcastss  0x3867(%rip),%ymm1        # 461c <_sk_callback_hsw+0x199>
+  DB  196,226,125,24,21,128,56,0,0        ; vbroadcastss  0x3880(%rip),%ymm2        # 462c <_sk_callback_hsw+0x195>
+  DB  196,226,125,24,13,123,56,0,0        ; vbroadcastss  0x387b(%rip),%ymm1        # 4630 <_sk_callback_hsw+0x199>
   DB  197,84,89,209                       ; vmulps        %ymm1,%ymm5,%ymm10
   DB  196,98,93,184,210                   ; vfmadd231ps   %ymm2,%ymm4,%ymm10
-  DB  196,98,125,24,45,89,56,0,0          ; vbroadcastss  0x3859(%rip),%ymm13        # 4620 <_sk_callback_hsw+0x19d>
+  DB  196,98,125,24,45,109,56,0,0         ; vbroadcastss  0x386d(%rip),%ymm13        # 4634 <_sk_callback_hsw+0x19d>
   DB  196,66,77,184,213                   ; vfmadd231ps   %ymm13,%ymm6,%ymm10
   DB  197,28,89,241                       ; vmulps        %ymm1,%ymm12,%ymm14
   DB  196,98,37,184,242                   ; vfmadd231ps   %ymm2,%ymm11,%ymm14
@@ -946,17 +946,17 @@ PUBLIC _sk_color_hsw
 _sk_color_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,208,0                ; vcmpeqps      %ymm8,%ymm3,%ymm10
-  DB  196,98,125,24,13,43,55,0,0          ; vbroadcastss  0x372b(%rip),%ymm9        # 4624 <_sk_callback_hsw+0x1a1>
+  DB  196,98,125,24,13,63,55,0,0          ; vbroadcastss  0x373f(%rip),%ymm9        # 4638 <_sk_callback_hsw+0x1a1>
   DB  197,52,94,219                       ; vdivps        %ymm3,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,172,89,192                      ; vmulps        %ymm0,%ymm10,%ymm0
   DB  197,172,89,201                      ; vmulps        %ymm1,%ymm10,%ymm1
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
-  DB  196,98,125,24,21,16,55,0,0          ; vbroadcastss  0x3710(%rip),%ymm10        # 4628 <_sk_callback_hsw+0x1a5>
-  DB  196,98,125,24,29,11,55,0,0          ; vbroadcastss  0x370b(%rip),%ymm11        # 462c <_sk_callback_hsw+0x1a9>
+  DB  196,98,125,24,21,36,55,0,0          ; vbroadcastss  0x3724(%rip),%ymm10        # 463c <_sk_callback_hsw+0x1a5>
+  DB  196,98,125,24,29,31,55,0,0          ; vbroadcastss  0x371f(%rip),%ymm11        # 4640 <_sk_callback_hsw+0x1a9>
   DB  196,65,84,89,227                    ; vmulps        %ymm11,%ymm5,%ymm12
   DB  196,66,93,184,226                   ; vfmadd231ps   %ymm10,%ymm4,%ymm12
-  DB  196,98,125,24,45,252,54,0,0         ; vbroadcastss  0x36fc(%rip),%ymm13        # 4630 <_sk_callback_hsw+0x1ad>
+  DB  196,98,125,24,45,16,55,0,0          ; vbroadcastss  0x3710(%rip),%ymm13        # 4644 <_sk_callback_hsw+0x1ad>
   DB  196,66,77,184,229                   ; vfmadd231ps   %ymm13,%ymm6,%ymm12
   DB  196,65,116,89,243                   ; vmulps        %ymm11,%ymm1,%ymm14
   DB  196,66,125,184,242                  ; vfmadd231ps   %ymm10,%ymm0,%ymm14
@@ -1022,17 +1022,17 @@ PUBLIC _sk_luminosity_hsw
 _sk_luminosity_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,68,194,208,0                 ; vcmpeqps      %ymm8,%ymm7,%ymm10
-  DB  196,98,125,24,13,212,53,0,0         ; vbroadcastss  0x35d4(%rip),%ymm9        # 4634 <_sk_callback_hsw+0x1b1>
+  DB  196,98,125,24,13,232,53,0,0         ; vbroadcastss  0x35e8(%rip),%ymm9        # 4648 <_sk_callback_hsw+0x1b1>
   DB  197,52,94,223                       ; vdivps        %ymm7,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,44,89,220                       ; vmulps        %ymm4,%ymm10,%ymm11
   DB  197,44,89,229                       ; vmulps        %ymm5,%ymm10,%ymm12
   DB  197,44,89,214                       ; vmulps        %ymm6,%ymm10,%ymm10
-  DB  196,98,125,24,45,185,53,0,0         ; vbroadcastss  0x35b9(%rip),%ymm13        # 4638 <_sk_callback_hsw+0x1b5>
-  DB  196,98,125,24,53,180,53,0,0         ; vbroadcastss  0x35b4(%rip),%ymm14        # 463c <_sk_callback_hsw+0x1b9>
+  DB  196,98,125,24,45,205,53,0,0         ; vbroadcastss  0x35cd(%rip),%ymm13        # 464c <_sk_callback_hsw+0x1b5>
+  DB  196,98,125,24,53,200,53,0,0         ; vbroadcastss  0x35c8(%rip),%ymm14        # 4650 <_sk_callback_hsw+0x1b9>
   DB  196,193,116,89,206                  ; vmulps        %ymm14,%ymm1,%ymm1
   DB  196,226,21,168,193                  ; vfmadd213ps   %ymm1,%ymm13,%ymm0
-  DB  196,98,125,24,61,165,53,0,0         ; vbroadcastss  0x35a5(%rip),%ymm15        # 4640 <_sk_callback_hsw+0x1bd>
+  DB  196,98,125,24,61,185,53,0,0         ; vbroadcastss  0x35b9(%rip),%ymm15        # 4654 <_sk_callback_hsw+0x1bd>
   DB  196,226,5,168,208                   ; vfmadd213ps   %ymm0,%ymm15,%ymm2
   DB  196,193,28,89,198                   ; vmulps        %ymm14,%ymm12,%ymm0
   DB  196,194,37,184,197                  ; vfmadd231ps   %ymm13,%ymm11,%ymm0
@@ -1106,7 +1106,7 @@ _sk_clamp_0_hsw LABEL PROC
 
 PUBLIC _sk_clamp_1_hsw
 _sk_clamp_1_hsw LABEL PROC
-  DB  196,98,125,24,5,103,52,0,0          ; vbroadcastss  0x3467(%rip),%ymm8        # 4644 <_sk_callback_hsw+0x1c1>
+  DB  196,98,125,24,5,123,52,0,0          ; vbroadcastss  0x347b(%rip),%ymm8        # 4658 <_sk_callback_hsw+0x1c1>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
@@ -1116,7 +1116,7 @@ _sk_clamp_1_hsw LABEL PROC
 
 PUBLIC _sk_clamp_a_hsw
 _sk_clamp_a_hsw LABEL PROC
-  DB  196,98,125,24,5,74,52,0,0           ; vbroadcastss  0x344a(%rip),%ymm8        # 4648 <_sk_callback_hsw+0x1c5>
+  DB  196,98,125,24,5,94,52,0,0           ; vbroadcastss  0x345e(%rip),%ymm8        # 465c <_sk_callback_hsw+0x1c5>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  197,252,93,195                      ; vminps        %ymm3,%ymm0,%ymm0
   DB  197,244,93,203                      ; vminps        %ymm3,%ymm1,%ymm1
@@ -1188,7 +1188,7 @@ PUBLIC _sk_unpremul_hsw
 _sk_unpremul_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,200,0                ; vcmpeqps      %ymm8,%ymm3,%ymm9
-  DB  196,98,125,24,21,146,51,0,0         ; vbroadcastss  0x3392(%rip),%ymm10        # 464c <_sk_callback_hsw+0x1c9>
+  DB  196,98,125,24,21,166,51,0,0         ; vbroadcastss  0x33a6(%rip),%ymm10        # 4660 <_sk_callback_hsw+0x1c9>
   DB  197,44,94,211                       ; vdivps        %ymm3,%ymm10,%ymm10
   DB  196,67,45,74,192,144                ; vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
@@ -1199,16 +1199,16 @@ _sk_unpremul_hsw LABEL PROC
 
 PUBLIC _sk_from_srgb_hsw
 _sk_from_srgb_hsw LABEL PROC
-  DB  196,98,125,24,5,115,51,0,0          ; vbroadcastss  0x3373(%rip),%ymm8        # 4650 <_sk_callback_hsw+0x1cd>
+  DB  196,98,125,24,5,135,51,0,0          ; vbroadcastss  0x3387(%rip),%ymm8        # 4664 <_sk_callback_hsw+0x1cd>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  197,124,89,208                      ; vmulps        %ymm0,%ymm0,%ymm10
-  DB  196,98,125,24,29,101,51,0,0         ; vbroadcastss  0x3365(%rip),%ymm11        # 4654 <_sk_callback_hsw+0x1d1>
-  DB  196,98,125,24,37,96,51,0,0          ; vbroadcastss  0x3360(%rip),%ymm12        # 4658 <_sk_callback_hsw+0x1d5>
+  DB  196,98,125,24,29,121,51,0,0         ; vbroadcastss  0x3379(%rip),%ymm11        # 4668 <_sk_callback_hsw+0x1d1>
+  DB  196,98,125,24,37,116,51,0,0         ; vbroadcastss  0x3374(%rip),%ymm12        # 466c <_sk_callback_hsw+0x1d5>
   DB  196,65,124,40,236                   ; vmovaps       %ymm12,%ymm13
   DB  196,66,125,168,235                  ; vfmadd213ps   %ymm11,%ymm0,%ymm13
-  DB  196,98,125,24,53,81,51,0,0          ; vbroadcastss  0x3351(%rip),%ymm14        # 465c <_sk_callback_hsw+0x1d9>
+  DB  196,98,125,24,53,101,51,0,0         ; vbroadcastss  0x3365(%rip),%ymm14        # 4670 <_sk_callback_hsw+0x1d9>
   DB  196,66,45,168,238                   ; vfmadd213ps   %ymm14,%ymm10,%ymm13
-  DB  196,98,125,24,21,71,51,0,0          ; vbroadcastss  0x3347(%rip),%ymm10        # 4660 <_sk_callback_hsw+0x1dd>
+  DB  196,98,125,24,21,91,51,0,0          ; vbroadcastss  0x335b(%rip),%ymm10        # 4674 <_sk_callback_hsw+0x1dd>
   DB  196,193,124,194,194,1               ; vcmpltps      %ymm10,%ymm0,%ymm0
   DB  196,195,21,74,193,0                 ; vblendvps     %ymm0,%ymm9,%ymm13,%ymm0
   DB  196,65,116,89,200                   ; vmulps        %ymm8,%ymm1,%ymm9
@@ -1232,16 +1232,16 @@ _sk_to_srgb_hsw LABEL PROC
   DB  197,124,82,192                      ; vrsqrtps      %ymm0,%ymm8
   DB  196,65,124,83,200                   ; vrcpps        %ymm8,%ymm9
   DB  196,65,124,82,208                   ; vrsqrtps      %ymm8,%ymm10
-  DB  196,98,125,24,5,225,50,0,0          ; vbroadcastss  0x32e1(%rip),%ymm8        # 4664 <_sk_callback_hsw+0x1e1>
+  DB  196,98,125,24,5,245,50,0,0          ; vbroadcastss  0x32f5(%rip),%ymm8        # 4678 <_sk_callback_hsw+0x1e1>
   DB  196,65,124,89,216                   ; vmulps        %ymm8,%ymm0,%ymm11
-  DB  196,98,125,24,37,215,50,0,0         ; vbroadcastss  0x32d7(%rip),%ymm12        # 4668 <_sk_callback_hsw+0x1e5>
-  DB  196,98,125,24,45,210,50,0,0         ; vbroadcastss  0x32d2(%rip),%ymm13        # 466c <_sk_callback_hsw+0x1e9>
+  DB  196,98,125,24,37,235,50,0,0         ; vbroadcastss  0x32eb(%rip),%ymm12        # 467c <_sk_callback_hsw+0x1e5>
+  DB  196,98,125,24,45,230,50,0,0         ; vbroadcastss  0x32e6(%rip),%ymm13        # 4680 <_sk_callback_hsw+0x1e9>
   DB  196,66,21,168,204                   ; vfmadd213ps   %ymm12,%ymm13,%ymm9
-  DB  196,98,125,24,53,200,50,0,0         ; vbroadcastss  0x32c8(%rip),%ymm14        # 4670 <_sk_callback_hsw+0x1ed>
+  DB  196,98,125,24,53,220,50,0,0         ; vbroadcastss  0x32dc(%rip),%ymm14        # 4684 <_sk_callback_hsw+0x1ed>
   DB  196,66,13,184,202                   ; vfmadd231ps   %ymm10,%ymm14,%ymm9
-  DB  196,98,125,24,21,190,50,0,0         ; vbroadcastss  0x32be(%rip),%ymm10        # 4674 <_sk_callback_hsw+0x1f1>
+  DB  196,98,125,24,21,210,50,0,0         ; vbroadcastss  0x32d2(%rip),%ymm10        # 4688 <_sk_callback_hsw+0x1f1>
   DB  196,65,44,93,201                    ; vminps        %ymm9,%ymm10,%ymm9
-  DB  196,98,125,24,61,180,50,0,0         ; vbroadcastss  0x32b4(%rip),%ymm15        # 4678 <_sk_callback_hsw+0x1f5>
+  DB  196,98,125,24,61,200,50,0,0         ; vbroadcastss  0x32c8(%rip),%ymm15        # 468c <_sk_callback_hsw+0x1f5>
   DB  196,193,124,194,199,1               ; vcmpltps      %ymm15,%ymm0,%ymm0
   DB  196,195,53,74,195,0                 ; vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   DB  197,124,82,201                      ; vrsqrtps      %ymm1,%ymm9
@@ -1272,26 +1272,26 @@ _sk_rgb_to_hsl_hsw LABEL PROC
   DB  197,124,93,201                      ; vminps        %ymm1,%ymm0,%ymm9
   DB  197,52,93,202                       ; vminps        %ymm2,%ymm9,%ymm9
   DB  196,65,60,92,209                    ; vsubps        %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,29,46,50,0,0          ; vbroadcastss  0x322e(%rip),%ymm11        # 467c <_sk_callback_hsw+0x1f9>
+  DB  196,98,125,24,29,66,50,0,0          ; vbroadcastss  0x3242(%rip),%ymm11        # 4690 <_sk_callback_hsw+0x1f9>
   DB  196,65,36,94,218                    ; vdivps        %ymm10,%ymm11,%ymm11
   DB  197,116,92,226                      ; vsubps        %ymm2,%ymm1,%ymm12
   DB  197,116,194,234,1                   ; vcmpltps      %ymm2,%ymm1,%ymm13
-  DB  196,98,125,24,53,27,50,0,0          ; vbroadcastss  0x321b(%rip),%ymm14        # 4680 <_sk_callback_hsw+0x1fd>
+  DB  196,98,125,24,53,47,50,0,0          ; vbroadcastss  0x322f(%rip),%ymm14        # 4694 <_sk_callback_hsw+0x1fd>
   DB  196,65,4,87,255                     ; vxorps        %ymm15,%ymm15,%ymm15
   DB  196,67,5,74,238,208                 ; vblendvps     %ymm13,%ymm14,%ymm15,%ymm13
   DB  196,66,37,168,229                   ; vfmadd213ps   %ymm13,%ymm11,%ymm12
   DB  197,236,92,208                      ; vsubps        %ymm0,%ymm2,%ymm2
   DB  197,124,92,233                      ; vsubps        %ymm1,%ymm0,%ymm13
-  DB  196,98,125,24,53,2,50,0,0           ; vbroadcastss  0x3202(%rip),%ymm14        # 4688 <_sk_callback_hsw+0x205>
+  DB  196,98,125,24,53,22,50,0,0          ; vbroadcastss  0x3216(%rip),%ymm14        # 469c <_sk_callback_hsw+0x205>
   DB  196,66,37,168,238                   ; vfmadd213ps   %ymm14,%ymm11,%ymm13
-  DB  196,98,125,24,53,240,49,0,0         ; vbroadcastss  0x31f0(%rip),%ymm14        # 4684 <_sk_callback_hsw+0x201>
+  DB  196,98,125,24,53,4,50,0,0           ; vbroadcastss  0x3204(%rip),%ymm14        # 4698 <_sk_callback_hsw+0x201>
   DB  196,194,37,168,214                  ; vfmadd213ps   %ymm14,%ymm11,%ymm2
   DB  197,188,194,201,0                   ; vcmpeqps      %ymm1,%ymm8,%ymm1
   DB  196,227,21,74,202,16                ; vblendvps     %ymm1,%ymm2,%ymm13,%ymm1
   DB  197,188,194,192,0                   ; vcmpeqps      %ymm0,%ymm8,%ymm0
   DB  196,195,117,74,196,0                ; vblendvps     %ymm0,%ymm12,%ymm1,%ymm0
   DB  196,193,60,88,201                   ; vaddps        %ymm9,%ymm8,%ymm1
-  DB  196,98,125,24,29,211,49,0,0         ; vbroadcastss  0x31d3(%rip),%ymm11        # 4690 <_sk_callback_hsw+0x20d>
+  DB  196,98,125,24,29,231,49,0,0         ; vbroadcastss  0x31e7(%rip),%ymm11        # 46a4 <_sk_callback_hsw+0x20d>
   DB  196,193,116,89,211                  ; vmulps        %ymm11,%ymm1,%ymm2
   DB  197,36,194,218,1                    ; vcmpltps      %ymm2,%ymm11,%ymm11
   DB  196,65,12,92,224                    ; vsubps        %ymm8,%ymm14,%ymm12
@@ -1301,7 +1301,7 @@ _sk_rgb_to_hsl_hsw LABEL PROC
   DB  197,172,94,201                      ; vdivps        %ymm1,%ymm10,%ymm1
   DB  196,195,125,74,199,128              ; vblendvps     %ymm8,%ymm15,%ymm0,%ymm0
   DB  196,195,117,74,207,128              ; vblendvps     %ymm8,%ymm15,%ymm1,%ymm1
-  DB  196,98,125,24,5,150,49,0,0          ; vbroadcastss  0x3196(%rip),%ymm8        # 468c <_sk_callback_hsw+0x209>
+  DB  196,98,125,24,5,170,49,0,0          ; vbroadcastss  0x31aa(%rip),%ymm8        # 46a0 <_sk_callback_hsw+0x209>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -1316,30 +1316,30 @@ _sk_hsl_to_rgb_hsw LABEL PROC
   DB  197,252,17,28,36                    ; vmovups       %ymm3,(%rsp)
   DB  197,252,40,233                      ; vmovaps       %ymm1,%ymm5
   DB  197,252,40,224                      ; vmovaps       %ymm0,%ymm4
-  DB  196,98,125,24,5,93,49,0,0           ; vbroadcastss  0x315d(%rip),%ymm8        # 4694 <_sk_callback_hsw+0x211>
+  DB  196,98,125,24,5,113,49,0,0          ; vbroadcastss  0x3171(%rip),%ymm8        # 46a8 <_sk_callback_hsw+0x211>
   DB  197,60,194,202,2                    ; vcmpleps      %ymm2,%ymm8,%ymm9
   DB  197,84,89,210                       ; vmulps        %ymm2,%ymm5,%ymm10
   DB  196,65,84,92,218                    ; vsubps        %ymm10,%ymm5,%ymm11
   DB  196,67,45,74,203,144                ; vblendvps     %ymm9,%ymm11,%ymm10,%ymm9
   DB  197,52,88,210                       ; vaddps        %ymm2,%ymm9,%ymm10
-  DB  196,98,125,24,13,64,49,0,0          ; vbroadcastss  0x3140(%rip),%ymm9        # 4698 <_sk_callback_hsw+0x215>
+  DB  196,98,125,24,13,84,49,0,0          ; vbroadcastss  0x3154(%rip),%ymm9        # 46ac <_sk_callback_hsw+0x215>
   DB  196,66,109,170,202                  ; vfmsub213ps   %ymm10,%ymm2,%ymm9
-  DB  196,98,125,24,29,54,49,0,0          ; vbroadcastss  0x3136(%rip),%ymm11        # 469c <_sk_callback_hsw+0x219>
+  DB  196,98,125,24,29,74,49,0,0          ; vbroadcastss  0x314a(%rip),%ymm11        # 46b0 <_sk_callback_hsw+0x219>
   DB  196,65,92,88,219                    ; vaddps        %ymm11,%ymm4,%ymm11
   DB  196,67,125,8,227,1                  ; vroundps      $0x1,%ymm11,%ymm12
   DB  196,65,36,92,252                    ; vsubps        %ymm12,%ymm11,%ymm15
   DB  196,65,44,92,217                    ; vsubps        %ymm9,%ymm10,%ymm11
-  DB  196,98,125,24,45,32,49,0,0          ; vbroadcastss  0x3120(%rip),%ymm13        # 46a4 <_sk_callback_hsw+0x221>
+  DB  196,98,125,24,45,52,49,0,0          ; vbroadcastss  0x3134(%rip),%ymm13        # 46b8 <_sk_callback_hsw+0x221>
   DB  196,193,4,89,197                    ; vmulps        %ymm13,%ymm15,%ymm0
-  DB  196,98,125,24,53,22,49,0,0          ; vbroadcastss  0x3116(%rip),%ymm14        # 46a8 <_sk_callback_hsw+0x225>
+  DB  196,98,125,24,53,42,49,0,0          ; vbroadcastss  0x312a(%rip),%ymm14        # 46bc <_sk_callback_hsw+0x225>
   DB  197,12,92,224                       ; vsubps        %ymm0,%ymm14,%ymm12
   DB  196,66,37,168,225                   ; vfmadd213ps   %ymm9,%ymm11,%ymm12
-  DB  196,226,125,24,29,252,48,0,0        ; vbroadcastss  0x30fc(%rip),%ymm3        # 46a0 <_sk_callback_hsw+0x21d>
+  DB  196,226,125,24,29,16,49,0,0         ; vbroadcastss  0x3110(%rip),%ymm3        # 46b4 <_sk_callback_hsw+0x21d>
   DB  196,193,100,194,255,2               ; vcmpleps      %ymm15,%ymm3,%ymm7
   DB  196,195,29,74,249,112               ; vblendvps     %ymm7,%ymm9,%ymm12,%ymm7
   DB  196,65,60,194,231,2                 ; vcmpleps      %ymm15,%ymm8,%ymm12
   DB  196,227,45,74,255,192               ; vblendvps     %ymm12,%ymm7,%ymm10,%ymm7
-  DB  196,98,125,24,37,231,48,0,0         ; vbroadcastss  0x30e7(%rip),%ymm12        # 46ac <_sk_callback_hsw+0x229>
+  DB  196,98,125,24,37,251,48,0,0         ; vbroadcastss  0x30fb(%rip),%ymm12        # 46c0 <_sk_callback_hsw+0x229>
   DB  196,65,28,194,255,2                 ; vcmpleps      %ymm15,%ymm12,%ymm15
   DB  196,194,37,168,193                  ; vfmadd213ps   %ymm9,%ymm11,%ymm0
   DB  196,99,125,74,255,240               ; vblendvps     %ymm15,%ymm7,%ymm0,%ymm15
@@ -1355,7 +1355,7 @@ _sk_hsl_to_rgb_hsw LABEL PROC
   DB  197,156,194,192,2                   ; vcmpleps      %ymm0,%ymm12,%ymm0
   DB  196,194,37,168,249                  ; vfmadd213ps   %ymm9,%ymm11,%ymm7
   DB  196,227,69,74,201,0                 ; vblendvps     %ymm0,%ymm1,%ymm7,%ymm1
-  DB  196,226,125,24,5,147,48,0,0         ; vbroadcastss  0x3093(%rip),%ymm0        # 46b0 <_sk_callback_hsw+0x22d>
+  DB  196,226,125,24,5,167,48,0,0         ; vbroadcastss  0x30a7(%rip),%ymm0        # 46c4 <_sk_callback_hsw+0x22d>
   DB  197,220,88,192                      ; vaddps        %ymm0,%ymm4,%ymm0
   DB  196,227,125,8,224,1                 ; vroundps      $0x1,%ymm0,%ymm4
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
@@ -1405,7 +1405,7 @@ _sk_scale_u8_hsw LABEL PROC
   DB  197,122,126,0                       ; vmovq         (%rax),%xmm8
   DB  196,66,125,49,192                   ; vpmovzxbd     %xmm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,205,47,0,0         ; vbroadcastss  0x2fcd(%rip),%ymm9        # 46b4 <_sk_callback_hsw+0x231>
+  DB  196,98,125,24,13,225,47,0,0         ; vbroadcastss  0x2fe1(%rip),%ymm9        # 46c8 <_sk_callback_hsw+0x231>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -1453,7 +1453,7 @@ _sk_lerp_u8_hsw LABEL PROC
   DB  197,122,126,0                       ; vmovq         (%rax),%xmm8
   DB  196,66,125,49,192                   ; vpmovzxbd     %xmm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,58,47,0,0          ; vbroadcastss  0x2f3a(%rip),%ymm9        # 46b8 <_sk_callback_hsw+0x235>
+  DB  196,98,125,24,13,78,47,0,0          ; vbroadcastss  0x2f4e(%rip),%ymm9        # 46cc <_sk_callback_hsw+0x235>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
   DB  196,226,61,168,196                  ; vfmadd213ps   %ymm4,%ymm8,%ymm0
@@ -1484,74 +1484,79 @@ _sk_lerp_565_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,149,0,0,0                    ; jne           1876 <_sk_lerp_565_hsw+0xa3>
-  DB  196,193,122,111,28,122              ; vmovdqu       (%r10,%rdi,2),%xmm3
-  DB  196,226,125,51,219                  ; vpmovzxwd     %xmm3,%ymm3
-  DB  196,98,125,88,5,199,46,0,0          ; vpbroadcastd  0x2ec7(%rip),%ymm8        # 46bc <_sk_callback_hsw+0x239>
-  DB  196,65,101,219,192                  ; vpand         %ymm8,%ymm3,%ymm8
-  DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,184,46,0,0         ; vbroadcastss  0x2eb8(%rip),%ymm9        # 46c0 <_sk_callback_hsw+0x23d>
-  DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,88,13,174,46,0,0         ; vpbroadcastd  0x2eae(%rip),%ymm9        # 46c4 <_sk_callback_hsw+0x241>
-  DB  196,65,101,219,201                  ; vpand         %ymm9,%ymm3,%ymm9
+  DB  15,133,169,0,0,0                    ; jne           188a <_sk_lerp_565_hsw+0xb7>
+  DB  196,65,122,111,4,122                ; vmovdqu       (%r10,%rdi,2),%xmm8
+  DB  196,66,125,51,192                   ; vpmovzxwd     %xmm8,%ymm8
+  DB  196,98,125,88,13,219,46,0,0         ; vpbroadcastd  0x2edb(%rip),%ymm9        # 46d0 <_sk_callback_hsw+0x239>
+  DB  196,65,61,219,201                   ; vpand         %ymm9,%ymm8,%ymm9
   DB  196,65,124,91,201                   ; vcvtdq2ps     %ymm9,%ymm9
-  DB  196,98,125,24,21,159,46,0,0         ; vbroadcastss  0x2e9f(%rip),%ymm10        # 46c8 <_sk_callback_hsw+0x245>
+  DB  196,98,125,24,21,204,46,0,0         ; vbroadcastss  0x2ecc(%rip),%ymm10        # 46d4 <_sk_callback_hsw+0x23d>
   DB  196,65,52,89,202                    ; vmulps        %ymm10,%ymm9,%ymm9
-  DB  196,98,125,88,21,149,46,0,0         ; vpbroadcastd  0x2e95(%rip),%ymm10        # 46cc <_sk_callback_hsw+0x249>
-  DB  196,193,101,219,218                 ; vpand         %ymm10,%ymm3,%ymm3
-  DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,21,135,46,0,0         ; vbroadcastss  0x2e87(%rip),%ymm10        # 46d0 <_sk_callback_hsw+0x24d>
-  DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
+  DB  196,98,125,88,21,194,46,0,0         ; vpbroadcastd  0x2ec2(%rip),%ymm10        # 46d8 <_sk_callback_hsw+0x241>
+  DB  196,65,61,219,210                   ; vpand         %ymm10,%ymm8,%ymm10
+  DB  196,65,124,91,210                   ; vcvtdq2ps     %ymm10,%ymm10
+  DB  196,98,125,24,29,179,46,0,0         ; vbroadcastss  0x2eb3(%rip),%ymm11        # 46dc <_sk_callback_hsw+0x245>
+  DB  196,65,44,89,211                    ; vmulps        %ymm11,%ymm10,%ymm10
+  DB  196,98,125,88,29,169,46,0,0         ; vpbroadcastd  0x2ea9(%rip),%ymm11        # 46e0 <_sk_callback_hsw+0x249>
+  DB  196,65,61,219,195                   ; vpand         %ymm11,%ymm8,%ymm8
+  DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
+  DB  196,98,125,24,29,154,46,0,0         ; vbroadcastss  0x2e9a(%rip),%ymm11        # 46e4 <_sk_callback_hsw+0x24d>
+  DB  196,65,60,89,195                    ; vmulps        %ymm11,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
-  DB  196,226,61,168,196                  ; vfmadd213ps   %ymm4,%ymm8,%ymm0
+  DB  196,226,53,168,196                  ; vfmadd213ps   %ymm4,%ymm9,%ymm0
   DB  197,244,92,205                      ; vsubps        %ymm5,%ymm1,%ymm1
-  DB  196,226,53,168,205                  ; vfmadd213ps   %ymm5,%ymm9,%ymm1
+  DB  196,226,45,168,205                  ; vfmadd213ps   %ymm5,%ymm10,%ymm1
   DB  197,236,92,214                      ; vsubps        %ymm6,%ymm2,%ymm2
-  DB  196,226,101,168,214                 ; vfmadd213ps   %ymm6,%ymm3,%ymm2
+  DB  196,226,61,168,214                  ; vfmadd213ps   %ymm6,%ymm8,%ymm2
+  DB  197,228,92,223                      ; vsubps        %ymm7,%ymm3,%ymm3
+  DB  196,98,101,168,207                  ; vfmadd213ps   %ymm7,%ymm3,%ymm9
+  DB  196,98,101,168,215                  ; vfmadd213ps   %ymm7,%ymm3,%ymm10
+  DB  196,98,101,168,199                  ; vfmadd213ps   %ymm7,%ymm3,%ymm8
+  DB  196,193,44,95,216                   ; vmaxps        %ymm8,%ymm10,%ymm3
+  DB  197,180,95,219                      ; vmaxps        %ymm3,%ymm9,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,96,46,0,0         ; vbroadcastss  0x2e60(%rip),%ymm3        # 46d4 <_sk_callback_hsw+0x251>
   DB  255,224                             ; jmpq          *%rax
   DB  65,137,200                          ; mov           %ecx,%r8d
   DB  65,128,224,7                        ; and           $0x7,%r8b
-  DB  197,225,239,219                     ; vpxor         %xmm3,%xmm3,%xmm3
+  DB  196,65,57,239,192                   ; vpxor         %xmm8,%xmm8,%xmm8
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,89,255,255,255               ; ja            17e7 <_sk_lerp_565_hsw+0x14>
+  DB  15,135,68,255,255,255               ; ja            17e7 <_sk_lerp_565_hsw+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 18e4 <_sk_lerp_565_hsw+0x111>
+  DB  76,141,13,74,0,0,0                  ; lea           0x4a(%rip),%r9        # 18f8 <_sk_lerp_565_hsw+0x125>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
-  DB  197,225,239,219                     ; vpxor         %xmm3,%xmm3,%xmm3
-  DB  196,193,97,196,92,122,12,6          ; vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm3,%xmm3
-  DB  196,193,97,196,92,122,10,5          ; vpinsrw       $0x5,0xa(%r10,%rdi,2),%xmm3,%xmm3
-  DB  196,193,97,196,92,122,8,4           ; vpinsrw       $0x4,0x8(%r10,%rdi,2),%xmm3,%xmm3
-  DB  196,193,97,196,92,122,6,3           ; vpinsrw       $0x3,0x6(%r10,%rdi,2),%xmm3,%xmm3
-  DB  196,193,97,196,92,122,4,2           ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm3,%xmm3
-  DB  196,193,97,196,92,122,2,1           ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm3,%xmm3
-  DB  196,193,97,196,28,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm3,%xmm3
-  DB  233,5,255,255,255                   ; jmpq          17e7 <_sk_lerp_565_hsw+0x14>
-  DB  102,144                             ; xchg          %ax,%ax
-  DB  242,255                             ; repnz         (bad)
+  DB  196,65,57,239,192                   ; vpxor         %xmm8,%xmm8,%xmm8
+  DB  196,65,57,196,68,122,12,6           ; vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm8,%xmm8
+  DB  196,65,57,196,68,122,10,5           ; vpinsrw       $0x5,0xa(%r10,%rdi,2),%xmm8,%xmm8
+  DB  196,65,57,196,68,122,8,4            ; vpinsrw       $0x4,0x8(%r10,%rdi,2),%xmm8,%xmm8
+  DB  196,65,57,196,68,122,6,3            ; vpinsrw       $0x3,0x6(%r10,%rdi,2),%xmm8,%xmm8
+  DB  196,65,57,196,68,122,4,2            ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
+  DB  196,65,57,196,68,122,2,1            ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
+  DB  196,65,57,196,4,122,0               ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
+  DB  233,239,254,255,255                 ; jmpq          17e7 <_sk_lerp_565_hsw+0x14>
+  DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  234                                 ; (bad)
   DB  255                                 ; (bad)
+  DB  236                                 ; in            (%dx),%al
   DB  255                                 ; (bad)
-  DB  255,226                             ; jmpq          *%rdx
   DB  255                                 ; (bad)
+  DB  255,228                             ; jmpq          *%rsp
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  218,255                             ; (bad)
   DB  255                                 ; (bad)
-  DB  255,210                             ; callq         *%rdx
+  DB  220,255                             ; fdivr         %st,%st(7)
+  DB  255                                 ; (bad)
+  DB  255,212                             ; callq         *%rsp
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,202                             ; dec           %edx
+  DB  255,204                             ; dec           %esp
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  190                                 ; .byte         0xbe
+  DB  191                                 ; .byte         0xbf
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
@@ -1563,23 +1568,23 @@ _sk_load_tables_hsw LABEL PROC
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  76,3,8                              ; add           (%rax),%r9
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,105                             ; jne           197e <_sk_load_tables_hsw+0x7e>
+  DB  117,105                             ; jne           1992 <_sk_load_tables_hsw+0x7e>
   DB  196,193,126,111,25                  ; vmovdqu       (%r9),%ymm3
-  DB  197,229,219,13,94,48,0,0            ; vpand         0x305e(%rip),%ymm3,%ymm1        # 4980 <_sk_callback_hsw+0x4fd>
+  DB  197,229,219,13,106,48,0,0           ; vpand         0x306a(%rip),%ymm3,%ymm1        # 49a0 <_sk_callback_hsw+0x509>
   DB  196,65,61,118,192                   ; vpcmpeqd      %ymm8,%ymm8,%ymm8
   DB  72,139,72,8                         ; mov           0x8(%rax),%rcx
   DB  76,139,72,16                        ; mov           0x10(%rax),%r9
   DB  197,237,118,210                     ; vpcmpeqd      %ymm2,%ymm2,%ymm2
   DB  196,226,109,146,4,137               ; vgatherdps    %ymm2,(%rcx,%ymm1,4),%ymm0
-  DB  196,226,101,0,21,94,48,0,0          ; vpshufb       0x305e(%rip),%ymm3,%ymm2        # 49a0 <_sk_callback_hsw+0x51d>
+  DB  196,226,101,0,21,106,48,0,0         ; vpshufb       0x306a(%rip),%ymm3,%ymm2        # 49c0 <_sk_callback_hsw+0x529>
   DB  196,65,53,118,201                   ; vpcmpeqd      %ymm9,%ymm9,%ymm9
   DB  196,194,53,146,12,145               ; vgatherdps    %ymm9,(%r9,%ymm2,4),%ymm1
   DB  72,139,64,24                        ; mov           0x18(%rax),%rax
-  DB  196,98,101,0,13,102,48,0,0          ; vpshufb       0x3066(%rip),%ymm3,%ymm9        # 49c0 <_sk_callback_hsw+0x53d>
+  DB  196,98,101,0,13,114,48,0,0          ; vpshufb       0x3072(%rip),%ymm3,%ymm9        # 49e0 <_sk_callback_hsw+0x549>
   DB  196,162,61,146,20,136               ; vgatherdps    %ymm8,(%rax,%ymm9,4),%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,102,45,0,0          ; vbroadcastss  0x2d66(%rip),%ymm8        # 46d8 <_sk_callback_hsw+0x255>
+  DB  196,98,125,24,5,98,45,0,0           ; vbroadcastss  0x2d62(%rip),%ymm8        # 46e8 <_sk_callback_hsw+0x251>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,137,193                          ; mov           %r8,%rcx
@@ -1592,7 +1597,7 @@ _sk_load_tables_hsw LABEL PROC
   DB  196,193,249,110,194                 ; vmovq         %r10,%xmm0
   DB  196,226,125,33,192                  ; vpmovsxbd     %xmm0,%ymm0
   DB  196,194,125,140,25                  ; vpmaskmovd    (%r9),%ymm0,%ymm3
-  DB  233,115,255,255,255                 ; jmpq          191a <_sk_load_tables_hsw+0x1a>
+  DB  233,115,255,255,255                 ; jmpq          192e <_sk_load_tables_hsw+0x1a>
 
 PUBLIC _sk_load_tables_u16_be_hsw
 _sk_load_tables_u16_be_hsw LABEL PROC
@@ -1600,7 +1605,7 @@ _sk_load_tables_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,201,0,0,0                    ; jne           1a86 <_sk_load_tables_u16_be_hsw+0xdf>
+  DB  15,133,201,0,0,0                    ; jne           1a9a <_sk_load_tables_u16_be_hsw+0xdf>
   DB  196,1,121,16,4,72                   ; vmovupd       (%r8,%r9,2),%xmm8
   DB  196,129,121,16,84,72,16             ; vmovupd       0x10(%r8,%r9,2),%xmm2
   DB  196,129,121,16,92,72,32             ; vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -1616,7 +1621,7 @@ _sk_load_tables_u16_be_hsw LABEL PROC
   DB  197,185,108,200                     ; vpunpcklqdq   %xmm0,%xmm8,%xmm1
   DB  197,185,109,208                     ; vpunpckhqdq   %xmm0,%xmm8,%xmm2
   DB  197,49,108,195                      ; vpunpcklqdq   %xmm3,%xmm9,%xmm8
-  DB  197,121,111,21,242,48,0,0           ; vmovdqa       0x30f2(%rip),%xmm10        # 4b00 <_sk_callback_hsw+0x67d>
+  DB  197,121,111,21,254,48,0,0           ; vmovdqa       0x30fe(%rip),%xmm10        # 4b20 <_sk_callback_hsw+0x689>
   DB  196,193,113,219,194                 ; vpand         %xmm10,%xmm1,%xmm0
   DB  196,226,125,51,200                  ; vpmovzxwd     %xmm0,%ymm1
   DB  196,65,37,118,219                   ; vpcmpeqd      %ymm11,%ymm11,%ymm11
@@ -1638,36 +1643,36 @@ _sk_load_tables_u16_be_hsw LABEL PROC
   DB  197,185,235,219                     ; vpor          %xmm3,%xmm8,%xmm3
   DB  196,226,125,51,219                  ; vpmovzxwd     %xmm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,95,44,0,0           ; vbroadcastss  0x2c5f(%rip),%ymm8        # 46dc <_sk_callback_hsw+0x259>
+  DB  196,98,125,24,5,91,44,0,0           ; vbroadcastss  0x2c5b(%rip),%ymm8        # 46ec <_sk_callback_hsw+0x255>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
   DB  196,1,123,16,4,72                   ; vmovsd        (%r8,%r9,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            1aec <_sk_load_tables_u16_be_hsw+0x145>
+  DB  116,85                              ; je            1b00 <_sk_load_tables_u16_be_hsw+0x145>
   DB  196,1,57,22,68,72,8                 ; vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            1aec <_sk_load_tables_u16_be_hsw+0x145>
+  DB  114,72                              ; jb            1b00 <_sk_load_tables_u16_be_hsw+0x145>
   DB  196,129,123,16,84,72,16             ; vmovsd        0x10(%r8,%r9,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            1af9 <_sk_load_tables_u16_be_hsw+0x152>
+  DB  116,72                              ; je            1b0d <_sk_load_tables_u16_be_hsw+0x152>
   DB  196,129,105,22,84,72,24             ; vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            1af9 <_sk_load_tables_u16_be_hsw+0x152>
+  DB  114,59                              ; jb            1b0d <_sk_load_tables_u16_be_hsw+0x152>
   DB  196,129,123,16,92,72,32             ; vmovsd        0x20(%r8,%r9,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,9,255,255,255                ; je            19d8 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  15,132,9,255,255,255                ; je            19ec <_sk_load_tables_u16_be_hsw+0x31>
   DB  196,129,97,22,92,72,40              ; vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,248,254,255,255              ; jb            19d8 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  15,130,248,254,255,255              ; jb            19ec <_sk_load_tables_u16_be_hsw+0x31>
   DB  196,1,122,126,76,72,48              ; vmovq         0x30(%r8,%r9,2),%xmm9
-  DB  233,236,254,255,255                 ; jmpq          19d8 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  233,236,254,255,255                 ; jmpq          19ec <_sk_load_tables_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,223,254,255,255                 ; jmpq          19d8 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  233,223,254,255,255                 ; jmpq          19ec <_sk_load_tables_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,214,254,255,255                 ; jmpq          19d8 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  233,214,254,255,255                 ; jmpq          19ec <_sk_load_tables_u16_be_hsw+0x31>
 
 PUBLIC _sk_load_tables_rgb_u16_be_hsw
 _sk_load_tables_rgb_u16_be_hsw LABEL PROC
@@ -1675,7 +1680,7 @@ _sk_load_tables_rgb_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,127                       ; lea           (%rdi,%rdi,2),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,193,0,0,0                    ; jne           1bd5 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
+  DB  15,133,193,0,0,0                    ; jne           1be9 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
   DB  196,129,122,111,4,72                ; vmovdqu       (%r8,%r9,2),%xmm0
   DB  196,129,122,111,84,72,12            ; vmovdqu       0xc(%r8,%r9,2),%xmm2
   DB  196,129,122,111,76,72,24            ; vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -1696,7 +1701,7 @@ _sk_load_tables_rgb_u16_be_hsw LABEL PROC
   DB  197,185,108,218                     ; vpunpcklqdq   %xmm2,%xmm8,%xmm3
   DB  197,185,109,210                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm2
   DB  197,121,108,193                     ; vpunpcklqdq   %xmm1,%xmm0,%xmm8
-  DB  197,121,111,13,146,47,0,0           ; vmovdqa       0x2f92(%rip),%xmm9        # 4b10 <_sk_callback_hsw+0x68d>
+  DB  197,121,111,13,158,47,0,0           ; vmovdqa       0x2f9e(%rip),%xmm9        # 4b30 <_sk_callback_hsw+0x699>
   DB  196,193,97,219,193                  ; vpand         %xmm9,%xmm3,%xmm0
   DB  196,226,125,51,200                  ; vpmovzxwd     %xmm0,%ymm1
   DB  197,229,118,219                     ; vpcmpeqd      %ymm3,%ymm3,%ymm3
@@ -1713,41 +1718,41 @@ _sk_load_tables_rgb_u16_be_hsw LABEL PROC
   DB  196,98,125,51,194                   ; vpmovzxwd     %xmm2,%ymm8
   DB  196,162,101,146,20,128              ; vgatherdps    %ymm3,(%rax,%ymm8,4),%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,13,43,0,0         ; vbroadcastss  0x2b0d(%rip),%ymm3        # 46e0 <_sk_callback_hsw+0x25d>
+  DB  196,226,125,24,29,9,43,0,0          ; vbroadcastss  0x2b09(%rip),%ymm3        # 46f0 <_sk_callback_hsw+0x259>
   DB  255,224                             ; jmpq          *%rax
   DB  196,129,121,110,4,72                ; vmovd         (%r8,%r9,2),%xmm0
   DB  196,129,121,196,68,72,4,2           ; vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           1bee <_sk_load_tables_rgb_u16_be_hsw+0xec>
-  DB  233,90,255,255,255                  ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  117,5                               ; jne           1c02 <_sk_load_tables_rgb_u16_be_hsw+0xec>
+  DB  233,90,255,255,255                  ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
   DB  196,129,121,110,76,72,6             ; vmovd         0x6(%r8,%r9,2),%xmm1
   DB  196,1,113,196,68,72,10,2            ; vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            1c1d <_sk_load_tables_rgb_u16_be_hsw+0x11b>
+  DB  114,26                              ; jb            1c31 <_sk_load_tables_rgb_u16_be_hsw+0x11b>
   DB  196,129,121,110,76,72,12            ; vmovd         0xc(%r8,%r9,2),%xmm1
   DB  196,129,113,196,84,72,16,2          ; vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           1c22 <_sk_load_tables_rgb_u16_be_hsw+0x120>
-  DB  233,43,255,255,255                  ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  DB  233,38,255,255,255                  ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           1c36 <_sk_load_tables_rgb_u16_be_hsw+0x120>
+  DB  233,43,255,255,255                  ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,38,255,255,255                  ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
   DB  196,129,121,110,76,72,18            ; vmovd         0x12(%r8,%r9,2),%xmm1
   DB  196,1,113,196,76,72,22,2            ; vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            1c51 <_sk_load_tables_rgb_u16_be_hsw+0x14f>
+  DB  114,26                              ; jb            1c65 <_sk_load_tables_rgb_u16_be_hsw+0x14f>
   DB  196,129,121,110,76,72,24            ; vmovd         0x18(%r8,%r9,2),%xmm1
   DB  196,129,113,196,76,72,28,2          ; vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           1c56 <_sk_load_tables_rgb_u16_be_hsw+0x154>
-  DB  233,247,254,255,255                 ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  DB  233,242,254,255,255                 ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           1c6a <_sk_load_tables_rgb_u16_be_hsw+0x154>
+  DB  233,247,254,255,255                 ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,242,254,255,255                 ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
   DB  196,129,121,110,92,72,30            ; vmovd         0x1e(%r8,%r9,2),%xmm3
   DB  196,1,97,196,92,72,34,2             ; vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            1c7f <_sk_load_tables_rgb_u16_be_hsw+0x17d>
+  DB  114,20                              ; jb            1c93 <_sk_load_tables_rgb_u16_be_hsw+0x17d>
   DB  196,129,121,110,92,72,36            ; vmovd         0x24(%r8,%r9,2),%xmm3
   DB  196,129,97,196,92,72,40,2           ; vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  DB  233,201,254,255,255                 ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  DB  233,196,254,255,255                 ; jmpq          1b48 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,201,254,255,255                 ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,196,254,255,255                 ; jmpq          1b5c <_sk_load_tables_rgb_u16_be_hsw+0x46>
 
 PUBLIC _sk_byte_tables_hsw
 _sk_byte_tables_hsw LABEL PROC
@@ -1758,7 +1763,7 @@ _sk_byte_tables_hsw LABEL PROC
   DB  65,84                               ; push          %r12
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,75,42,0,0           ; vbroadcastss  0x2a4b(%rip),%ymm8        # 46e4 <_sk_callback_hsw+0x261>
+  DB  196,98,125,24,5,71,42,0,0           ; vbroadcastss  0x2a47(%rip),%ymm8        # 46f4 <_sk_callback_hsw+0x25d>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,195,249,22,192,1                ; vpextrq       $0x1,%xmm0,%r8
@@ -1795,7 +1800,7 @@ _sk_byte_tables_hsw LABEL PROC
   DB  196,227,121,32,197,7                ; vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,156,41,0,0         ; vbroadcastss  0x299c(%rip),%ymm9        # 46e8 <_sk_callback_hsw+0x265>
+  DB  196,98,125,24,13,152,41,0,0         ; vbroadcastss  0x2998(%rip),%ymm9        # 46f8 <_sk_callback_hsw+0x261>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -1954,7 +1959,7 @@ _sk_byte_tables_rgb_hsw LABEL PROC
   DB  196,227,121,32,197,7                ; vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,213,38,0,0         ; vbroadcastss  0x26d5(%rip),%ymm9        # 46ec <_sk_callback_hsw+0x269>
+  DB  196,98,125,24,13,209,38,0,0         ; vbroadcastss  0x26d1(%rip),%ymm9        # 46fc <_sk_callback_hsw+0x265>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -2107,33 +2112,33 @@ _sk_parametric_r_hsw LABEL PROC
   DB  196,66,125,168,211                  ; vfmadd213ps   %ymm11,%ymm0,%ymm10
   DB  196,226,125,24,0                    ; vbroadcastss  (%rax),%ymm0
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,136,36,0,0         ; vbroadcastss  0x2488(%rip),%ymm12        # 46f0 <_sk_callback_hsw+0x26d>
-  DB  196,98,125,24,45,131,36,0,0         ; vbroadcastss  0x2483(%rip),%ymm13        # 46f4 <_sk_callback_hsw+0x271>
+  DB  196,98,125,24,37,132,36,0,0         ; vbroadcastss  0x2484(%rip),%ymm12        # 4700 <_sk_callback_hsw+0x269>
+  DB  196,98,125,24,45,127,36,0,0         ; vbroadcastss  0x247f(%rip),%ymm13        # 4704 <_sk_callback_hsw+0x26d>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,121,36,0,0         ; vbroadcastss  0x2479(%rip),%ymm13        # 46f8 <_sk_callback_hsw+0x275>
+  DB  196,98,125,24,45,117,36,0,0         ; vbroadcastss  0x2475(%rip),%ymm13        # 4708 <_sk_callback_hsw+0x271>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,111,36,0,0         ; vbroadcastss  0x246f(%rip),%ymm13        # 46fc <_sk_callback_hsw+0x279>
+  DB  196,98,125,24,45,107,36,0,0         ; vbroadcastss  0x246b(%rip),%ymm13        # 470c <_sk_callback_hsw+0x275>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,101,36,0,0         ; vbroadcastss  0x2465(%rip),%ymm11        # 4700 <_sk_callback_hsw+0x27d>
+  DB  196,98,125,24,29,97,36,0,0          ; vbroadcastss  0x2461(%rip),%ymm11        # 4710 <_sk_callback_hsw+0x279>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,91,36,0,0          ; vbroadcastss  0x245b(%rip),%ymm12        # 4704 <_sk_callback_hsw+0x281>
+  DB  196,98,125,24,37,87,36,0,0          ; vbroadcastss  0x2457(%rip),%ymm12        # 4714 <_sk_callback_hsw+0x27d>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,81,36,0,0          ; vbroadcastss  0x2451(%rip),%ymm12        # 4708 <_sk_callback_hsw+0x285>
+  DB  196,98,125,24,37,77,36,0,0          ; vbroadcastss  0x244d(%rip),%ymm12        # 4718 <_sk_callback_hsw+0x281>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  196,99,125,8,208,1                  ; vroundps      $0x1,%ymm0,%ymm10
   DB  196,65,124,92,210                   ; vsubps        %ymm10,%ymm0,%ymm10
-  DB  196,98,125,24,29,50,36,0,0          ; vbroadcastss  0x2432(%rip),%ymm11        # 470c <_sk_callback_hsw+0x289>
+  DB  196,98,125,24,29,46,36,0,0          ; vbroadcastss  0x242e(%rip),%ymm11        # 471c <_sk_callback_hsw+0x285>
   DB  196,193,124,88,195                  ; vaddps        %ymm11,%ymm0,%ymm0
-  DB  196,98,125,24,29,40,36,0,0          ; vbroadcastss  0x2428(%rip),%ymm11        # 4710 <_sk_callback_hsw+0x28d>
+  DB  196,98,125,24,29,36,36,0,0          ; vbroadcastss  0x2424(%rip),%ymm11        # 4720 <_sk_callback_hsw+0x289>
   DB  196,98,45,172,216                   ; vfnmadd213ps  %ymm0,%ymm10,%ymm11
-  DB  196,226,125,24,5,30,36,0,0          ; vbroadcastss  0x241e(%rip),%ymm0        # 4714 <_sk_callback_hsw+0x291>
+  DB  196,226,125,24,5,26,36,0,0          ; vbroadcastss  0x241a(%rip),%ymm0        # 4724 <_sk_callback_hsw+0x28d>
   DB  196,193,124,92,194                  ; vsubps        %ymm10,%ymm0,%ymm0
-  DB  196,98,125,24,21,20,36,0,0          ; vbroadcastss  0x2414(%rip),%ymm10        # 4718 <_sk_callback_hsw+0x295>
+  DB  196,98,125,24,21,16,36,0,0          ; vbroadcastss  0x2410(%rip),%ymm10        # 4728 <_sk_callback_hsw+0x291>
   DB  197,172,94,192                      ; vdivps        %ymm0,%ymm10,%ymm0
   DB  197,164,88,192                      ; vaddps        %ymm0,%ymm11,%ymm0
-  DB  196,98,125,24,21,7,36,0,0           ; vbroadcastss  0x2407(%rip),%ymm10        # 471c <_sk_callback_hsw+0x299>
+  DB  196,98,125,24,21,3,36,0,0           ; vbroadcastss  0x2403(%rip),%ymm10        # 472c <_sk_callback_hsw+0x295>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2141,7 +2146,7 @@ _sk_parametric_r_hsw LABEL PROC
   DB  196,195,125,74,193,128              ; vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,124,95,192                  ; vmaxps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,222,35,0,0          ; vbroadcastss  0x23de(%rip),%ymm8        # 4720 <_sk_callback_hsw+0x29d>
+  DB  196,98,125,24,5,218,35,0,0          ; vbroadcastss  0x23da(%rip),%ymm8        # 4730 <_sk_callback_hsw+0x299>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2159,33 +2164,33 @@ _sk_parametric_g_hsw LABEL PROC
   DB  196,66,117,168,211                  ; vfmadd213ps   %ymm11,%ymm1,%ymm10
   DB  196,226,125,24,8                    ; vbroadcastss  (%rax),%ymm1
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,150,35,0,0         ; vbroadcastss  0x2396(%rip),%ymm12        # 4724 <_sk_callback_hsw+0x2a1>
-  DB  196,98,125,24,45,145,35,0,0         ; vbroadcastss  0x2391(%rip),%ymm13        # 4728 <_sk_callback_hsw+0x2a5>
+  DB  196,98,125,24,37,146,35,0,0         ; vbroadcastss  0x2392(%rip),%ymm12        # 4734 <_sk_callback_hsw+0x29d>
+  DB  196,98,125,24,45,141,35,0,0         ; vbroadcastss  0x238d(%rip),%ymm13        # 4738 <_sk_callback_hsw+0x2a1>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,135,35,0,0         ; vbroadcastss  0x2387(%rip),%ymm13        # 472c <_sk_callback_hsw+0x2a9>
+  DB  196,98,125,24,45,131,35,0,0         ; vbroadcastss  0x2383(%rip),%ymm13        # 473c <_sk_callback_hsw+0x2a5>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,125,35,0,0         ; vbroadcastss  0x237d(%rip),%ymm13        # 4730 <_sk_callback_hsw+0x2ad>
+  DB  196,98,125,24,45,121,35,0,0         ; vbroadcastss  0x2379(%rip),%ymm13        # 4740 <_sk_callback_hsw+0x2a9>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,115,35,0,0         ; vbroadcastss  0x2373(%rip),%ymm11        # 4734 <_sk_callback_hsw+0x2b1>
+  DB  196,98,125,24,29,111,35,0,0         ; vbroadcastss  0x236f(%rip),%ymm11        # 4744 <_sk_callback_hsw+0x2ad>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,105,35,0,0         ; vbroadcastss  0x2369(%rip),%ymm12        # 4738 <_sk_callback_hsw+0x2b5>
+  DB  196,98,125,24,37,101,35,0,0         ; vbroadcastss  0x2365(%rip),%ymm12        # 4748 <_sk_callback_hsw+0x2b1>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,95,35,0,0          ; vbroadcastss  0x235f(%rip),%ymm12        # 473c <_sk_callback_hsw+0x2b9>
+  DB  196,98,125,24,37,91,35,0,0          ; vbroadcastss  0x235b(%rip),%ymm12        # 474c <_sk_callback_hsw+0x2b5>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  196,99,125,8,209,1                  ; vroundps      $0x1,%ymm1,%ymm10
   DB  196,65,116,92,210                   ; vsubps        %ymm10,%ymm1,%ymm10
-  DB  196,98,125,24,29,64,35,0,0          ; vbroadcastss  0x2340(%rip),%ymm11        # 4740 <_sk_callback_hsw+0x2bd>
+  DB  196,98,125,24,29,60,35,0,0          ; vbroadcastss  0x233c(%rip),%ymm11        # 4750 <_sk_callback_hsw+0x2b9>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,54,35,0,0          ; vbroadcastss  0x2336(%rip),%ymm11        # 4744 <_sk_callback_hsw+0x2c1>
+  DB  196,98,125,24,29,50,35,0,0          ; vbroadcastss  0x2332(%rip),%ymm11        # 4754 <_sk_callback_hsw+0x2bd>
   DB  196,98,45,172,217                   ; vfnmadd213ps  %ymm1,%ymm10,%ymm11
-  DB  196,226,125,24,13,44,35,0,0         ; vbroadcastss  0x232c(%rip),%ymm1        # 4748 <_sk_callback_hsw+0x2c5>
+  DB  196,226,125,24,13,40,35,0,0         ; vbroadcastss  0x2328(%rip),%ymm1        # 4758 <_sk_callback_hsw+0x2c1>
   DB  196,193,116,92,202                  ; vsubps        %ymm10,%ymm1,%ymm1
-  DB  196,98,125,24,21,34,35,0,0          ; vbroadcastss  0x2322(%rip),%ymm10        # 474c <_sk_callback_hsw+0x2c9>
+  DB  196,98,125,24,21,30,35,0,0          ; vbroadcastss  0x231e(%rip),%ymm10        # 475c <_sk_callback_hsw+0x2c5>
   DB  197,172,94,201                      ; vdivps        %ymm1,%ymm10,%ymm1
   DB  197,164,88,201                      ; vaddps        %ymm1,%ymm11,%ymm1
-  DB  196,98,125,24,21,21,35,0,0          ; vbroadcastss  0x2315(%rip),%ymm10        # 4750 <_sk_callback_hsw+0x2cd>
+  DB  196,98,125,24,21,17,35,0,0          ; vbroadcastss  0x2311(%rip),%ymm10        # 4760 <_sk_callback_hsw+0x2c9>
   DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2193,7 +2198,7 @@ _sk_parametric_g_hsw LABEL PROC
   DB  196,195,117,74,201,128              ; vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,116,95,200                  ; vmaxps        %ymm8,%ymm1,%ymm1
-  DB  196,98,125,24,5,236,34,0,0          ; vbroadcastss  0x22ec(%rip),%ymm8        # 4754 <_sk_callback_hsw+0x2d1>
+  DB  196,98,125,24,5,232,34,0,0          ; vbroadcastss  0x22e8(%rip),%ymm8        # 4764 <_sk_callback_hsw+0x2cd>
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2211,33 +2216,33 @@ _sk_parametric_b_hsw LABEL PROC
   DB  196,66,109,168,211                  ; vfmadd213ps   %ymm11,%ymm2,%ymm10
   DB  196,226,125,24,16                   ; vbroadcastss  (%rax),%ymm2
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,164,34,0,0         ; vbroadcastss  0x22a4(%rip),%ymm12        # 4758 <_sk_callback_hsw+0x2d5>
-  DB  196,98,125,24,45,159,34,0,0         ; vbroadcastss  0x229f(%rip),%ymm13        # 475c <_sk_callback_hsw+0x2d9>
+  DB  196,98,125,24,37,160,34,0,0         ; vbroadcastss  0x22a0(%rip),%ymm12        # 4768 <_sk_callback_hsw+0x2d1>
+  DB  196,98,125,24,45,155,34,0,0         ; vbroadcastss  0x229b(%rip),%ymm13        # 476c <_sk_callback_hsw+0x2d5>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,149,34,0,0         ; vbroadcastss  0x2295(%rip),%ymm13        # 4760 <_sk_callback_hsw+0x2dd>
+  DB  196,98,125,24,45,145,34,0,0         ; vbroadcastss  0x2291(%rip),%ymm13        # 4770 <_sk_callback_hsw+0x2d9>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,139,34,0,0         ; vbroadcastss  0x228b(%rip),%ymm13        # 4764 <_sk_callback_hsw+0x2e1>
+  DB  196,98,125,24,45,135,34,0,0         ; vbroadcastss  0x2287(%rip),%ymm13        # 4774 <_sk_callback_hsw+0x2dd>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,129,34,0,0         ; vbroadcastss  0x2281(%rip),%ymm11        # 4768 <_sk_callback_hsw+0x2e5>
+  DB  196,98,125,24,29,125,34,0,0         ; vbroadcastss  0x227d(%rip),%ymm11        # 4778 <_sk_callback_hsw+0x2e1>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,119,34,0,0         ; vbroadcastss  0x2277(%rip),%ymm12        # 476c <_sk_callback_hsw+0x2e9>
+  DB  196,98,125,24,37,115,34,0,0         ; vbroadcastss  0x2273(%rip),%ymm12        # 477c <_sk_callback_hsw+0x2e5>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,109,34,0,0         ; vbroadcastss  0x226d(%rip),%ymm12        # 4770 <_sk_callback_hsw+0x2ed>
+  DB  196,98,125,24,37,105,34,0,0         ; vbroadcastss  0x2269(%rip),%ymm12        # 4780 <_sk_callback_hsw+0x2e9>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  196,99,125,8,210,1                  ; vroundps      $0x1,%ymm2,%ymm10
   DB  196,65,108,92,210                   ; vsubps        %ymm10,%ymm2,%ymm10
-  DB  196,98,125,24,29,78,34,0,0          ; vbroadcastss  0x224e(%rip),%ymm11        # 4774 <_sk_callback_hsw+0x2f1>
+  DB  196,98,125,24,29,74,34,0,0          ; vbroadcastss  0x224a(%rip),%ymm11        # 4784 <_sk_callback_hsw+0x2ed>
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
-  DB  196,98,125,24,29,68,34,0,0          ; vbroadcastss  0x2244(%rip),%ymm11        # 4778 <_sk_callback_hsw+0x2f5>
+  DB  196,98,125,24,29,64,34,0,0          ; vbroadcastss  0x2240(%rip),%ymm11        # 4788 <_sk_callback_hsw+0x2f1>
   DB  196,98,45,172,218                   ; vfnmadd213ps  %ymm2,%ymm10,%ymm11
-  DB  196,226,125,24,21,58,34,0,0         ; vbroadcastss  0x223a(%rip),%ymm2        # 477c <_sk_callback_hsw+0x2f9>
+  DB  196,226,125,24,21,54,34,0,0         ; vbroadcastss  0x2236(%rip),%ymm2        # 478c <_sk_callback_hsw+0x2f5>
   DB  196,193,108,92,210                  ; vsubps        %ymm10,%ymm2,%ymm2
-  DB  196,98,125,24,21,48,34,0,0          ; vbroadcastss  0x2230(%rip),%ymm10        # 4780 <_sk_callback_hsw+0x2fd>
+  DB  196,98,125,24,21,44,34,0,0          ; vbroadcastss  0x222c(%rip),%ymm10        # 4790 <_sk_callback_hsw+0x2f9>
   DB  197,172,94,210                      ; vdivps        %ymm2,%ymm10,%ymm2
   DB  197,164,88,210                      ; vaddps        %ymm2,%ymm11,%ymm2
-  DB  196,98,125,24,21,35,34,0,0          ; vbroadcastss  0x2223(%rip),%ymm10        # 4784 <_sk_callback_hsw+0x301>
+  DB  196,98,125,24,21,31,34,0,0          ; vbroadcastss  0x221f(%rip),%ymm10        # 4794 <_sk_callback_hsw+0x2fd>
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  197,253,91,210                      ; vcvtps2dq     %ymm2,%ymm2
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2245,7 +2250,7 @@ _sk_parametric_b_hsw LABEL PROC
   DB  196,195,109,74,209,128              ; vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,108,95,208                  ; vmaxps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,250,33,0,0          ; vbroadcastss  0x21fa(%rip),%ymm8        # 4788 <_sk_callback_hsw+0x305>
+  DB  196,98,125,24,5,246,33,0,0          ; vbroadcastss  0x21f6(%rip),%ymm8        # 4798 <_sk_callback_hsw+0x301>
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2263,33 +2268,33 @@ _sk_parametric_a_hsw LABEL PROC
   DB  196,66,101,168,211                  ; vfmadd213ps   %ymm11,%ymm3,%ymm10
   DB  196,226,125,24,24                   ; vbroadcastss  (%rax),%ymm3
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,178,33,0,0         ; vbroadcastss  0x21b2(%rip),%ymm12        # 478c <_sk_callback_hsw+0x309>
-  DB  196,98,125,24,45,173,33,0,0         ; vbroadcastss  0x21ad(%rip),%ymm13        # 4790 <_sk_callback_hsw+0x30d>
+  DB  196,98,125,24,37,174,33,0,0         ; vbroadcastss  0x21ae(%rip),%ymm12        # 479c <_sk_callback_hsw+0x305>
+  DB  196,98,125,24,45,169,33,0,0         ; vbroadcastss  0x21a9(%rip),%ymm13        # 47a0 <_sk_callback_hsw+0x309>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,163,33,0,0         ; vbroadcastss  0x21a3(%rip),%ymm13        # 4794 <_sk_callback_hsw+0x311>
+  DB  196,98,125,24,45,159,33,0,0         ; vbroadcastss  0x219f(%rip),%ymm13        # 47a4 <_sk_callback_hsw+0x30d>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,153,33,0,0         ; vbroadcastss  0x2199(%rip),%ymm13        # 4798 <_sk_callback_hsw+0x315>
+  DB  196,98,125,24,45,149,33,0,0         ; vbroadcastss  0x2195(%rip),%ymm13        # 47a8 <_sk_callback_hsw+0x311>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,143,33,0,0         ; vbroadcastss  0x218f(%rip),%ymm11        # 479c <_sk_callback_hsw+0x319>
+  DB  196,98,125,24,29,139,33,0,0         ; vbroadcastss  0x218b(%rip),%ymm11        # 47ac <_sk_callback_hsw+0x315>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,133,33,0,0         ; vbroadcastss  0x2185(%rip),%ymm12        # 47a0 <_sk_callback_hsw+0x31d>
+  DB  196,98,125,24,37,129,33,0,0         ; vbroadcastss  0x2181(%rip),%ymm12        # 47b0 <_sk_callback_hsw+0x319>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,123,33,0,0         ; vbroadcastss  0x217b(%rip),%ymm12        # 47a4 <_sk_callback_hsw+0x321>
+  DB  196,98,125,24,37,119,33,0,0         ; vbroadcastss  0x2177(%rip),%ymm12        # 47b4 <_sk_callback_hsw+0x31d>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
   DB  196,99,125,8,211,1                  ; vroundps      $0x1,%ymm3,%ymm10
   DB  196,65,100,92,210                   ; vsubps        %ymm10,%ymm3,%ymm10
-  DB  196,98,125,24,29,92,33,0,0          ; vbroadcastss  0x215c(%rip),%ymm11        # 47a8 <_sk_callback_hsw+0x325>
+  DB  196,98,125,24,29,88,33,0,0          ; vbroadcastss  0x2158(%rip),%ymm11        # 47b8 <_sk_callback_hsw+0x321>
   DB  196,193,100,88,219                  ; vaddps        %ymm11,%ymm3,%ymm3
-  DB  196,98,125,24,29,82,33,0,0          ; vbroadcastss  0x2152(%rip),%ymm11        # 47ac <_sk_callback_hsw+0x329>
+  DB  196,98,125,24,29,78,33,0,0          ; vbroadcastss  0x214e(%rip),%ymm11        # 47bc <_sk_callback_hsw+0x325>
   DB  196,98,45,172,219                   ; vfnmadd213ps  %ymm3,%ymm10,%ymm11
-  DB  196,226,125,24,29,72,33,0,0         ; vbroadcastss  0x2148(%rip),%ymm3        # 47b0 <_sk_callback_hsw+0x32d>
+  DB  196,226,125,24,29,68,33,0,0         ; vbroadcastss  0x2144(%rip),%ymm3        # 47c0 <_sk_callback_hsw+0x329>
   DB  196,193,100,92,218                  ; vsubps        %ymm10,%ymm3,%ymm3
-  DB  196,98,125,24,21,62,33,0,0          ; vbroadcastss  0x213e(%rip),%ymm10        # 47b4 <_sk_callback_hsw+0x331>
+  DB  196,98,125,24,21,58,33,0,0          ; vbroadcastss  0x213a(%rip),%ymm10        # 47c4 <_sk_callback_hsw+0x32d>
   DB  197,172,94,219                      ; vdivps        %ymm3,%ymm10,%ymm3
   DB  197,164,88,219                      ; vaddps        %ymm3,%ymm11,%ymm3
-  DB  196,98,125,24,21,49,33,0,0          ; vbroadcastss  0x2131(%rip),%ymm10        # 47b8 <_sk_callback_hsw+0x335>
+  DB  196,98,125,24,21,45,33,0,0          ; vbroadcastss  0x212d(%rip),%ymm10        # 47c8 <_sk_callback_hsw+0x331>
   DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
   DB  197,253,91,219                      ; vcvtps2dq     %ymm3,%ymm3
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2297,33 +2302,33 @@ _sk_parametric_a_hsw LABEL PROC
   DB  196,195,101,74,217,128              ; vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,100,95,216                  ; vmaxps        %ymm8,%ymm3,%ymm3
-  DB  196,98,125,24,5,8,33,0,0            ; vbroadcastss  0x2108(%rip),%ymm8        # 47bc <_sk_callback_hsw+0x339>
+  DB  196,98,125,24,5,4,33,0,0            ; vbroadcastss  0x2104(%rip),%ymm8        # 47cc <_sk_callback_hsw+0x335>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_lab_to_xyz_hsw
 _sk_lab_to_xyz_hsw LABEL PROC
-  DB  196,98,125,24,5,250,32,0,0          ; vbroadcastss  0x20fa(%rip),%ymm8        # 47c0 <_sk_callback_hsw+0x33d>
-  DB  196,98,125,24,13,245,32,0,0         ; vbroadcastss  0x20f5(%rip),%ymm9        # 47c4 <_sk_callback_hsw+0x341>
-  DB  196,98,125,24,21,240,32,0,0         ; vbroadcastss  0x20f0(%rip),%ymm10        # 47c8 <_sk_callback_hsw+0x345>
+  DB  196,98,125,24,5,246,32,0,0          ; vbroadcastss  0x20f6(%rip),%ymm8        # 47d0 <_sk_callback_hsw+0x339>
+  DB  196,98,125,24,13,241,32,0,0         ; vbroadcastss  0x20f1(%rip),%ymm9        # 47d4 <_sk_callback_hsw+0x33d>
+  DB  196,98,125,24,21,236,32,0,0         ; vbroadcastss  0x20ec(%rip),%ymm10        # 47d8 <_sk_callback_hsw+0x341>
   DB  196,194,53,168,202                  ; vfmadd213ps   %ymm10,%ymm9,%ymm1
   DB  196,194,53,168,210                  ; vfmadd213ps   %ymm10,%ymm9,%ymm2
-  DB  196,98,125,24,13,225,32,0,0         ; vbroadcastss  0x20e1(%rip),%ymm9        # 47cc <_sk_callback_hsw+0x349>
+  DB  196,98,125,24,13,221,32,0,0         ; vbroadcastss  0x20dd(%rip),%ymm9        # 47dc <_sk_callback_hsw+0x345>
   DB  196,66,125,184,200                  ; vfmadd231ps   %ymm8,%ymm0,%ymm9
-  DB  196,226,125,24,5,215,32,0,0         ; vbroadcastss  0x20d7(%rip),%ymm0        # 47d0 <_sk_callback_hsw+0x34d>
+  DB  196,226,125,24,5,211,32,0,0         ; vbroadcastss  0x20d3(%rip),%ymm0        # 47e0 <_sk_callback_hsw+0x349>
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
-  DB  196,98,125,24,5,206,32,0,0          ; vbroadcastss  0x20ce(%rip),%ymm8        # 47d4 <_sk_callback_hsw+0x351>
+  DB  196,98,125,24,5,202,32,0,0          ; vbroadcastss  0x20ca(%rip),%ymm8        # 47e4 <_sk_callback_hsw+0x34d>
   DB  196,98,117,168,192                  ; vfmadd213ps   %ymm0,%ymm1,%ymm8
-  DB  196,98,125,24,13,196,32,0,0         ; vbroadcastss  0x20c4(%rip),%ymm9        # 47d8 <_sk_callback_hsw+0x355>
+  DB  196,98,125,24,13,192,32,0,0         ; vbroadcastss  0x20c0(%rip),%ymm9        # 47e8 <_sk_callback_hsw+0x351>
   DB  196,98,109,172,200                  ; vfnmadd213ps  %ymm0,%ymm2,%ymm9
   DB  196,193,60,89,200                   ; vmulps        %ymm8,%ymm8,%ymm1
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
-  DB  196,226,125,24,21,177,32,0,0        ; vbroadcastss  0x20b1(%rip),%ymm2        # 47dc <_sk_callback_hsw+0x359>
+  DB  196,226,125,24,21,173,32,0,0        ; vbroadcastss  0x20ad(%rip),%ymm2        # 47ec <_sk_callback_hsw+0x355>
   DB  197,108,194,209,1                   ; vcmpltps      %ymm1,%ymm2,%ymm10
-  DB  196,98,125,24,29,167,32,0,0         ; vbroadcastss  0x20a7(%rip),%ymm11        # 47e0 <_sk_callback_hsw+0x35d>
+  DB  196,98,125,24,29,163,32,0,0         ; vbroadcastss  0x20a3(%rip),%ymm11        # 47f0 <_sk_callback_hsw+0x359>
   DB  196,65,60,88,195                    ; vaddps        %ymm11,%ymm8,%ymm8
-  DB  196,98,125,24,37,157,32,0,0         ; vbroadcastss  0x209d(%rip),%ymm12        # 47e4 <_sk_callback_hsw+0x361>
+  DB  196,98,125,24,37,153,32,0,0         ; vbroadcastss  0x2099(%rip),%ymm12        # 47f4 <_sk_callback_hsw+0x35d>
   DB  196,65,60,89,196                    ; vmulps        %ymm12,%ymm8,%ymm8
   DB  196,99,61,74,193,160                ; vblendvps     %ymm10,%ymm1,%ymm8,%ymm8
   DB  197,252,89,200                      ; vmulps        %ymm0,%ymm0,%ymm1
@@ -2338,9 +2343,9 @@ _sk_lab_to_xyz_hsw LABEL PROC
   DB  196,65,52,88,203                    ; vaddps        %ymm11,%ymm9,%ymm9
   DB  196,65,52,89,204                    ; vmulps        %ymm12,%ymm9,%ymm9
   DB  196,227,53,74,208,32                ; vblendvps     %ymm2,%ymm0,%ymm9,%ymm2
-  DB  196,226,125,24,5,82,32,0,0          ; vbroadcastss  0x2052(%rip),%ymm0        # 47e8 <_sk_callback_hsw+0x365>
+  DB  196,226,125,24,5,78,32,0,0          ; vbroadcastss  0x204e(%rip),%ymm0        # 47f8 <_sk_callback_hsw+0x361>
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
-  DB  196,98,125,24,5,73,32,0,0           ; vbroadcastss  0x2049(%rip),%ymm8        # 47ec <_sk_callback_hsw+0x369>
+  DB  196,98,125,24,5,69,32,0,0           ; vbroadcastss  0x2045(%rip),%ymm8        # 47fc <_sk_callback_hsw+0x365>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2352,11 +2357,11 @@ _sk_load_a8_hsw LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,45                              ; jne           27e9 <_sk_load_a8_hsw+0x3d>
+  DB  117,45                              ; jne           27fd <_sk_load_a8_hsw+0x3d>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,30,32,0,0         ; vbroadcastss  0x201e(%rip),%ymm1        # 47f0 <_sk_callback_hsw+0x36d>
+  DB  196,226,125,24,13,26,32,0,0         ; vbroadcastss  0x201a(%rip),%ymm1        # 4800 <_sk_callback_hsw+0x369>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -2373,9 +2378,9 @@ _sk_load_a8_hsw LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           27f1 <_sk_load_a8_hsw+0x45>
+  DB  117,234                             ; jne           2805 <_sk_load_a8_hsw+0x45>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,178                             ; jmp           27c0 <_sk_load_a8_hsw+0x14>
+  DB  235,178                             ; jmp           27d4 <_sk_load_a8_hsw+0x14>
 
 PUBLIC _sk_gather_a8_hsw
 _sk_gather_a8_hsw LABEL PROC
@@ -2419,7 +2424,7 @@ _sk_gather_a8_hsw LABEL PROC
   DB  196,227,121,32,192,7                ; vpinsrb       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,41,31,0,0         ; vbroadcastss  0x1f29(%rip),%ymm1        # 47f4 <_sk_callback_hsw+0x371>
+  DB  196,226,125,24,13,37,31,0,0         ; vbroadcastss  0x1f25(%rip),%ymm1        # 4804 <_sk_callback_hsw+0x36d>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -2435,14 +2440,14 @@ PUBLIC _sk_store_a8_hsw
 _sk_store_a8_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,4,31,0,0            ; vbroadcastss  0x1f04(%rip),%ymm8        # 47f8 <_sk_callback_hsw+0x375>
+  DB  196,98,125,24,5,0,31,0,0            ; vbroadcastss  0x1f00(%rip),%ymm8        # 4808 <_sk_callback_hsw+0x371>
   DB  196,65,100,89,192                   ; vmulps        %ymm8,%ymm3,%ymm8
   DB  196,65,125,91,192                   ; vcvtps2dq     %ymm8,%ymm8
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  196,65,57,103,192                   ; vpackuswb     %xmm8,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           291d <_sk_store_a8_hsw+0x37>
+  DB  117,10                              ; jne           2931 <_sk_store_a8_hsw+0x37>
   DB  196,65,123,17,4,58                  ; vmovsd        %xmm8,(%r10,%rdi,1)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2450,10 +2455,10 @@ _sk_store_a8_hsw LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            2919 <_sk_store_a8_hsw+0x33>
+  DB  119,236                             ; ja            292d <_sk_store_a8_hsw+0x33>
   DB  196,66,121,48,192                   ; vpmovzxbw     %xmm8,%xmm8
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 2980 <_sk_store_a8_hsw+0x9a>
+  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 2994 <_sk_store_a8_hsw+0x9a>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2464,7 +2469,7 @@ _sk_store_a8_hsw LABEL PROC
   DB  196,67,121,20,68,58,2,4             ; vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   DB  196,67,121,20,68,58,1,2             ; vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   DB  196,67,121,20,4,58,0                ; vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  DB  235,154                             ; jmp           2919 <_sk_store_a8_hsw+0x33>
+  DB  235,154                             ; jmp           292d <_sk_store_a8_hsw+0x33>
   DB  144                                 ; nop
   DB  246,255                             ; idiv          %bh
   DB  255                                 ; (bad)
@@ -2496,14 +2501,14 @@ _sk_load_g8_hsw LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,50                              ; jne           29de <_sk_load_g8_hsw+0x42>
+  DB  117,50                              ; jne           29f2 <_sk_load_g8_hsw+0x42>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,58,30,0,0         ; vbroadcastss  0x1e3a(%rip),%ymm1        # 47fc <_sk_callback_hsw+0x379>
+  DB  196,226,125,24,13,54,30,0,0         ; vbroadcastss  0x1e36(%rip),%ymm1        # 480c <_sk_callback_hsw+0x375>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,47,30,0,0         ; vbroadcastss  0x1e2f(%rip),%ymm3        # 4800 <_sk_callback_hsw+0x37d>
+  DB  196,226,125,24,29,43,30,0,0         ; vbroadcastss  0x1e2b(%rip),%ymm3        # 4810 <_sk_callback_hsw+0x379>
   DB  76,137,193                          ; mov           %r8,%rcx
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
@@ -2517,9 +2522,9 @@ _sk_load_g8_hsw LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           29e6 <_sk_load_g8_hsw+0x4a>
+  DB  117,234                             ; jne           29fa <_sk_load_g8_hsw+0x4a>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,173                             ; jmp           29b0 <_sk_load_g8_hsw+0x14>
+  DB  235,173                             ; jmp           29c4 <_sk_load_g8_hsw+0x14>
 
 PUBLIC _sk_gather_g8_hsw
 _sk_gather_g8_hsw LABEL PROC
@@ -2563,10 +2568,10 @@ _sk_gather_g8_hsw LABEL PROC
   DB  196,227,121,32,192,7                ; vpinsrb       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,68,29,0,0         ; vbroadcastss  0x1d44(%rip),%ymm1        # 4804 <_sk_callback_hsw+0x381>
+  DB  196,226,125,24,13,64,29,0,0         ; vbroadcastss  0x1d40(%rip),%ymm1        # 4814 <_sk_callback_hsw+0x37d>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,57,29,0,0         ; vbroadcastss  0x1d39(%rip),%ymm3        # 4808 <_sk_callback_hsw+0x385>
+  DB  196,226,125,24,29,53,29,0,0         ; vbroadcastss  0x1d35(%rip),%ymm3        # 4818 <_sk_callback_hsw+0x381>
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
   DB  91                                  ; pop           %rbx
@@ -2580,9 +2585,9 @@ _sk_gather_i8_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            2aef <_sk_gather_i8_hsw+0xf>
+  DB  116,5                               ; je            2b03 <_sk_gather_i8_hsw+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           2af1 <_sk_gather_i8_hsw+0x11>
+  DB  235,2                               ; jmp           2b05 <_sk_gather_i8_hsw+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,87                               ; push          %r15
   DB  65,86                               ; push          %r14
@@ -2620,14 +2625,14 @@ _sk_gather_i8_hsw LABEL PROC
   DB  73,139,64,8                         ; mov           0x8(%r8),%rax
   DB  197,245,118,201                     ; vpcmpeqd      %ymm1,%ymm1,%ymm1
   DB  196,226,117,144,28,128              ; vpgatherdd    %ymm1,(%rax,%ymm0,4),%ymm3
-  DB  197,229,219,5,65,30,0,0             ; vpand         0x1e41(%rip),%ymm3,%ymm0        # 49e0 <_sk_callback_hsw+0x55d>
+  DB  197,229,219,5,77,30,0,0             ; vpand         0x1e4d(%rip),%ymm3,%ymm0        # 4a00 <_sk_callback_hsw+0x569>
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,96,28,0,0           ; vbroadcastss  0x1c60(%rip),%ymm8        # 480c <_sk_callback_hsw+0x389>
+  DB  196,98,125,24,5,92,28,0,0           ; vbroadcastss  0x1c5c(%rip),%ymm8        # 481c <_sk_callback_hsw+0x385>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,226,101,0,13,70,30,0,0          ; vpshufb       0x1e46(%rip),%ymm3,%ymm1        # 4a00 <_sk_callback_hsw+0x57d>
+  DB  196,226,101,0,13,82,30,0,0          ; vpshufb       0x1e52(%rip),%ymm3,%ymm1        # 4a20 <_sk_callback_hsw+0x589>
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,226,101,0,21,84,30,0,0          ; vpshufb       0x1e54(%rip),%ymm3,%ymm2        # 4a20 <_sk_callback_hsw+0x59d>
+  DB  196,226,101,0,21,96,30,0,0          ; vpshufb       0x1e60(%rip),%ymm3,%ymm2        # 4a40 <_sk_callback_hsw+0x5a9>
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
@@ -2646,35 +2651,35 @@ _sk_load_565_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,114                             ; jne           2c6c <_sk_load_565_hsw+0x7c>
+  DB  117,114                             ; jne           2c80 <_sk_load_565_hsw+0x7c>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  196,226,125,51,208                  ; vpmovzxwd     %xmm0,%ymm2
-  DB  196,226,125,88,5,2,28,0,0           ; vpbroadcastd  0x1c02(%rip),%ymm0        # 4810 <_sk_callback_hsw+0x38d>
+  DB  196,226,125,88,5,254,27,0,0         ; vpbroadcastd  0x1bfe(%rip),%ymm0        # 4820 <_sk_callback_hsw+0x389>
   DB  197,237,219,192                     ; vpand         %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,245,27,0,0        ; vbroadcastss  0x1bf5(%rip),%ymm1        # 4814 <_sk_callback_hsw+0x391>
+  DB  196,226,125,24,13,241,27,0,0        ; vbroadcastss  0x1bf1(%rip),%ymm1        # 4824 <_sk_callback_hsw+0x38d>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,236,27,0,0        ; vpbroadcastd  0x1bec(%rip),%ymm1        # 4818 <_sk_callback_hsw+0x395>
+  DB  196,226,125,88,13,232,27,0,0        ; vpbroadcastd  0x1be8(%rip),%ymm1        # 4828 <_sk_callback_hsw+0x391>
   DB  197,237,219,201                     ; vpand         %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,223,27,0,0        ; vbroadcastss  0x1bdf(%rip),%ymm3        # 481c <_sk_callback_hsw+0x399>
+  DB  196,226,125,24,29,219,27,0,0        ; vbroadcastss  0x1bdb(%rip),%ymm3        # 482c <_sk_callback_hsw+0x395>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,88,29,214,27,0,0        ; vpbroadcastd  0x1bd6(%rip),%ymm3        # 4820 <_sk_callback_hsw+0x39d>
+  DB  196,226,125,88,29,210,27,0,0        ; vpbroadcastd  0x1bd2(%rip),%ymm3        # 4830 <_sk_callback_hsw+0x399>
   DB  197,237,219,211                     ; vpand         %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,201,27,0,0        ; vbroadcastss  0x1bc9(%rip),%ymm3        # 4824 <_sk_callback_hsw+0x3a1>
+  DB  196,226,125,24,29,197,27,0,0        ; vbroadcastss  0x1bc5(%rip),%ymm3        # 4834 <_sk_callback_hsw+0x39d>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,190,27,0,0        ; vbroadcastss  0x1bbe(%rip),%ymm3        # 4828 <_sk_callback_hsw+0x3a5>
+  DB  196,226,125,24,29,186,27,0,0        ; vbroadcastss  0x1bba(%rip),%ymm3        # 4838 <_sk_callback_hsw+0x3a1>
   DB  255,224                             ; jmpq          *%rax
   DB  65,137,200                          ; mov           %ecx,%r8d
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,128                             ; ja            2c00 <_sk_load_565_hsw+0x10>
+  DB  119,128                             ; ja            2c14 <_sk_load_565_hsw+0x10>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 2cd4 <_sk_load_565_hsw+0xe4>
+  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 2ce8 <_sk_load_565_hsw+0xe4>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2686,7 +2691,7 @@ _sk_load_565_hsw LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,44,255,255,255                  ; jmpq          2c00 <_sk_load_565_hsw+0x10>
+  DB  233,44,255,255,255                  ; jmpq          2c14 <_sk_load_565_hsw+0x10>
   DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -2754,23 +2759,23 @@ _sk_gather_565_hsw LABEL PROC
   DB  65,15,183,4,88                      ; movzwl        (%r8,%rbx,2),%eax
   DB  197,249,196,192,7                   ; vpinsrw       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,51,208                  ; vpmovzxwd     %xmm0,%ymm2
-  DB  196,226,125,88,5,129,26,0,0         ; vpbroadcastd  0x1a81(%rip),%ymm0        # 482c <_sk_callback_hsw+0x3a9>
+  DB  196,226,125,88,5,125,26,0,0         ; vpbroadcastd  0x1a7d(%rip),%ymm0        # 483c <_sk_callback_hsw+0x3a5>
   DB  197,237,219,192                     ; vpand         %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,116,26,0,0        ; vbroadcastss  0x1a74(%rip),%ymm1        # 4830 <_sk_callback_hsw+0x3ad>
+  DB  196,226,125,24,13,112,26,0,0        ; vbroadcastss  0x1a70(%rip),%ymm1        # 4840 <_sk_callback_hsw+0x3a9>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,107,26,0,0        ; vpbroadcastd  0x1a6b(%rip),%ymm1        # 4834 <_sk_callback_hsw+0x3b1>
+  DB  196,226,125,88,13,103,26,0,0        ; vpbroadcastd  0x1a67(%rip),%ymm1        # 4844 <_sk_callback_hsw+0x3ad>
   DB  197,237,219,201                     ; vpand         %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,94,26,0,0         ; vbroadcastss  0x1a5e(%rip),%ymm3        # 4838 <_sk_callback_hsw+0x3b5>
+  DB  196,226,125,24,29,90,26,0,0         ; vbroadcastss  0x1a5a(%rip),%ymm3        # 4848 <_sk_callback_hsw+0x3b1>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,88,29,85,26,0,0         ; vpbroadcastd  0x1a55(%rip),%ymm3        # 483c <_sk_callback_hsw+0x3b9>
+  DB  196,226,125,88,29,81,26,0,0         ; vpbroadcastd  0x1a51(%rip),%ymm3        # 484c <_sk_callback_hsw+0x3b5>
   DB  197,237,219,211                     ; vpand         %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,72,26,0,0         ; vbroadcastss  0x1a48(%rip),%ymm3        # 4840 <_sk_callback_hsw+0x3bd>
+  DB  196,226,125,24,29,68,26,0,0         ; vbroadcastss  0x1a44(%rip),%ymm3        # 4850 <_sk_callback_hsw+0x3b9>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,61,26,0,0         ; vbroadcastss  0x1a3d(%rip),%ymm3        # 4844 <_sk_callback_hsw+0x3c1>
+  DB  196,226,125,24,29,57,26,0,0         ; vbroadcastss  0x1a39(%rip),%ymm3        # 4854 <_sk_callback_hsw+0x3bd>
   DB  91                                  ; pop           %rbx
   DB  65,92                               ; pop           %r12
   DB  65,94                               ; pop           %r14
@@ -2781,11 +2786,11 @@ PUBLIC _sk_store_565_hsw
 _sk_store_565_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,42,26,0,0           ; vbroadcastss  0x1a2a(%rip),%ymm8        # 4848 <_sk_callback_hsw+0x3c5>
+  DB  196,98,125,24,5,38,26,0,0           ; vbroadcastss  0x1a26(%rip),%ymm8        # 4858 <_sk_callback_hsw+0x3c1>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,53,114,241,11               ; vpslld        $0xb,%ymm9,%ymm9
-  DB  196,98,125,24,21,21,26,0,0          ; vbroadcastss  0x1a15(%rip),%ymm10        # 484c <_sk_callback_hsw+0x3c9>
+  DB  196,98,125,24,21,17,26,0,0          ; vbroadcastss  0x1a11(%rip),%ymm10        # 485c <_sk_callback_hsw+0x3c5>
   DB  196,65,116,89,210                   ; vmulps        %ymm10,%ymm1,%ymm10
   DB  196,65,125,91,210                   ; vcvtps2dq     %ymm10,%ymm10
   DB  196,193,45,114,242,5                ; vpslld        $0x5,%ymm10,%ymm10
@@ -2796,7 +2801,7 @@ _sk_store_565_hsw LABEL PROC
   DB  196,67,125,57,193,1                 ; vextracti128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           2e75 <_sk_store_565_hsw+0x65>
+  DB  117,10                              ; jne           2e89 <_sk_store_565_hsw+0x65>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2804,9 +2809,9 @@ _sk_store_565_hsw LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            2e71 <_sk_store_565_hsw+0x61>
+  DB  119,236                             ; ja            2e85 <_sk_store_565_hsw+0x61>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 2ed4 <_sk_store_565_hsw+0xc4>
+  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 2ee8 <_sk_store_565_hsw+0xc4>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2817,7 +2822,7 @@ _sk_store_565_hsw LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           2e71 <_sk_store_565_hsw+0x61>
+  DB  235,159                             ; jmp           2e85 <_sk_store_565_hsw+0x61>
   DB  102,144                             ; xchg          %ax,%ax
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -2848,28 +2853,28 @@ _sk_load_4444_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,138,0,0,0                    ; jne           2f88 <_sk_load_4444_hsw+0x98>
+  DB  15,133,138,0,0,0                    ; jne           2f9c <_sk_load_4444_hsw+0x98>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  196,226,125,51,216                  ; vpmovzxwd     %xmm0,%ymm3
-  DB  196,226,125,88,5,62,25,0,0          ; vpbroadcastd  0x193e(%rip),%ymm0        # 4850 <_sk_callback_hsw+0x3cd>
+  DB  196,226,125,88,5,58,25,0,0          ; vpbroadcastd  0x193a(%rip),%ymm0        # 4860 <_sk_callback_hsw+0x3c9>
   DB  197,229,219,192                     ; vpand         %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,49,25,0,0         ; vbroadcastss  0x1931(%rip),%ymm1        # 4854 <_sk_callback_hsw+0x3d1>
+  DB  196,226,125,24,13,45,25,0,0         ; vbroadcastss  0x192d(%rip),%ymm1        # 4864 <_sk_callback_hsw+0x3cd>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,40,25,0,0         ; vpbroadcastd  0x1928(%rip),%ymm1        # 4858 <_sk_callback_hsw+0x3d5>
+  DB  196,226,125,88,13,36,25,0,0         ; vpbroadcastd  0x1924(%rip),%ymm1        # 4868 <_sk_callback_hsw+0x3d1>
   DB  197,229,219,201                     ; vpand         %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,27,25,0,0         ; vbroadcastss  0x191b(%rip),%ymm2        # 485c <_sk_callback_hsw+0x3d9>
+  DB  196,226,125,24,21,23,25,0,0         ; vbroadcastss  0x1917(%rip),%ymm2        # 486c <_sk_callback_hsw+0x3d5>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,88,21,18,25,0,0         ; vpbroadcastd  0x1912(%rip),%ymm2        # 4860 <_sk_callback_hsw+0x3dd>
+  DB  196,226,125,88,21,14,25,0,0         ; vpbroadcastd  0x190e(%rip),%ymm2        # 4870 <_sk_callback_hsw+0x3d9>
   DB  197,229,219,210                     ; vpand         %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,5,25,0,0            ; vbroadcastss  0x1905(%rip),%ymm8        # 4864 <_sk_callback_hsw+0x3e1>
+  DB  196,98,125,24,5,1,25,0,0            ; vbroadcastss  0x1901(%rip),%ymm8        # 4874 <_sk_callback_hsw+0x3dd>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,88,5,251,24,0,0          ; vpbroadcastd  0x18fb(%rip),%ymm8        # 4868 <_sk_callback_hsw+0x3e5>
+  DB  196,98,125,88,5,247,24,0,0          ; vpbroadcastd  0x18f7(%rip),%ymm8        # 4878 <_sk_callback_hsw+0x3e1>
   DB  196,193,101,219,216                 ; vpand         %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,237,24,0,0          ; vbroadcastss  0x18ed(%rip),%ymm8        # 486c <_sk_callback_hsw+0x3e9>
+  DB  196,98,125,24,5,233,24,0,0          ; vbroadcastss  0x18e9(%rip),%ymm8        # 487c <_sk_callback_hsw+0x3e5>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2878,9 +2883,9 @@ _sk_load_4444_hsw LABEL PROC
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,100,255,255,255              ; ja            2f04 <_sk_load_4444_hsw+0x14>
+  DB  15,135,100,255,255,255              ; ja            2f18 <_sk_load_4444_hsw+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 2ff4 <_sk_load_4444_hsw+0x104>
+  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 3008 <_sk_load_4444_hsw+0x104>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2892,7 +2897,7 @@ _sk_load_4444_hsw LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,16,255,255,255                  ; jmpq          2f04 <_sk_load_4444_hsw+0x14>
+  DB  233,16,255,255,255                  ; jmpq          2f18 <_sk_load_4444_hsw+0x14>
   DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -2960,25 +2965,25 @@ _sk_gather_4444_hsw LABEL PROC
   DB  65,15,183,4,88                      ; movzwl        (%r8,%rbx,2),%eax
   DB  197,249,196,192,7                   ; vpinsrw       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,51,216                  ; vpmovzxwd     %xmm0,%ymm3
-  DB  196,226,125,88,5,165,23,0,0         ; vpbroadcastd  0x17a5(%rip),%ymm0        # 4870 <_sk_callback_hsw+0x3ed>
+  DB  196,226,125,88,5,161,23,0,0         ; vpbroadcastd  0x17a1(%rip),%ymm0        # 4880 <_sk_callback_hsw+0x3e9>
   DB  197,229,219,192                     ; vpand         %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,152,23,0,0        ; vbroadcastss  0x1798(%rip),%ymm1        # 4874 <_sk_callback_hsw+0x3f1>
+  DB  196,226,125,24,13,148,23,0,0        ; vbroadcastss  0x1794(%rip),%ymm1        # 4884 <_sk_callback_hsw+0x3ed>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,143,23,0,0        ; vpbroadcastd  0x178f(%rip),%ymm1        # 4878 <_sk_callback_hsw+0x3f5>
+  DB  196,226,125,88,13,139,23,0,0        ; vpbroadcastd  0x178b(%rip),%ymm1        # 4888 <_sk_callback_hsw+0x3f1>
   DB  197,229,219,201                     ; vpand         %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,130,23,0,0        ; vbroadcastss  0x1782(%rip),%ymm2        # 487c <_sk_callback_hsw+0x3f9>
+  DB  196,226,125,24,21,126,23,0,0        ; vbroadcastss  0x177e(%rip),%ymm2        # 488c <_sk_callback_hsw+0x3f5>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,88,21,121,23,0,0        ; vpbroadcastd  0x1779(%rip),%ymm2        # 4880 <_sk_callback_hsw+0x3fd>
+  DB  196,226,125,88,21,117,23,0,0        ; vpbroadcastd  0x1775(%rip),%ymm2        # 4890 <_sk_callback_hsw+0x3f9>
   DB  197,229,219,210                     ; vpand         %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,108,23,0,0          ; vbroadcastss  0x176c(%rip),%ymm8        # 4884 <_sk_callback_hsw+0x401>
+  DB  196,98,125,24,5,104,23,0,0          ; vbroadcastss  0x1768(%rip),%ymm8        # 4894 <_sk_callback_hsw+0x3fd>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,88,5,98,23,0,0           ; vpbroadcastd  0x1762(%rip),%ymm8        # 4888 <_sk_callback_hsw+0x405>
+  DB  196,98,125,88,5,94,23,0,0           ; vpbroadcastd  0x175e(%rip),%ymm8        # 4898 <_sk_callback_hsw+0x401>
   DB  196,193,101,219,216                 ; vpand         %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,84,23,0,0           ; vbroadcastss  0x1754(%rip),%ymm8        # 488c <_sk_callback_hsw+0x409>
+  DB  196,98,125,24,5,80,23,0,0           ; vbroadcastss  0x1750(%rip),%ymm8        # 489c <_sk_callback_hsw+0x405>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -2991,7 +2996,7 @@ PUBLIC _sk_store_4444_hsw
 _sk_store_4444_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,58,23,0,0           ; vbroadcastss  0x173a(%rip),%ymm8        # 4890 <_sk_callback_hsw+0x40d>
+  DB  196,98,125,24,5,54,23,0,0           ; vbroadcastss  0x1736(%rip),%ymm8        # 48a0 <_sk_callback_hsw+0x409>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,53,114,241,12               ; vpslld        $0xc,%ymm9,%ymm9
@@ -3009,7 +3014,7 @@ _sk_store_4444_hsw LABEL PROC
   DB  196,67,125,57,193,1                 ; vextracti128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           31b9 <_sk_store_4444_hsw+0x71>
+  DB  117,10                              ; jne           31cd <_sk_store_4444_hsw+0x71>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -3017,9 +3022,9 @@ _sk_store_4444_hsw LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            31b5 <_sk_store_4444_hsw+0x6d>
+  DB  119,236                             ; ja            31c9 <_sk_store_4444_hsw+0x6d>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3218 <_sk_store_4444_hsw+0xd0>
+  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 322c <_sk_store_4444_hsw+0xd0>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -3030,7 +3035,7 @@ _sk_store_4444_hsw LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           31b5 <_sk_store_4444_hsw+0x6d>
+  DB  235,159                             ; jmp           31c9 <_sk_store_4444_hsw+0x6d>
   DB  102,144                             ; xchg          %ax,%ax
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -3063,16 +3068,16 @@ _sk_load_8888_hsw LABEL PROC
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  76,3,8                              ; add           (%rax),%r9
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,88                              ; jne           32a1 <_sk_load_8888_hsw+0x6d>
+  DB  117,88                              ; jne           32b5 <_sk_load_8888_hsw+0x6d>
   DB  196,193,126,111,25                  ; vmovdqu       (%r9),%ymm3
-  DB  197,229,219,5,234,23,0,0            ; vpand         0x17ea(%rip),%ymm3,%ymm0        # 4a40 <_sk_callback_hsw+0x5bd>
+  DB  197,229,219,5,246,23,0,0            ; vpand         0x17f6(%rip),%ymm3,%ymm0        # 4a60 <_sk_callback_hsw+0x5c9>
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,49,22,0,0           ; vbroadcastss  0x1631(%rip),%ymm8        # 4894 <_sk_callback_hsw+0x411>
+  DB  196,98,125,24,5,45,22,0,0           ; vbroadcastss  0x162d(%rip),%ymm8        # 48a4 <_sk_callback_hsw+0x40d>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,226,101,0,13,239,23,0,0         ; vpshufb       0x17ef(%rip),%ymm3,%ymm1        # 4a60 <_sk_callback_hsw+0x5dd>
+  DB  196,226,101,0,13,251,23,0,0         ; vpshufb       0x17fb(%rip),%ymm3,%ymm1        # 4a80 <_sk_callback_hsw+0x5e9>
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,226,101,0,21,253,23,0,0         ; vpshufb       0x17fd(%rip),%ymm3,%ymm2        # 4a80 <_sk_callback_hsw+0x5fd>
+  DB  196,226,101,0,21,9,24,0,0           ; vpshufb       0x1809(%rip),%ymm3,%ymm2        # 4aa0 <_sk_callback_hsw+0x609>
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
@@ -3089,7 +3094,7 @@ _sk_load_8888_hsw LABEL PROC
   DB  196,225,249,110,192                 ; vmovq         %rax,%xmm0
   DB  196,226,125,33,192                  ; vpmovsxbd     %xmm0,%ymm0
   DB  196,194,125,140,25                  ; vpmaskmovd    (%r9),%ymm0,%ymm3
-  DB  235,135                             ; jmp           324e <_sk_load_8888_hsw+0x1a>
+  DB  235,135                             ; jmp           3262 <_sk_load_8888_hsw+0x1a>
 
 PUBLIC _sk_gather_8888_hsw
 _sk_gather_8888_hsw LABEL PROC
@@ -3102,14 +3107,14 @@ _sk_gather_8888_hsw LABEL PROC
   DB  197,245,254,192                     ; vpaddd        %ymm0,%ymm1,%ymm0
   DB  197,245,118,201                     ; vpcmpeqd      %ymm1,%ymm1,%ymm1
   DB  196,194,117,144,28,128              ; vpgatherdd    %ymm1,(%r8,%ymm0,4),%ymm3
-  DB  197,229,219,5,171,23,0,0            ; vpand         0x17ab(%rip),%ymm3,%ymm0        # 4aa0 <_sk_callback_hsw+0x61d>
+  DB  197,229,219,5,183,23,0,0            ; vpand         0x17b7(%rip),%ymm3,%ymm0        # 4ac0 <_sk_callback_hsw+0x629>
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,150,21,0,0          ; vbroadcastss  0x1596(%rip),%ymm8        # 4898 <_sk_callback_hsw+0x415>
+  DB  196,98,125,24,5,146,21,0,0          ; vbroadcastss  0x1592(%rip),%ymm8        # 48a8 <_sk_callback_hsw+0x411>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,226,101,0,13,176,23,0,0         ; vpshufb       0x17b0(%rip),%ymm3,%ymm1        # 4ac0 <_sk_callback_hsw+0x63d>
+  DB  196,226,101,0,13,188,23,0,0         ; vpshufb       0x17bc(%rip),%ymm3,%ymm1        # 4ae0 <_sk_callback_hsw+0x649>
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,226,101,0,21,190,23,0,0         ; vpshufb       0x17be(%rip),%ymm3,%ymm2        # 4ae0 <_sk_callback_hsw+0x65d>
+  DB  196,226,101,0,21,202,23,0,0         ; vpshufb       0x17ca(%rip),%ymm3,%ymm2        # 4b00 <_sk_callback_hsw+0x669>
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
@@ -3124,7 +3129,7 @@ _sk_store_8888_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  76,3,8                              ; add           (%rax),%r9
-  DB  196,98,125,24,5,70,21,0,0           ; vbroadcastss  0x1546(%rip),%ymm8        # 489c <_sk_callback_hsw+0x419>
+  DB  196,98,125,24,5,66,21,0,0           ; vbroadcastss  0x1542(%rip),%ymm8        # 48ac <_sk_callback_hsw+0x415>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,65,116,89,208                   ; vmulps        %ymm8,%ymm1,%ymm10
@@ -3140,7 +3145,7 @@ _sk_store_8888_hsw LABEL PROC
   DB  196,65,45,235,192                   ; vpor          %ymm8,%ymm10,%ymm8
   DB  196,65,53,235,192                   ; vpor          %ymm8,%ymm9,%ymm8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,12                              ; jne           33b0 <_sk_store_8888_hsw+0x73>
+  DB  117,12                              ; jne           33c4 <_sk_store_8888_hsw+0x73>
   DB  196,65,126,127,1                    ; vmovdqu       %ymm8,(%r9)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,137,193                          ; mov           %r8,%rcx
@@ -3153,14 +3158,14 @@ _sk_store_8888_hsw LABEL PROC
   DB  196,97,249,110,200                  ; vmovq         %rax,%xmm9
   DB  196,66,125,33,201                   ; vpmovsxbd     %xmm9,%ymm9
   DB  196,66,53,142,1                     ; vpmaskmovd    %ymm8,%ymm9,(%r9)
-  DB  235,211                             ; jmp           33a9 <_sk_store_8888_hsw+0x6c>
+  DB  235,211                             ; jmp           33bd <_sk_store_8888_hsw+0x6c>
 
 PUBLIC _sk_load_f16_hsw
 _sk_load_f16_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,97                              ; jne           3441 <_sk_load_f16_hsw+0x6b>
+  DB  117,97                              ; jne           3455 <_sk_load_f16_hsw+0x6b>
   DB  197,121,16,4,248                    ; vmovupd       (%rax,%rdi,8),%xmm8
   DB  197,249,16,84,248,16                ; vmovupd       0x10(%rax,%rdi,8),%xmm2
   DB  197,249,16,92,248,32                ; vmovupd       0x20(%rax,%rdi,8),%xmm3
@@ -3186,29 +3191,29 @@ _sk_load_f16_hsw LABEL PROC
   DB  197,123,16,4,248                    ; vmovsd        (%rax,%rdi,8),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,79                              ; je            34a0 <_sk_load_f16_hsw+0xca>
+  DB  116,79                              ; je            34b4 <_sk_load_f16_hsw+0xca>
   DB  197,57,22,68,248,8                  ; vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,67                              ; jb            34a0 <_sk_load_f16_hsw+0xca>
+  DB  114,67                              ; jb            34b4 <_sk_load_f16_hsw+0xca>
   DB  197,251,16,84,248,16                ; vmovsd        0x10(%rax,%rdi,8),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,68                              ; je            34ad <_sk_load_f16_hsw+0xd7>
+  DB  116,68                              ; je            34c1 <_sk_load_f16_hsw+0xd7>
   DB  197,233,22,84,248,24                ; vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,56                              ; jb            34ad <_sk_load_f16_hsw+0xd7>
+  DB  114,56                              ; jb            34c1 <_sk_load_f16_hsw+0xd7>
   DB  197,251,16,92,248,32                ; vmovsd        0x20(%rax,%rdi,8),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,114,255,255,255              ; je            33f7 <_sk_load_f16_hsw+0x21>
+  DB  15,132,114,255,255,255              ; je            340b <_sk_load_f16_hsw+0x21>
   DB  197,225,22,92,248,40                ; vmovhpd       0x28(%rax,%rdi,8),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,98,255,255,255               ; jb            33f7 <_sk_load_f16_hsw+0x21>
+  DB  15,130,98,255,255,255               ; jb            340b <_sk_load_f16_hsw+0x21>
   DB  197,122,126,76,248,48               ; vmovq         0x30(%rax,%rdi,8),%xmm9
-  DB  233,87,255,255,255                  ; jmpq          33f7 <_sk_load_f16_hsw+0x21>
+  DB  233,87,255,255,255                  ; jmpq          340b <_sk_load_f16_hsw+0x21>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,74,255,255,255                  ; jmpq          33f7 <_sk_load_f16_hsw+0x21>
+  DB  233,74,255,255,255                  ; jmpq          340b <_sk_load_f16_hsw+0x21>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,65,255,255,255                  ; jmpq          33f7 <_sk_load_f16_hsw+0x21>
+  DB  233,65,255,255,255                  ; jmpq          340b <_sk_load_f16_hsw+0x21>
 
 PUBLIC _sk_gather_f16_hsw
 _sk_gather_f16_hsw LABEL PROC
@@ -3262,7 +3267,7 @@ _sk_store_f16_hsw LABEL PROC
   DB  196,65,57,98,205                    ; vpunpckldq    %xmm13,%xmm8,%xmm9
   DB  196,65,57,106,197                   ; vpunpckhdq    %xmm13,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,27                              ; jne           35a5 <_sk_store_f16_hsw+0x65>
+  DB  117,27                              ; jne           35b9 <_sk_store_f16_hsw+0x65>
   DB  197,120,17,28,248                   ; vmovups       %xmm11,(%rax,%rdi,8)
   DB  197,120,17,84,248,16                ; vmovups       %xmm10,0x10(%rax,%rdi,8)
   DB  197,120,17,76,248,32                ; vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -3271,22 +3276,22 @@ _sk_store_f16_hsw LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  197,121,214,28,248                  ; vmovq         %xmm11,(%rax,%rdi,8)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,241                             ; je            35a1 <_sk_store_f16_hsw+0x61>
+  DB  116,241                             ; je            35b5 <_sk_store_f16_hsw+0x61>
   DB  197,121,23,92,248,8                 ; vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,229                             ; jb            35a1 <_sk_store_f16_hsw+0x61>
+  DB  114,229                             ; jb            35b5 <_sk_store_f16_hsw+0x61>
   DB  197,121,214,84,248,16               ; vmovq         %xmm10,0x10(%rax,%rdi,8)
-  DB  116,221                             ; je            35a1 <_sk_store_f16_hsw+0x61>
+  DB  116,221                             ; je            35b5 <_sk_store_f16_hsw+0x61>
   DB  197,121,23,84,248,24                ; vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,209                             ; jb            35a1 <_sk_store_f16_hsw+0x61>
+  DB  114,209                             ; jb            35b5 <_sk_store_f16_hsw+0x61>
   DB  197,121,214,76,248,32               ; vmovq         %xmm9,0x20(%rax,%rdi,8)
-  DB  116,201                             ; je            35a1 <_sk_store_f16_hsw+0x61>
+  DB  116,201                             ; je            35b5 <_sk_store_f16_hsw+0x61>
   DB  197,121,23,76,248,40                ; vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,189                             ; jb            35a1 <_sk_store_f16_hsw+0x61>
+  DB  114,189                             ; jb            35b5 <_sk_store_f16_hsw+0x61>
   DB  197,121,214,68,248,48               ; vmovq         %xmm8,0x30(%rax,%rdi,8)
-  DB  235,181                             ; jmp           35a1 <_sk_store_f16_hsw+0x61>
+  DB  235,181                             ; jmp           35b5 <_sk_store_f16_hsw+0x61>
 
 PUBLIC _sk_load_u16_be_hsw
 _sk_load_u16_be_hsw LABEL PROC
@@ -3294,7 +3299,7 @@ _sk_load_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,204,0,0,0                    ; jne           36ce <_sk_load_u16_be_hsw+0xe2>
+  DB  15,133,204,0,0,0                    ; jne           36e2 <_sk_load_u16_be_hsw+0xe2>
   DB  196,65,121,16,4,64                  ; vmovupd       (%r8,%rax,2),%xmm8
   DB  196,193,121,16,84,64,16             ; vmovupd       0x10(%r8,%rax,2),%xmm2
   DB  196,193,121,16,92,64,32             ; vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -3313,7 +3318,7 @@ _sk_load_u16_be_hsw LABEL PROC
   DB  197,241,235,192                     ; vpor          %xmm0,%xmm1,%xmm0
   DB  196,226,125,51,192                  ; vpmovzxwd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,21,61,18,0,0          ; vbroadcastss  0x123d(%rip),%ymm10        # 48a0 <_sk_callback_hsw+0x41d>
+  DB  196,98,125,24,21,57,18,0,0          ; vbroadcastss  0x1239(%rip),%ymm10        # 48b0 <_sk_callback_hsw+0x419>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,185,109,202                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -3341,29 +3346,29 @@ _sk_load_u16_be_hsw LABEL PROC
   DB  196,65,123,16,4,64                  ; vmovsd        (%r8,%rax,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            3734 <_sk_load_u16_be_hsw+0x148>
+  DB  116,85                              ; je            3748 <_sk_load_u16_be_hsw+0x148>
   DB  196,65,57,22,68,64,8                ; vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            3734 <_sk_load_u16_be_hsw+0x148>
+  DB  114,72                              ; jb            3748 <_sk_load_u16_be_hsw+0x148>
   DB  196,193,123,16,84,64,16             ; vmovsd        0x10(%r8,%rax,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            3741 <_sk_load_u16_be_hsw+0x155>
+  DB  116,72                              ; je            3755 <_sk_load_u16_be_hsw+0x155>
   DB  196,193,105,22,84,64,24             ; vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            3741 <_sk_load_u16_be_hsw+0x155>
+  DB  114,59                              ; jb            3755 <_sk_load_u16_be_hsw+0x155>
   DB  196,193,123,16,92,64,32             ; vmovsd        0x20(%r8,%rax,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,6,255,255,255                ; je            361d <_sk_load_u16_be_hsw+0x31>
+  DB  15,132,6,255,255,255                ; je            3631 <_sk_load_u16_be_hsw+0x31>
   DB  196,193,97,22,92,64,40              ; vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,245,254,255,255              ; jb            361d <_sk_load_u16_be_hsw+0x31>
+  DB  15,130,245,254,255,255              ; jb            3631 <_sk_load_u16_be_hsw+0x31>
   DB  196,65,122,126,76,64,48             ; vmovq         0x30(%r8,%rax,2),%xmm9
-  DB  233,233,254,255,255                 ; jmpq          361d <_sk_load_u16_be_hsw+0x31>
+  DB  233,233,254,255,255                 ; jmpq          3631 <_sk_load_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,220,254,255,255                 ; jmpq          361d <_sk_load_u16_be_hsw+0x31>
+  DB  233,220,254,255,255                 ; jmpq          3631 <_sk_load_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,211,254,255,255                 ; jmpq          361d <_sk_load_u16_be_hsw+0x31>
+  DB  233,211,254,255,255                 ; jmpq          3631 <_sk_load_u16_be_hsw+0x31>
 
 PUBLIC _sk_load_rgb_u16_be_hsw
 _sk_load_rgb_u16_be_hsw LABEL PROC
@@ -3371,7 +3376,7 @@ _sk_load_rgb_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,127                        ; lea           (%rdi,%rdi,2),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,204,0,0,0                    ; jne           3828 <_sk_load_rgb_u16_be_hsw+0xde>
+  DB  15,133,204,0,0,0                    ; jne           383c <_sk_load_rgb_u16_be_hsw+0xde>
   DB  196,193,122,111,4,64                ; vmovdqu       (%r8,%rax,2),%xmm0
   DB  196,193,122,111,84,64,12            ; vmovdqu       0xc(%r8,%rax,2),%xmm2
   DB  196,193,122,111,76,64,24            ; vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -3395,7 +3400,7 @@ _sk_load_rgb_u16_be_hsw LABEL PROC
   DB  197,241,235,192                     ; vpor          %xmm0,%xmm1,%xmm0
   DB  196,226,125,51,192                  ; vpmovzxwd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,21,206,16,0,0         ; vbroadcastss  0x10ce(%rip),%ymm10        # 48a4 <_sk_callback_hsw+0x421>
+  DB  196,98,125,24,21,202,16,0,0         ; vbroadcastss  0x10ca(%rip),%ymm10        # 48b4 <_sk_callback_hsw+0x41d>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,185,109,202                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -3412,48 +3417,48 @@ _sk_load_rgb_u16_be_hsw LABEL PROC
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,130,16,0,0        ; vbroadcastss  0x1082(%rip),%ymm3        # 48a8 <_sk_callback_hsw+0x425>
+  DB  196,226,125,24,29,126,16,0,0        ; vbroadcastss  0x107e(%rip),%ymm3        # 48b8 <_sk_callback_hsw+0x421>
   DB  255,224                             ; jmpq          *%rax
   DB  196,193,121,110,4,64                ; vmovd         (%r8,%rax,2),%xmm0
   DB  196,193,121,196,68,64,4,2           ; vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           3841 <_sk_load_rgb_u16_be_hsw+0xf7>
-  DB  233,79,255,255,255                  ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  117,5                               ; jne           3855 <_sk_load_rgb_u16_be_hsw+0xf7>
+  DB  233,79,255,255,255                  ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
   DB  196,193,121,110,76,64,6             ; vmovd         0x6(%r8,%rax,2),%xmm1
   DB  196,65,113,196,68,64,10,2           ; vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            3870 <_sk_load_rgb_u16_be_hsw+0x126>
+  DB  114,26                              ; jb            3884 <_sk_load_rgb_u16_be_hsw+0x126>
   DB  196,193,121,110,76,64,12            ; vmovd         0xc(%r8,%rax,2),%xmm1
   DB  196,193,113,196,84,64,16,2          ; vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           3875 <_sk_load_rgb_u16_be_hsw+0x12b>
-  DB  233,32,255,255,255                  ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
-  DB  233,27,255,255,255                  ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           3889 <_sk_load_rgb_u16_be_hsw+0x12b>
+  DB  233,32,255,255,255                  ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,27,255,255,255                  ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
   DB  196,193,121,110,76,64,18            ; vmovd         0x12(%r8,%rax,2),%xmm1
   DB  196,65,113,196,76,64,22,2           ; vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            38a4 <_sk_load_rgb_u16_be_hsw+0x15a>
+  DB  114,26                              ; jb            38b8 <_sk_load_rgb_u16_be_hsw+0x15a>
   DB  196,193,121,110,76,64,24            ; vmovd         0x18(%r8,%rax,2),%xmm1
   DB  196,193,113,196,76,64,28,2          ; vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           38a9 <_sk_load_rgb_u16_be_hsw+0x15f>
-  DB  233,236,254,255,255                 ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
-  DB  233,231,254,255,255                 ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           38bd <_sk_load_rgb_u16_be_hsw+0x15f>
+  DB  233,236,254,255,255                 ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,231,254,255,255                 ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
   DB  196,193,121,110,92,64,30            ; vmovd         0x1e(%r8,%rax,2),%xmm3
   DB  196,65,97,196,92,64,34,2            ; vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            38d2 <_sk_load_rgb_u16_be_hsw+0x188>
+  DB  114,20                              ; jb            38e6 <_sk_load_rgb_u16_be_hsw+0x188>
   DB  196,193,121,110,92,64,36            ; vmovd         0x24(%r8,%rax,2),%xmm3
   DB  196,193,97,196,92,64,40,2           ; vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  DB  233,190,254,255,255                 ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
-  DB  233,185,254,255,255                 ; jmpq          3790 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,190,254,255,255                 ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,185,254,255,255                 ; jmpq          37a4 <_sk_load_rgb_u16_be_hsw+0x46>
 
 PUBLIC _sk_store_u16_be_hsw
 _sk_store_u16_be_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
-  DB  196,98,125,24,5,191,15,0,0          ; vbroadcastss  0xfbf(%rip),%ymm8        # 48ac <_sk_callback_hsw+0x429>
+  DB  196,98,125,24,5,187,15,0,0          ; vbroadcastss  0xfbb(%rip),%ymm8        # 48bc <_sk_callback_hsw+0x425>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,67,125,25,202,1                 ; vextractf128  $0x1,%ymm9,%xmm10
@@ -3491,7 +3496,7 @@ _sk_store_u16_be_hsw LABEL PROC
   DB  196,65,17,98,200                    ; vpunpckldq    %xmm8,%xmm13,%xmm9
   DB  196,65,17,106,192                   ; vpunpckhdq    %xmm8,%xmm13,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,31                              ; jne           39d1 <_sk_store_u16_be_hsw+0xfa>
+  DB  117,31                              ; jne           39e5 <_sk_store_u16_be_hsw+0xfa>
   DB  196,65,120,17,28,64                 ; vmovups       %xmm11,(%r8,%rax,2)
   DB  196,65,120,17,84,64,16              ; vmovups       %xmm10,0x10(%r8,%rax,2)
   DB  196,65,120,17,76,64,32              ; vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -3500,31 +3505,31 @@ _sk_store_u16_be_hsw LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,214,28,64                ; vmovq         %xmm11,(%r8,%rax,2)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  116,240                             ; je            39e1 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,23,92,64,8               ; vmovhpd       %xmm11,0x8(%r8,%rax,2)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  114,227                             ; jb            39e1 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,214,84,64,16             ; vmovq         %xmm10,0x10(%r8,%rax,2)
-  DB  116,218                             ; je            39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  116,218                             ; je            39e1 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,23,84,64,24              ; vmovhpd       %xmm10,0x18(%r8,%rax,2)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  114,205                             ; jb            39e1 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,214,76,64,32             ; vmovq         %xmm9,0x20(%r8,%rax,2)
-  DB  116,196                             ; je            39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  116,196                             ; je            39e1 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,23,76,64,40              ; vmovhpd       %xmm9,0x28(%r8,%rax,2)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,183                             ; jb            39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  114,183                             ; jb            39e1 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,214,68,64,48             ; vmovq         %xmm8,0x30(%r8,%rax,2)
-  DB  235,174                             ; jmp           39cd <_sk_store_u16_be_hsw+0xf6>
+  DB  235,174                             ; jmp           39e1 <_sk_store_u16_be_hsw+0xf6>
 
 PUBLIC _sk_load_f32_hsw
 _sk_load_f32_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  119,110                             ; ja            3a95 <_sk_load_f32_hsw+0x76>
+  DB  119,110                             ; ja            3aa9 <_sk_load_f32_hsw+0x76>
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
-  DB  76,141,21,135,0,0,0                 ; lea           0x87(%rip),%r10        # 3ac0 <_sk_load_f32_hsw+0xa1>
+  DB  76,141,21,135,0,0,0                 ; lea           0x87(%rip),%r10        # 3ad4 <_sk_load_f32_hsw+0xa1>
   DB  73,99,4,138                         ; movslq        (%r10,%rcx,4),%rax
   DB  76,1,208                            ; add           %r10,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -3583,7 +3588,7 @@ _sk_store_f32_hsw LABEL PROC
   DB  196,65,37,20,196                    ; vunpcklpd     %ymm12,%ymm11,%ymm8
   DB  196,65,37,21,220                    ; vunpckhpd     %ymm12,%ymm11,%ymm11
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,55                              ; jne           3b4d <_sk_store_f32_hsw+0x6d>
+  DB  117,55                              ; jne           3b61 <_sk_store_f32_hsw+0x6d>
   DB  196,67,45,24,225,1                  ; vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   DB  196,67,61,24,235,1                  ; vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   DB  196,67,45,6,201,49                  ; vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -3596,22 +3601,22 @@ _sk_store_f32_hsw LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,17,20,128                ; vmovupd       %xmm10,(%r8,%rax,4)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            3b49 <_sk_store_f32_hsw+0x69>
+  DB  116,240                             ; je            3b5d <_sk_store_f32_hsw+0x69>
   DB  196,65,121,17,76,128,16             ; vmovupd       %xmm9,0x10(%r8,%rax,4)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            3b49 <_sk_store_f32_hsw+0x69>
+  DB  114,227                             ; jb            3b5d <_sk_store_f32_hsw+0x69>
   DB  196,65,121,17,68,128,32             ; vmovupd       %xmm8,0x20(%r8,%rax,4)
-  DB  116,218                             ; je            3b49 <_sk_store_f32_hsw+0x69>
+  DB  116,218                             ; je            3b5d <_sk_store_f32_hsw+0x69>
   DB  196,65,121,17,92,128,48             ; vmovupd       %xmm11,0x30(%r8,%rax,4)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            3b49 <_sk_store_f32_hsw+0x69>
+  DB  114,205                             ; jb            3b5d <_sk_store_f32_hsw+0x69>
   DB  196,67,125,25,84,128,64,1           ; vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  DB  116,195                             ; je            3b49 <_sk_store_f32_hsw+0x69>
+  DB  116,195                             ; je            3b5d <_sk_store_f32_hsw+0x69>
   DB  196,67,125,25,76,128,80,1           ; vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,181                             ; jb            3b49 <_sk_store_f32_hsw+0x69>
+  DB  114,181                             ; jb            3b5d <_sk_store_f32_hsw+0x69>
   DB  196,67,125,25,68,128,96,1           ; vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  DB  235,171                             ; jmp           3b49 <_sk_store_f32_hsw+0x69>
+  DB  235,171                             ; jmp           3b5d <_sk_store_f32_hsw+0x69>
 
 PUBLIC _sk_clamp_x_hsw
 _sk_clamp_x_hsw LABEL PROC
@@ -3707,11 +3712,11 @@ _sk_mirror_y_hsw LABEL PROC
 
 PUBLIC _sk_luminance_to_alpha_hsw
 _sk_luminance_to_alpha_hsw LABEL PROC
-  DB  196,226,125,24,29,217,11,0,0        ; vbroadcastss  0xbd9(%rip),%ymm3        # 48b0 <_sk_callback_hsw+0x42d>
-  DB  196,98,125,24,5,212,11,0,0          ; vbroadcastss  0xbd4(%rip),%ymm8        # 48b4 <_sk_callback_hsw+0x431>
+  DB  196,226,125,24,29,213,11,0,0        ; vbroadcastss  0xbd5(%rip),%ymm3        # 48c0 <_sk_callback_hsw+0x429>
+  DB  196,98,125,24,5,208,11,0,0          ; vbroadcastss  0xbd0(%rip),%ymm8        # 48c4 <_sk_callback_hsw+0x42d>
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  196,226,125,184,203                 ; vfmadd231ps   %ymm3,%ymm0,%ymm1
-  DB  196,226,125,24,29,197,11,0,0        ; vbroadcastss  0xbc5(%rip),%ymm3        # 48b8 <_sk_callback_hsw+0x435>
+  DB  196,226,125,24,29,193,11,0,0        ; vbroadcastss  0xbc1(%rip),%ymm3        # 48c8 <_sk_callback_hsw+0x431>
   DB  196,226,109,168,217                 ; vfmadd213ps   %ymm1,%ymm2,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -3846,7 +3851,7 @@ _sk_linear_gradient_hsw LABEL PROC
   DB  196,98,125,24,72,28                 ; vbroadcastss  0x1c(%rax),%ymm9
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  15,132,143,0,0,0                    ; je            3fcb <_sk_linear_gradient_hsw+0xb5>
+  DB  15,132,143,0,0,0                    ; je            3fdf <_sk_linear_gradient_hsw+0xb5>
   DB  72,139,64,8                         ; mov           0x8(%rax),%rax
   DB  72,131,192,32                       ; add           $0x20,%rax
   DB  196,65,28,87,228                    ; vxorps        %ymm12,%ymm12,%ymm12
@@ -3873,8 +3878,8 @@ _sk_linear_gradient_hsw LABEL PROC
   DB  196,67,13,74,201,208                ; vblendvps     %ymm13,%ymm9,%ymm14,%ymm9
   DB  72,131,192,36                       ; add           $0x24,%rax
   DB  73,255,200                          ; dec           %r8
-  DB  117,140                             ; jne           3f55 <_sk_linear_gradient_hsw+0x3f>
-  DB  235,17                              ; jmp           3fdc <_sk_linear_gradient_hsw+0xc6>
+  DB  117,140                             ; jne           3f69 <_sk_linear_gradient_hsw+0x3f>
+  DB  235,17                              ; jmp           3ff0 <_sk_linear_gradient_hsw+0xc6>
   DB  197,244,87,201                      ; vxorps        %ymm1,%ymm1,%ymm1
   DB  197,236,87,210                      ; vxorps        %ymm2,%ymm2,%ymm2
   DB  197,228,87,219                      ; vxorps        %ymm3,%ymm3,%ymm3
@@ -3917,24 +3922,24 @@ _sk_xy_to_polar_unit_hsw LABEL PROC
   DB  196,65,52,95,226                    ; vmaxps        %ymm10,%ymm9,%ymm12
   DB  196,65,36,94,220                    ; vdivps        %ymm12,%ymm11,%ymm11
   DB  196,65,36,89,227                    ; vmulps        %ymm11,%ymm11,%ymm12
-  DB  196,98,125,24,45,69,8,0,0           ; vbroadcastss  0x845(%rip),%ymm13        # 48bc <_sk_callback_hsw+0x439>
-  DB  196,98,125,24,53,64,8,0,0           ; vbroadcastss  0x840(%rip),%ymm14        # 48c0 <_sk_callback_hsw+0x43d>
+  DB  196,98,125,24,45,65,8,0,0           ; vbroadcastss  0x841(%rip),%ymm13        # 48cc <_sk_callback_hsw+0x435>
+  DB  196,98,125,24,53,60,8,0,0           ; vbroadcastss  0x83c(%rip),%ymm14        # 48d0 <_sk_callback_hsw+0x439>
   DB  196,66,29,184,245                   ; vfmadd231ps   %ymm13,%ymm12,%ymm14
-  DB  196,98,125,24,45,54,8,0,0           ; vbroadcastss  0x836(%rip),%ymm13        # 48c4 <_sk_callback_hsw+0x441>
+  DB  196,98,125,24,45,50,8,0,0           ; vbroadcastss  0x832(%rip),%ymm13        # 48d4 <_sk_callback_hsw+0x43d>
   DB  196,66,29,184,238                   ; vfmadd231ps   %ymm14,%ymm12,%ymm13
-  DB  196,98,125,24,53,44,8,0,0           ; vbroadcastss  0x82c(%rip),%ymm14        # 48c8 <_sk_callback_hsw+0x445>
+  DB  196,98,125,24,53,40,8,0,0           ; vbroadcastss  0x828(%rip),%ymm14        # 48d8 <_sk_callback_hsw+0x441>
   DB  196,66,29,184,245                   ; vfmadd231ps   %ymm13,%ymm12,%ymm14
   DB  196,65,36,89,222                    ; vmulps        %ymm14,%ymm11,%ymm11
   DB  196,65,52,194,202,1                 ; vcmpltps      %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,21,23,8,0,0           ; vbroadcastss  0x817(%rip),%ymm10        # 48cc <_sk_callback_hsw+0x449>
+  DB  196,98,125,24,21,19,8,0,0           ; vbroadcastss  0x813(%rip),%ymm10        # 48dc <_sk_callback_hsw+0x445>
   DB  196,65,44,92,211                    ; vsubps        %ymm11,%ymm10,%ymm10
   DB  196,67,37,74,202,144                ; vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   DB  196,193,124,194,192,1               ; vcmpltps      %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,21,1,8,0,0            ; vbroadcastss  0x801(%rip),%ymm10        # 48d0 <_sk_callback_hsw+0x44d>
+  DB  196,98,125,24,21,253,7,0,0          ; vbroadcastss  0x7fd(%rip),%ymm10        # 48e0 <_sk_callback_hsw+0x449>
   DB  196,65,44,92,209                    ; vsubps        %ymm9,%ymm10,%ymm10
   DB  196,195,53,74,194,0                 ; vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   DB  196,65,116,194,200,1                ; vcmpltps      %ymm8,%ymm1,%ymm9
-  DB  196,98,125,24,21,235,7,0,0          ; vbroadcastss  0x7eb(%rip),%ymm10        # 48d4 <_sk_callback_hsw+0x451>
+  DB  196,98,125,24,21,231,7,0,0          ; vbroadcastss  0x7e7(%rip),%ymm10        # 48e4 <_sk_callback_hsw+0x44d>
   DB  197,44,92,208                       ; vsubps        %ymm0,%ymm10,%ymm10
   DB  196,195,125,74,194,144              ; vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   DB  196,65,124,194,200,3                ; vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -3954,7 +3959,7 @@ _sk_xy_to_radius_hsw LABEL PROC
 PUBLIC _sk_save_xy_hsw
 _sk_save_xy_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,180,7,0,0           ; vbroadcastss  0x7b4(%rip),%ymm8        # 48d8 <_sk_callback_hsw+0x455>
+  DB  196,98,125,24,5,176,7,0,0           ; vbroadcastss  0x7b0(%rip),%ymm8        # 48e8 <_sk_callback_hsw+0x451>
   DB  196,65,124,88,200                   ; vaddps        %ymm8,%ymm0,%ymm9
   DB  196,67,125,8,209,1                  ; vroundps      $0x1,%ymm9,%ymm10
   DB  196,65,52,92,202                    ; vsubps        %ymm10,%ymm9,%ymm9
@@ -3984,9 +3989,9 @@ _sk_accumulate_hsw LABEL PROC
 PUBLIC _sk_bilinear_nx_hsw
 _sk_bilinear_nx_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,72,7,0,0           ; vbroadcastss  0x748(%rip),%ymm0        # 48dc <_sk_callback_hsw+0x459>
+  DB  196,226,125,24,5,68,7,0,0           ; vbroadcastss  0x744(%rip),%ymm0        # 48ec <_sk_callback_hsw+0x455>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,63,7,0,0            ; vbroadcastss  0x73f(%rip),%ymm8        # 48e0 <_sk_callback_hsw+0x45d>
+  DB  196,98,125,24,5,59,7,0,0            ; vbroadcastss  0x73b(%rip),%ymm8        # 48f0 <_sk_callback_hsw+0x459>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -3995,7 +4000,7 @@ _sk_bilinear_nx_hsw LABEL PROC
 PUBLIC _sk_bilinear_px_hsw
 _sk_bilinear_px_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,39,7,0,0           ; vbroadcastss  0x727(%rip),%ymm0        # 48e4 <_sk_callback_hsw+0x461>
+  DB  196,226,125,24,5,35,7,0,0           ; vbroadcastss  0x723(%rip),%ymm0        # 48f4 <_sk_callback_hsw+0x45d>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -4005,9 +4010,9 @@ _sk_bilinear_px_hsw LABEL PROC
 PUBLIC _sk_bilinear_ny_hsw
 _sk_bilinear_ny_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,11,7,0,0          ; vbroadcastss  0x70b(%rip),%ymm1        # 48e8 <_sk_callback_hsw+0x465>
+  DB  196,226,125,24,13,7,7,0,0           ; vbroadcastss  0x707(%rip),%ymm1        # 48f8 <_sk_callback_hsw+0x461>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,1,7,0,0             ; vbroadcastss  0x701(%rip),%ymm8        # 48ec <_sk_callback_hsw+0x469>
+  DB  196,98,125,24,5,253,6,0,0           ; vbroadcastss  0x6fd(%rip),%ymm8        # 48fc <_sk_callback_hsw+0x465>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4016,7 +4021,7 @@ _sk_bilinear_ny_hsw LABEL PROC
 PUBLIC _sk_bilinear_py_hsw
 _sk_bilinear_py_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,233,6,0,0         ; vbroadcastss  0x6e9(%rip),%ymm1        # 48f0 <_sk_callback_hsw+0x46d>
+  DB  196,226,125,24,13,229,6,0,0         ; vbroadcastss  0x6e5(%rip),%ymm1        # 4900 <_sk_callback_hsw+0x469>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -4026,13 +4031,13 @@ _sk_bilinear_py_hsw LABEL PROC
 PUBLIC _sk_bicubic_n3x_hsw
 _sk_bicubic_n3x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,204,6,0,0          ; vbroadcastss  0x6cc(%rip),%ymm0        # 48f4 <_sk_callback_hsw+0x471>
+  DB  196,226,125,24,5,200,6,0,0          ; vbroadcastss  0x6c8(%rip),%ymm0        # 4904 <_sk_callback_hsw+0x46d>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,195,6,0,0           ; vbroadcastss  0x6c3(%rip),%ymm8        # 48f8 <_sk_callback_hsw+0x475>
+  DB  196,98,125,24,5,191,6,0,0           ; vbroadcastss  0x6bf(%rip),%ymm8        # 4908 <_sk_callback_hsw+0x471>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,180,6,0,0          ; vbroadcastss  0x6b4(%rip),%ymm10        # 48fc <_sk_callback_hsw+0x479>
-  DB  196,98,125,24,29,175,6,0,0          ; vbroadcastss  0x6af(%rip),%ymm11        # 4900 <_sk_callback_hsw+0x47d>
+  DB  196,98,125,24,21,176,6,0,0          ; vbroadcastss  0x6b0(%rip),%ymm10        # 490c <_sk_callback_hsw+0x475>
+  DB  196,98,125,24,29,171,6,0,0          ; vbroadcastss  0x6ab(%rip),%ymm11        # 4910 <_sk_callback_hsw+0x479>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,36,89,193                    ; vmulps        %ymm9,%ymm11,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -4042,16 +4047,16 @@ _sk_bicubic_n3x_hsw LABEL PROC
 PUBLIC _sk_bicubic_n1x_hsw
 _sk_bicubic_n1x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,146,6,0,0          ; vbroadcastss  0x692(%rip),%ymm0        # 4904 <_sk_callback_hsw+0x481>
+  DB  196,226,125,24,5,142,6,0,0          ; vbroadcastss  0x68e(%rip),%ymm0        # 4914 <_sk_callback_hsw+0x47d>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,137,6,0,0           ; vbroadcastss  0x689(%rip),%ymm8        # 4908 <_sk_callback_hsw+0x485>
+  DB  196,98,125,24,5,133,6,0,0           ; vbroadcastss  0x685(%rip),%ymm8        # 4918 <_sk_callback_hsw+0x481>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,127,6,0,0          ; vbroadcastss  0x67f(%rip),%ymm9        # 490c <_sk_callback_hsw+0x489>
-  DB  196,98,125,24,21,122,6,0,0          ; vbroadcastss  0x67a(%rip),%ymm10        # 4910 <_sk_callback_hsw+0x48d>
+  DB  196,98,125,24,13,123,6,0,0          ; vbroadcastss  0x67b(%rip),%ymm9        # 491c <_sk_callback_hsw+0x485>
+  DB  196,98,125,24,21,118,6,0,0          ; vbroadcastss  0x676(%rip),%ymm10        # 4920 <_sk_callback_hsw+0x489>
   DB  196,66,61,168,209                   ; vfmadd213ps   %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,13,112,6,0,0          ; vbroadcastss  0x670(%rip),%ymm9        # 4914 <_sk_callback_hsw+0x491>
+  DB  196,98,125,24,13,108,6,0,0          ; vbroadcastss  0x66c(%rip),%ymm9        # 4924 <_sk_callback_hsw+0x48d>
   DB  196,66,61,184,202                   ; vfmadd231ps   %ymm10,%ymm8,%ymm9
-  DB  196,98,125,24,21,102,6,0,0          ; vbroadcastss  0x666(%rip),%ymm10        # 4918 <_sk_callback_hsw+0x495>
+  DB  196,98,125,24,21,98,6,0,0           ; vbroadcastss  0x662(%rip),%ymm10        # 4928 <_sk_callback_hsw+0x491>
   DB  196,66,61,184,209                   ; vfmadd231ps   %ymm9,%ymm8,%ymm10
   DB  197,124,17,144,128,0,0,0            ; vmovups       %ymm10,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4060,14 +4065,14 @@ _sk_bicubic_n1x_hsw LABEL PROC
 PUBLIC _sk_bicubic_p1x_hsw
 _sk_bicubic_p1x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,78,6,0,0            ; vbroadcastss  0x64e(%rip),%ymm8        # 491c <_sk_callback_hsw+0x499>
+  DB  196,98,125,24,5,74,6,0,0            ; vbroadcastss  0x64a(%rip),%ymm8        # 492c <_sk_callback_hsw+0x495>
   DB  197,188,88,0                        ; vaddps        (%rax),%ymm8,%ymm0
   DB  197,124,16,72,64                    ; vmovups       0x40(%rax),%ymm9
-  DB  196,98,125,24,21,64,6,0,0           ; vbroadcastss  0x640(%rip),%ymm10        # 4920 <_sk_callback_hsw+0x49d>
-  DB  196,98,125,24,29,59,6,0,0           ; vbroadcastss  0x63b(%rip),%ymm11        # 4924 <_sk_callback_hsw+0x4a1>
+  DB  196,98,125,24,21,60,6,0,0           ; vbroadcastss  0x63c(%rip),%ymm10        # 4930 <_sk_callback_hsw+0x499>
+  DB  196,98,125,24,29,55,6,0,0           ; vbroadcastss  0x637(%rip),%ymm11        # 4934 <_sk_callback_hsw+0x49d>
   DB  196,66,53,168,218                   ; vfmadd213ps   %ymm10,%ymm9,%ymm11
   DB  196,66,53,168,216                   ; vfmadd213ps   %ymm8,%ymm9,%ymm11
-  DB  196,98,125,24,5,44,6,0,0            ; vbroadcastss  0x62c(%rip),%ymm8        # 4928 <_sk_callback_hsw+0x4a5>
+  DB  196,98,125,24,5,40,6,0,0            ; vbroadcastss  0x628(%rip),%ymm8        # 4938 <_sk_callback_hsw+0x4a1>
   DB  196,66,53,184,195                   ; vfmadd231ps   %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4076,12 +4081,12 @@ _sk_bicubic_p1x_hsw LABEL PROC
 PUBLIC _sk_bicubic_p3x_hsw
 _sk_bicubic_p3x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,20,6,0,0           ; vbroadcastss  0x614(%rip),%ymm0        # 492c <_sk_callback_hsw+0x4a9>
+  DB  196,226,125,24,5,16,6,0,0           ; vbroadcastss  0x610(%rip),%ymm0        # 493c <_sk_callback_hsw+0x4a5>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,1,6,0,0            ; vbroadcastss  0x601(%rip),%ymm10        # 4930 <_sk_callback_hsw+0x4ad>
-  DB  196,98,125,24,29,252,5,0,0          ; vbroadcastss  0x5fc(%rip),%ymm11        # 4934 <_sk_callback_hsw+0x4b1>
+  DB  196,98,125,24,21,253,5,0,0          ; vbroadcastss  0x5fd(%rip),%ymm10        # 4940 <_sk_callback_hsw+0x4a9>
+  DB  196,98,125,24,29,248,5,0,0          ; vbroadcastss  0x5f8(%rip),%ymm11        # 4944 <_sk_callback_hsw+0x4ad>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,52,89,195                    ; vmulps        %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -4091,13 +4096,13 @@ _sk_bicubic_p3x_hsw LABEL PROC
 PUBLIC _sk_bicubic_n3y_hsw
 _sk_bicubic_n3y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,223,5,0,0         ; vbroadcastss  0x5df(%rip),%ymm1        # 4938 <_sk_callback_hsw+0x4b5>
+  DB  196,226,125,24,13,219,5,0,0         ; vbroadcastss  0x5db(%rip),%ymm1        # 4948 <_sk_callback_hsw+0x4b1>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,213,5,0,0           ; vbroadcastss  0x5d5(%rip),%ymm8        # 493c <_sk_callback_hsw+0x4b9>
+  DB  196,98,125,24,5,209,5,0,0           ; vbroadcastss  0x5d1(%rip),%ymm8        # 494c <_sk_callback_hsw+0x4b5>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,198,5,0,0          ; vbroadcastss  0x5c6(%rip),%ymm10        # 4940 <_sk_callback_hsw+0x4bd>
-  DB  196,98,125,24,29,193,5,0,0          ; vbroadcastss  0x5c1(%rip),%ymm11        # 4944 <_sk_callback_hsw+0x4c1>
+  DB  196,98,125,24,21,194,5,0,0          ; vbroadcastss  0x5c2(%rip),%ymm10        # 4950 <_sk_callback_hsw+0x4b9>
+  DB  196,98,125,24,29,189,5,0,0          ; vbroadcastss  0x5bd(%rip),%ymm11        # 4954 <_sk_callback_hsw+0x4bd>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,36,89,193                    ; vmulps        %ymm9,%ymm11,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -4107,16 +4112,16 @@ _sk_bicubic_n3y_hsw LABEL PROC
 PUBLIC _sk_bicubic_n1y_hsw
 _sk_bicubic_n1y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,164,5,0,0         ; vbroadcastss  0x5a4(%rip),%ymm1        # 4948 <_sk_callback_hsw+0x4c5>
+  DB  196,226,125,24,13,160,5,0,0         ; vbroadcastss  0x5a0(%rip),%ymm1        # 4958 <_sk_callback_hsw+0x4c1>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,154,5,0,0           ; vbroadcastss  0x59a(%rip),%ymm8        # 494c <_sk_callback_hsw+0x4c9>
+  DB  196,98,125,24,5,150,5,0,0           ; vbroadcastss  0x596(%rip),%ymm8        # 495c <_sk_callback_hsw+0x4c5>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,144,5,0,0          ; vbroadcastss  0x590(%rip),%ymm9        # 4950 <_sk_callback_hsw+0x4cd>
-  DB  196,98,125,24,21,139,5,0,0          ; vbroadcastss  0x58b(%rip),%ymm10        # 4954 <_sk_callback_hsw+0x4d1>
+  DB  196,98,125,24,13,140,5,0,0          ; vbroadcastss  0x58c(%rip),%ymm9        # 4960 <_sk_callback_hsw+0x4c9>
+  DB  196,98,125,24,21,135,5,0,0          ; vbroadcastss  0x587(%rip),%ymm10        # 4964 <_sk_callback_hsw+0x4cd>
   DB  196,66,61,168,209                   ; vfmadd213ps   %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,13,129,5,0,0          ; vbroadcastss  0x581(%rip),%ymm9        # 4958 <_sk_callback_hsw+0x4d5>
+  DB  196,98,125,24,13,125,5,0,0          ; vbroadcastss  0x57d(%rip),%ymm9        # 4968 <_sk_callback_hsw+0x4d1>
   DB  196,66,61,184,202                   ; vfmadd231ps   %ymm10,%ymm8,%ymm9
-  DB  196,98,125,24,21,119,5,0,0          ; vbroadcastss  0x577(%rip),%ymm10        # 495c <_sk_callback_hsw+0x4d9>
+  DB  196,98,125,24,21,115,5,0,0          ; vbroadcastss  0x573(%rip),%ymm10        # 496c <_sk_callback_hsw+0x4d5>
   DB  196,66,61,184,209                   ; vfmadd231ps   %ymm9,%ymm8,%ymm10
   DB  197,124,17,144,160,0,0,0            ; vmovups       %ymm10,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4125,14 +4130,14 @@ _sk_bicubic_n1y_hsw LABEL PROC
 PUBLIC _sk_bicubic_p1y_hsw
 _sk_bicubic_p1y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,95,5,0,0            ; vbroadcastss  0x55f(%rip),%ymm8        # 4960 <_sk_callback_hsw+0x4dd>
+  DB  196,98,125,24,5,91,5,0,0            ; vbroadcastss  0x55b(%rip),%ymm8        # 4970 <_sk_callback_hsw+0x4d9>
   DB  197,188,88,72,32                    ; vaddps        0x20(%rax),%ymm8,%ymm1
   DB  197,124,16,72,96                    ; vmovups       0x60(%rax),%ymm9
-  DB  196,98,125,24,21,80,5,0,0           ; vbroadcastss  0x550(%rip),%ymm10        # 4964 <_sk_callback_hsw+0x4e1>
-  DB  196,98,125,24,29,75,5,0,0           ; vbroadcastss  0x54b(%rip),%ymm11        # 4968 <_sk_callback_hsw+0x4e5>
+  DB  196,98,125,24,21,76,5,0,0           ; vbroadcastss  0x54c(%rip),%ymm10        # 4974 <_sk_callback_hsw+0x4dd>
+  DB  196,98,125,24,29,71,5,0,0           ; vbroadcastss  0x547(%rip),%ymm11        # 4978 <_sk_callback_hsw+0x4e1>
   DB  196,66,53,168,218                   ; vfmadd213ps   %ymm10,%ymm9,%ymm11
   DB  196,66,53,168,216                   ; vfmadd213ps   %ymm8,%ymm9,%ymm11
-  DB  196,98,125,24,5,60,5,0,0            ; vbroadcastss  0x53c(%rip),%ymm8        # 496c <_sk_callback_hsw+0x4e9>
+  DB  196,98,125,24,5,56,5,0,0            ; vbroadcastss  0x538(%rip),%ymm8        # 497c <_sk_callback_hsw+0x4e5>
   DB  196,66,53,184,195                   ; vfmadd231ps   %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4141,12 +4146,12 @@ _sk_bicubic_p1y_hsw LABEL PROC
 PUBLIC _sk_bicubic_p3y_hsw
 _sk_bicubic_p3y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,36,5,0,0          ; vbroadcastss  0x524(%rip),%ymm1        # 4970 <_sk_callback_hsw+0x4ed>
+  DB  196,226,125,24,13,32,5,0,0          ; vbroadcastss  0x520(%rip),%ymm1        # 4980 <_sk_callback_hsw+0x4e9>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,16,5,0,0           ; vbroadcastss  0x510(%rip),%ymm10        # 4974 <_sk_callback_hsw+0x4f1>
-  DB  196,98,125,24,29,11,5,0,0           ; vbroadcastss  0x50b(%rip),%ymm11        # 4978 <_sk_callback_hsw+0x4f5>
+  DB  196,98,125,24,21,12,5,0,0           ; vbroadcastss  0x50c(%rip),%ymm10        # 4984 <_sk_callback_hsw+0x4ed>
+  DB  196,98,125,24,29,7,5,0,0            ; vbroadcastss  0x507(%rip),%ymm11        # 4988 <_sk_callback_hsw+0x4f1>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,52,89,195                    ; vmulps        %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -4261,25 +4266,25 @@ ALIGN 4
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 4651 <.literal4+0xb5>
+  DB  71,225,61                           ; rex.RXB       loope 4665 <.literal4+0xb5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 4661 <.literal4+0xc5>
+  DB  71,225,61                           ; rex.RXB       loope 4675 <.literal4+0xc5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 4671 <.literal4+0xd5>
+  DB  71,225,61                           ; rex.RXB       loope 4685 <.literal4+0xd5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 4681 <.literal4+0xe5>
+  DB  71,225,61                           ; rex.RXB       loope 4695 <.literal4+0xe5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -4328,24 +4333,26 @@ ALIGN 4
   DB  190,129,128,128,59                  ; mov           $0x3b808081,%esi
   DB  129,128,128,59,0,248,0,0,8,33       ; addl          $0x21080000,-0x7ffc480(%rax)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        46cd <.literal4+0x131>
+  DB  224,7                               ; loopne        46e1 <.literal4+0x131>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
   DB  31                                  ; (bad)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,8                                 ; add           %cl,(%rax)
-  DB  33,4,61,0,0,128,63                  ; and           %eax,0x3f800000(,%rdi,1)
-  DB  129,128,128,59,128,0,128,55,0,0     ; addl          $0x3780,0x803b80(%rax)
+  DB  33,4,61,129,128,128,59              ; and           %eax,0x3b808081(,%rdi,1)
+  DB  128,0,128                           ; addb          $0x80,(%rax)
+  DB  55                                  ; (bad)
+  DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
   DB  0,52,255                            ; add           %dh,(%rdi,%rdi,8)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            46f8 <.literal4+0x15c>
+  DB  127,0                               ; jg            4708 <.literal4+0x158>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4771 <.literal4+0x1d5>
+  DB  119,115                             ; ja            4781 <.literal4+0x1d1>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4359,10 +4366,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            472c <.literal4+0x190>
+  DB  127,0                               ; jg            473c <.literal4+0x18c>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            47a5 <.literal4+0x209>
+  DB  119,115                             ; ja            47b5 <.literal4+0x205>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4376,10 +4383,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4760 <.literal4+0x1c4>
+  DB  127,0                               ; jg            4770 <.literal4+0x1c0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            47d9 <.literal4+0x23d>
+  DB  119,115                             ; ja            47e9 <.literal4+0x239>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4393,10 +4400,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4794 <.literal4+0x1f8>
+  DB  127,0                               ; jg            47a4 <.literal4+0x1f4>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            480d <.literal4+0x271>
+  DB  119,115                             ; ja            481d <.literal4+0x26d>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4409,7 +4416,7 @@ ALIGN 4
   DB  0,75,0                              ; add           %cl,0x0(%rbx)
   DB  0,128,63,0,0,200                    ; add           %al,-0x37ffffc1(%rax)
   DB  66,0,0                              ; rex.X         add %al,(%rax)
-  DB  127,67                              ; jg            480b <.literal4+0x26f>
+  DB  127,67                              ; jg            481b <.literal4+0x26b>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -4421,10 +4428,10 @@ ALIGN 4
   DB  190,80,128,3,62                     ; mov           $0x3e038050,%esi
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           482b <.literal4+0x28f>
+  DB  118,63                              ; jbe           483b <.literal4+0x28b>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            483f <.literal4+0x2a3>
+  DB  127,67                              ; jg            484f <.literal4+0x29f>
   DB  129,128,128,59,0,0,128,63,129,128   ; addl          $0x80813f80,0x3b80(%rax)
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,128,63,129,128,128                ; add           %al,-0x7f7f7ec1(%rax)
@@ -4433,7 +4440,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4821 <.literal4+0x285>
+  DB  224,7                               ; loopne        4831 <.literal4+0x281>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -4445,7 +4452,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        483d <.literal4+0x2a1>
+  DB  224,7                               ; loopne        484d <.literal4+0x29d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -4456,7 +4463,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            4892 <.literal4+0x2f6>
+  DB  124,66                              ; jl            48a2 <.literal4+0x2f2>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,55,0,15                 ; mov           %ecx,0xf003788(%rax)
@@ -4474,9 +4481,9 @@ ALIGN 4
   DB  137,136,136,59,15,0                 ; mov           %ecx,0xf3b88(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,61,0,0                  ; mov           %ecx,0x3d88(%rax)
-  DB  112,65                              ; jo            48d5 <.literal4+0x339>
+  DB  112,65                              ; jo            48e5 <.literal4+0x335>
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            48e3 <.literal4+0x347>
+  DB  127,67                              ; jg            48f3 <.literal4+0x343>
   DB  128,0,128                           ; addb          $0x80,(%rax)
   DB  55                                  ; (bad)
   DB  128,0,128                           ; addb          $0x80,(%rax)
@@ -4484,7 +4491,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  255                                 ; (bad)
-  DB  127,71                              ; jg            48f7 <.literal4+0x35b>
+  DB  127,71                              ; jg            4907 <.literal4+0x357>
   DB  208                                 ; (bad)
   DB  179,89                              ; mov           $0x59,%bl
   DB  62,89                               ; ds            pop %rcx
@@ -4581,16 +4588,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a0049a8 <_sk_callback_hsw+0xa000525>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a0049c8 <_sk_callback_hsw+0xa000531>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 120049b0 <_sk_callback_hsw+0x1200052d>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 120049d0 <_sk_callback_hsw+0x12000539>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a0049b8 <_sk_callback_hsw+0x1a000535>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a0049d8 <_sk_callback_hsw+0x1a000541>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 30049c0 <_sk_callback_hsw+0x300053d>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 30049e0 <_sk_callback_hsw+0x3000549>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4633,16 +4640,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004a08 <_sk_callback_hsw+0xa000585>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004a28 <_sk_callback_hsw+0xa000591>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004a10 <_sk_callback_hsw+0x1200058d>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004a30 <_sk_callback_hsw+0x12000599>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004a18 <_sk_callback_hsw+0x1a000595>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004a38 <_sk_callback_hsw+0x1a0005a1>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004a20 <_sk_callback_hsw+0x300059d>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004a40 <_sk_callback_hsw+0x30005a9>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4685,16 +4692,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004a68 <_sk_callback_hsw+0xa0005e5>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004a88 <_sk_callback_hsw+0xa0005f1>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004a70 <_sk_callback_hsw+0x120005ed>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004a90 <_sk_callback_hsw+0x120005f9>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004a78 <_sk_callback_hsw+0x1a0005f5>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004a98 <_sk_callback_hsw+0x1a000601>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004a80 <_sk_callback_hsw+0x30005fd>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004aa0 <_sk_callback_hsw+0x3000609>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4737,16 +4744,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004ac8 <_sk_callback_hsw+0xa000645>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004ae8 <_sk_callback_hsw+0xa000651>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004ad0 <_sk_callback_hsw+0x1200064d>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004af0 <_sk_callback_hsw+0x12000659>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004ad8 <_sk_callback_hsw+0x1a000655>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004af8 <_sk_callback_hsw+0x1a000661>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004ae0 <_sk_callback_hsw+0x300065d>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004b00 <_sk_callback_hsw+0x3000669>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4888,14 +4895,14 @@ _sk_seed_shader_avx LABEL PROC
   DB  197,249,112,192,0                   ; vpshufd       $0x0,%xmm0,%xmm0
   DB  196,227,125,24,192,1                ; vinsertf128   $0x1,%xmm0,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,252,91,0,0        ; vbroadcastss  0x5bfc(%rip),%ymm1        # 5d5c <_sk_callback_avx+0x11c>
+  DB  196,226,125,24,13,32,92,0,0         ; vbroadcastss  0x5c20(%rip),%ymm1        # 5d80 <_sk_callback_avx+0x11c>
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
   DB  197,252,88,2                        ; vaddps        (%rdx),%ymm0,%ymm0
   DB  196,226,125,24,16                   ; vbroadcastss  (%rax),%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  197,236,88,201                      ; vaddps        %ymm1,%ymm2,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,21,224,91,0,0        ; vbroadcastss  0x5be0(%rip),%ymm2        # 5d60 <_sk_callback_avx+0x120>
+  DB  196,226,125,24,21,4,92,0,0          ; vbroadcastss  0x5c04(%rip),%ymm2        # 5d84 <_sk_callback_avx+0x120>
   DB  197,228,87,219                      ; vxorps        %ymm3,%ymm3,%ymm3
   DB  197,220,87,228                      ; vxorps        %ymm4,%ymm4,%ymm4
   DB  197,212,87,237                      ; vxorps        %ymm5,%ymm5,%ymm5
@@ -4915,7 +4922,7 @@ _sk_dither_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  196,66,125,24,8                     ; vbroadcastss  (%r8),%ymm9
   DB  196,65,60,87,209                    ; vxorps        %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,29,152,91,0,0         ; vbroadcastss  0x5b98(%rip),%ymm11        # 5d64 <_sk_callback_avx+0x124>
+  DB  196,98,125,24,29,188,91,0,0         ; vbroadcastss  0x5bbc(%rip),%ymm11        # 5d88 <_sk_callback_avx+0x124>
   DB  196,65,44,84,203                    ; vandps        %ymm11,%ymm10,%ymm9
   DB  196,193,25,114,241,5                ; vpslld        $0x5,%xmm9,%xmm12
   DB  196,67,125,25,201,1                 ; vextractf128  $0x1,%ymm9,%xmm9
@@ -4926,8 +4933,8 @@ _sk_dither_avx LABEL PROC
   DB  196,67,125,25,219,1                 ; vextractf128  $0x1,%ymm11,%xmm11
   DB  196,193,33,114,243,4                ; vpslld        $0x4,%xmm11,%xmm11
   DB  196,67,29,24,219,1                  ; vinsertf128   $0x1,%xmm11,%ymm12,%ymm11
-  DB  196,98,125,24,37,89,91,0,0          ; vbroadcastss  0x5b59(%rip),%ymm12        # 5d68 <_sk_callback_avx+0x128>
-  DB  196,98,125,24,45,84,91,0,0          ; vbroadcastss  0x5b54(%rip),%ymm13        # 5d6c <_sk_callback_avx+0x12c>
+  DB  196,98,125,24,37,125,91,0,0         ; vbroadcastss  0x5b7d(%rip),%ymm12        # 5d8c <_sk_callback_avx+0x128>
+  DB  196,98,125,24,45,120,91,0,0         ; vbroadcastss  0x5b78(%rip),%ymm13        # 5d90 <_sk_callback_avx+0x12c>
   DB  196,65,44,84,245                    ; vandps        %ymm13,%ymm10,%ymm14
   DB  196,193,1,114,246,2                 ; vpslld        $0x2,%xmm14,%xmm15
   DB  196,67,125,25,246,1                 ; vextractf128  $0x1,%ymm14,%xmm14
@@ -4954,9 +4961,9 @@ _sk_dither_avx LABEL PROC
   DB  196,65,60,86,193                    ; vorps         %ymm9,%ymm8,%ymm8
   DB  196,65,60,86,194                    ; vorps         %ymm10,%ymm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,191,90,0,0         ; vbroadcastss  0x5abf(%rip),%ymm9        # 5d70 <_sk_callback_avx+0x130>
+  DB  196,98,125,24,13,227,90,0,0         ; vbroadcastss  0x5ae3(%rip),%ymm9        # 5d94 <_sk_callback_avx+0x130>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,181,90,0,0         ; vbroadcastss  0x5ab5(%rip),%ymm9        # 5d74 <_sk_callback_avx+0x134>
+  DB  196,98,125,24,13,217,90,0,0         ; vbroadcastss  0x5ad9(%rip),%ymm9        # 5d98 <_sk_callback_avx+0x134>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  196,98,125,24,72,8                  ; vbroadcastss  0x8(%rax),%ymm9
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
@@ -4988,7 +4995,7 @@ _sk_clear_avx LABEL PROC
 PUBLIC _sk_srcatop_avx
 _sk_srcatop_avx LABEL PROC
   DB  197,252,89,199                      ; vmulps        %ymm7,%ymm0,%ymm0
-  DB  196,98,125,24,5,91,90,0,0           ; vbroadcastss  0x5a5b(%rip),%ymm8        # 5d78 <_sk_callback_avx+0x138>
+  DB  196,98,125,24,5,127,90,0,0          ; vbroadcastss  0x5a7f(%rip),%ymm8        # 5d9c <_sk_callback_avx+0x138>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,204                       ; vmulps        %ymm4,%ymm8,%ymm9
   DB  197,180,88,192                      ; vaddps        %ymm0,%ymm9,%ymm0
@@ -5007,7 +5014,7 @@ _sk_srcatop_avx LABEL PROC
 PUBLIC _sk_dstatop_avx
 _sk_dstatop_avx LABEL PROC
   DB  197,100,89,196                      ; vmulps        %ymm4,%ymm3,%ymm8
-  DB  196,98,125,24,13,29,90,0,0          ; vbroadcastss  0x5a1d(%rip),%ymm9        # 5d7c <_sk_callback_avx+0x13c>
+  DB  196,98,125,24,13,65,90,0,0          ; vbroadcastss  0x5a41(%rip),%ymm9        # 5da0 <_sk_callback_avx+0x13c>
   DB  197,52,92,207                       ; vsubps        %ymm7,%ymm9,%ymm9
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
   DB  197,188,88,192                      ; vaddps        %ymm0,%ymm8,%ymm0
@@ -5043,7 +5050,7 @@ _sk_dstin_avx LABEL PROC
 
 PUBLIC _sk_srcout_avx
 _sk_srcout_avx LABEL PROC
-  DB  196,98,125,24,5,188,89,0,0          ; vbroadcastss  0x59bc(%rip),%ymm8        # 5d80 <_sk_callback_avx+0x140>
+  DB  196,98,125,24,5,224,89,0,0          ; vbroadcastss  0x59e0(%rip),%ymm8        # 5da4 <_sk_callback_avx+0x140>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -5054,7 +5061,7 @@ _sk_srcout_avx LABEL PROC
 
 PUBLIC _sk_dstout_avx
 _sk_dstout_avx LABEL PROC
-  DB  196,226,125,24,5,159,89,0,0         ; vbroadcastss  0x599f(%rip),%ymm0        # 5d84 <_sk_callback_avx+0x144>
+  DB  196,226,125,24,5,195,89,0,0         ; vbroadcastss  0x59c3(%rip),%ymm0        # 5da8 <_sk_callback_avx+0x144>
   DB  197,252,92,219                      ; vsubps        %ymm3,%ymm0,%ymm3
   DB  197,228,89,196                      ; vmulps        %ymm4,%ymm3,%ymm0
   DB  197,228,89,205                      ; vmulps        %ymm5,%ymm3,%ymm1
@@ -5065,7 +5072,7 @@ _sk_dstout_avx LABEL PROC
 
 PUBLIC _sk_srcover_avx
 _sk_srcover_avx LABEL PROC
-  DB  196,98,125,24,5,130,89,0,0          ; vbroadcastss  0x5982(%rip),%ymm8        # 5d88 <_sk_callback_avx+0x148>
+  DB  196,98,125,24,5,166,89,0,0          ; vbroadcastss  0x59a6(%rip),%ymm8        # 5dac <_sk_callback_avx+0x148>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,204                       ; vmulps        %ymm4,%ymm8,%ymm9
   DB  197,180,88,192                      ; vaddps        %ymm0,%ymm9,%ymm0
@@ -5080,7 +5087,7 @@ _sk_srcover_avx LABEL PROC
 
 PUBLIC _sk_dstover_avx
 _sk_dstover_avx LABEL PROC
-  DB  196,98,125,24,5,85,89,0,0           ; vbroadcastss  0x5955(%rip),%ymm8        # 5d8c <_sk_callback_avx+0x14c>
+  DB  196,98,125,24,5,121,89,0,0          ; vbroadcastss  0x5979(%rip),%ymm8        # 5db0 <_sk_callback_avx+0x14c>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,252,88,196                      ; vaddps        %ymm4,%ymm0,%ymm0
@@ -5104,7 +5111,7 @@ _sk_modulate_avx LABEL PROC
 
 PUBLIC _sk_multiply_avx
 _sk_multiply_avx LABEL PROC
-  DB  196,98,125,24,5,20,89,0,0           ; vbroadcastss  0x5914(%rip),%ymm8        # 5d90 <_sk_callback_avx+0x150>
+  DB  196,98,125,24,5,56,89,0,0           ; vbroadcastss  0x5938(%rip),%ymm8        # 5db4 <_sk_callback_avx+0x150>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,208                       ; vmulps        %ymm0,%ymm9,%ymm10
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5158,7 +5165,7 @@ _sk_screen_avx LABEL PROC
 
 PUBLIC _sk_xor__avx
 _sk_xor__avx LABEL PROC
-  DB  196,98,125,24,5,99,88,0,0           ; vbroadcastss  0x5863(%rip),%ymm8        # 5d94 <_sk_callback_avx+0x154>
+  DB  196,98,125,24,5,135,88,0,0          ; vbroadcastss  0x5887(%rip),%ymm8        # 5db8 <_sk_callback_avx+0x154>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5193,7 +5200,7 @@ _sk_darken_avx LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,95,209                  ; vmaxps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,227,87,0,0          ; vbroadcastss  0x57e3(%rip),%ymm8        # 5d98 <_sk_callback_avx+0x158>
+  DB  196,98,125,24,5,7,88,0,0            ; vbroadcastss  0x5807(%rip),%ymm8        # 5dbc <_sk_callback_avx+0x158>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5217,7 +5224,7 @@ _sk_lighten_avx LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,143,87,0,0          ; vbroadcastss  0x578f(%rip),%ymm8        # 5d9c <_sk_callback_avx+0x15c>
+  DB  196,98,125,24,5,179,87,0,0          ; vbroadcastss  0x57b3(%rip),%ymm8        # 5dc0 <_sk_callback_avx+0x15c>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5244,7 +5251,7 @@ _sk_difference_avx LABEL PROC
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,47,87,0,0           ; vbroadcastss  0x572f(%rip),%ymm8        # 5da0 <_sk_callback_avx+0x160>
+  DB  196,98,125,24,5,83,87,0,0           ; vbroadcastss  0x5753(%rip),%ymm8        # 5dc4 <_sk_callback_avx+0x160>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5265,7 +5272,7 @@ _sk_exclusion_avx LABEL PROC
   DB  197,236,89,214                      ; vmulps        %ymm6,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,234,86,0,0          ; vbroadcastss  0x56ea(%rip),%ymm8        # 5da4 <_sk_callback_avx+0x164>
+  DB  196,98,125,24,5,14,87,0,0           ; vbroadcastss  0x570e(%rip),%ymm8        # 5dc8 <_sk_callback_avx+0x164>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5274,7 +5281,7 @@ _sk_exclusion_avx LABEL PROC
 
 PUBLIC _sk_colorburn_avx
 _sk_colorburn_avx LABEL PROC
-  DB  196,98,125,24,5,213,86,0,0          ; vbroadcastss  0x56d5(%rip),%ymm8        # 5da8 <_sk_callback_avx+0x168>
+  DB  196,98,125,24,5,249,86,0,0          ; vbroadcastss  0x56f9(%rip),%ymm8        # 5dcc <_sk_callback_avx+0x168>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,216                       ; vmulps        %ymm0,%ymm9,%ymm11
   DB  196,65,44,87,210                    ; vxorps        %ymm10,%ymm10,%ymm10
@@ -5334,7 +5341,7 @@ _sk_colorburn_avx LABEL PROC
 PUBLIC _sk_colordodge_avx
 _sk_colordodge_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
-  DB  196,98,125,24,13,209,85,0,0         ; vbroadcastss  0x55d1(%rip),%ymm9        # 5dac <_sk_callback_avx+0x16c>
+  DB  196,98,125,24,13,245,85,0,0         ; vbroadcastss  0x55f5(%rip),%ymm9        # 5dd0 <_sk_callback_avx+0x16c>
   DB  197,52,92,215                       ; vsubps        %ymm7,%ymm9,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,52,92,203                       ; vsubps        %ymm3,%ymm9,%ymm9
@@ -5389,7 +5396,7 @@ _sk_colordodge_avx LABEL PROC
 
 PUBLIC _sk_hardlight_avx
 _sk_hardlight_avx LABEL PROC
-  DB  196,98,125,24,5,227,84,0,0          ; vbroadcastss  0x54e3(%rip),%ymm8        # 5db0 <_sk_callback_avx+0x170>
+  DB  196,98,125,24,5,7,85,0,0            ; vbroadcastss  0x5507(%rip),%ymm8        # 5dd4 <_sk_callback_avx+0x170>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,200                       ; vmulps        %ymm0,%ymm10,%ymm9
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5442,7 +5449,7 @@ _sk_hardlight_avx LABEL PROC
 
 PUBLIC _sk_overlay_avx
 _sk_overlay_avx LABEL PROC
-  DB  196,98,125,24,5,12,84,0,0           ; vbroadcastss  0x540c(%rip),%ymm8        # 5db4 <_sk_callback_avx+0x174>
+  DB  196,98,125,24,5,48,84,0,0           ; vbroadcastss  0x5430(%rip),%ymm8        # 5dd8 <_sk_callback_avx+0x174>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,200                       ; vmulps        %ymm0,%ymm10,%ymm9
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5507,10 +5514,10 @@ _sk_softlight_avx LABEL PROC
   DB  196,65,60,88,192                    ; vaddps        %ymm8,%ymm8,%ymm8
   DB  196,65,60,89,216                    ; vmulps        %ymm8,%ymm8,%ymm11
   DB  196,65,60,88,195                    ; vaddps        %ymm11,%ymm8,%ymm8
-  DB  196,98,125,24,29,255,82,0,0         ; vbroadcastss  0x52ff(%rip),%ymm11        # 5dbc <_sk_callback_avx+0x17c>
+  DB  196,98,125,24,29,35,83,0,0          ; vbroadcastss  0x5323(%rip),%ymm11        # 5de0 <_sk_callback_avx+0x17c>
   DB  196,65,28,88,235                    ; vaddps        %ymm11,%ymm12,%ymm13
   DB  196,65,20,89,192                    ; vmulps        %ymm8,%ymm13,%ymm8
-  DB  196,98,125,24,45,240,82,0,0         ; vbroadcastss  0x52f0(%rip),%ymm13        # 5dc0 <_sk_callback_avx+0x180>
+  DB  196,98,125,24,45,20,83,0,0          ; vbroadcastss  0x5314(%rip),%ymm13        # 5de4 <_sk_callback_avx+0x180>
   DB  196,65,28,89,245                    ; vmulps        %ymm13,%ymm12,%ymm14
   DB  196,65,12,88,192                    ; vaddps        %ymm8,%ymm14,%ymm8
   DB  196,65,124,82,244                   ; vrsqrtps      %ymm12,%ymm14
@@ -5521,7 +5528,7 @@ _sk_softlight_avx LABEL PROC
   DB  197,4,194,255,2                     ; vcmpleps      %ymm7,%ymm15,%ymm15
   DB  196,67,13,74,240,240                ; vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   DB  197,116,88,249                      ; vaddps        %ymm1,%ymm1,%ymm15
-  DB  196,98,125,24,5,174,82,0,0          ; vbroadcastss  0x52ae(%rip),%ymm8        # 5db8 <_sk_callback_avx+0x178>
+  DB  196,98,125,24,5,210,82,0,0          ; vbroadcastss  0x52d2(%rip),%ymm8        # 5ddc <_sk_callback_avx+0x178>
   DB  196,65,60,92,228                    ; vsubps        %ymm12,%ymm8,%ymm12
   DB  197,132,92,195                      ; vsubps        %ymm3,%ymm15,%ymm0
   DB  196,65,124,89,228                   ; vmulps        %ymm12,%ymm0,%ymm12
@@ -5617,7 +5624,7 @@ PUBLIC _sk_hue_avx
 _sk_hue_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,208,0                ; vcmpeqps      %ymm8,%ymm3,%ymm10
-  DB  196,98,125,24,13,14,81,0,0          ; vbroadcastss  0x510e(%rip),%ymm9        # 5dc4 <_sk_callback_avx+0x184>
+  DB  196,98,125,24,13,50,81,0,0          ; vbroadcastss  0x5132(%rip),%ymm9        # 5de8 <_sk_callback_avx+0x184>
   DB  197,52,94,219                       ; vdivps        %ymm3,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,172,89,192                      ; vmulps        %ymm0,%ymm10,%ymm0
@@ -5646,12 +5653,12 @@ _sk_hue_avx LABEL PROC
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  196,193,108,94,212                  ; vdivps        %ymm12,%ymm2,%ymm2
   DB  196,195,109,74,208,208              ; vblendvps     %ymm13,%ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,21,131,80,0,0         ; vbroadcastss  0x5083(%rip),%ymm10        # 5dc8 <_sk_callback_avx+0x188>
+  DB  196,98,125,24,21,167,80,0,0         ; vbroadcastss  0x50a7(%rip),%ymm10        # 5dec <_sk_callback_avx+0x188>
   DB  196,65,92,89,218                    ; vmulps        %ymm10,%ymm4,%ymm11
-  DB  196,98,125,24,37,121,80,0,0         ; vbroadcastss  0x5079(%rip),%ymm12        # 5dcc <_sk_callback_avx+0x18c>
+  DB  196,98,125,24,37,157,80,0,0         ; vbroadcastss  0x509d(%rip),%ymm12        # 5df0 <_sk_callback_avx+0x18c>
   DB  196,65,84,89,236                    ; vmulps        %ymm12,%ymm5,%ymm13
   DB  196,65,36,88,221                    ; vaddps        %ymm13,%ymm11,%ymm11
-  DB  196,98,125,24,45,106,80,0,0         ; vbroadcastss  0x506a(%rip),%ymm13        # 5dd0 <_sk_callback_avx+0x190>
+  DB  196,98,125,24,45,142,80,0,0         ; vbroadcastss  0x508e(%rip),%ymm13        # 5df4 <_sk_callback_avx+0x190>
   DB  196,65,76,89,245                    ; vmulps        %ymm13,%ymm6,%ymm14
   DB  196,65,36,88,222                    ; vaddps        %ymm14,%ymm11,%ymm11
   DB  196,65,124,89,242                   ; vmulps        %ymm10,%ymm0,%ymm14
@@ -5723,7 +5730,7 @@ PUBLIC _sk_saturation_avx
 _sk_saturation_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,68,194,208,0                 ; vcmpeqps      %ymm8,%ymm7,%ymm10
-  DB  196,98,125,24,13,39,79,0,0          ; vbroadcastss  0x4f27(%rip),%ymm9        # 5dd4 <_sk_callback_avx+0x194>
+  DB  196,98,125,24,13,75,79,0,0          ; vbroadcastss  0x4f4b(%rip),%ymm9        # 5df8 <_sk_callback_avx+0x194>
   DB  197,52,94,223                       ; vdivps        %ymm7,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,44,89,220                       ; vmulps        %ymm4,%ymm10,%ymm11
@@ -5752,12 +5759,12 @@ _sk_saturation_avx LABEL PROC
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  197,252,94,194                      ; vdivps        %ymm2,%ymm0,%ymm0
   DB  196,195,125,74,192,208              ; vblendvps     %ymm13,%ymm8,%ymm0,%ymm0
-  DB  196,226,125,24,13,163,78,0,0        ; vbroadcastss  0x4ea3(%rip),%ymm1        # 5dd8 <_sk_callback_avx+0x198>
+  DB  196,226,125,24,13,199,78,0,0        ; vbroadcastss  0x4ec7(%rip),%ymm1        # 5dfc <_sk_callback_avx+0x198>
   DB  197,220,89,209                      ; vmulps        %ymm1,%ymm4,%ymm2
-  DB  196,98,125,24,21,154,78,0,0         ; vbroadcastss  0x4e9a(%rip),%ymm10        # 5ddc <_sk_callback_avx+0x19c>
+  DB  196,98,125,24,21,190,78,0,0         ; vbroadcastss  0x4ebe(%rip),%ymm10        # 5e00 <_sk_callback_avx+0x19c>
   DB  196,65,84,89,234                    ; vmulps        %ymm10,%ymm5,%ymm13
   DB  196,193,108,88,213                  ; vaddps        %ymm13,%ymm2,%ymm2
-  DB  196,98,125,24,45,139,78,0,0         ; vbroadcastss  0x4e8b(%rip),%ymm13        # 5de0 <_sk_callback_avx+0x1a0>
+  DB  196,98,125,24,45,175,78,0,0         ; vbroadcastss  0x4eaf(%rip),%ymm13        # 5e04 <_sk_callback_avx+0x1a0>
   DB  196,65,76,89,245                    ; vmulps        %ymm13,%ymm6,%ymm14
   DB  196,193,108,88,214                  ; vaddps        %ymm14,%ymm2,%ymm2
   DB  197,36,89,241                       ; vmulps        %ymm1,%ymm11,%ymm14
@@ -5829,18 +5836,18 @@ PUBLIC _sk_color_avx
 _sk_color_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,208,0                ; vcmpeqps      %ymm8,%ymm3,%ymm10
-  DB  196,98,125,24,13,76,77,0,0          ; vbroadcastss  0x4d4c(%rip),%ymm9        # 5de4 <_sk_callback_avx+0x1a4>
+  DB  196,98,125,24,13,112,77,0,0         ; vbroadcastss  0x4d70(%rip),%ymm9        # 5e08 <_sk_callback_avx+0x1a4>
   DB  197,52,94,219                       ; vdivps        %ymm3,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,172,89,192                      ; vmulps        %ymm0,%ymm10,%ymm0
   DB  197,172,89,201                      ; vmulps        %ymm1,%ymm10,%ymm1
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
-  DB  196,98,125,24,21,49,77,0,0          ; vbroadcastss  0x4d31(%rip),%ymm10        # 5de8 <_sk_callback_avx+0x1a8>
+  DB  196,98,125,24,21,85,77,0,0          ; vbroadcastss  0x4d55(%rip),%ymm10        # 5e0c <_sk_callback_avx+0x1a8>
   DB  196,65,92,89,218                    ; vmulps        %ymm10,%ymm4,%ymm11
-  DB  196,98,125,24,37,39,77,0,0          ; vbroadcastss  0x4d27(%rip),%ymm12        # 5dec <_sk_callback_avx+0x1ac>
+  DB  196,98,125,24,37,75,77,0,0          ; vbroadcastss  0x4d4b(%rip),%ymm12        # 5e10 <_sk_callback_avx+0x1ac>
   DB  196,65,84,89,236                    ; vmulps        %ymm12,%ymm5,%ymm13
   DB  196,65,36,88,221                    ; vaddps        %ymm13,%ymm11,%ymm11
-  DB  196,98,125,24,45,24,77,0,0          ; vbroadcastss  0x4d18(%rip),%ymm13        # 5df0 <_sk_callback_avx+0x1b0>
+  DB  196,98,125,24,45,60,77,0,0          ; vbroadcastss  0x4d3c(%rip),%ymm13        # 5e14 <_sk_callback_avx+0x1b0>
   DB  196,65,76,89,245                    ; vmulps        %ymm13,%ymm6,%ymm14
   DB  196,65,36,88,222                    ; vaddps        %ymm14,%ymm11,%ymm11
   DB  196,65,124,89,242                   ; vmulps        %ymm10,%ymm0,%ymm14
@@ -5912,18 +5919,18 @@ PUBLIC _sk_luminosity_avx
 _sk_luminosity_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,68,194,208,0                 ; vcmpeqps      %ymm8,%ymm7,%ymm10
-  DB  196,98,125,24,13,213,75,0,0         ; vbroadcastss  0x4bd5(%rip),%ymm9        # 5df4 <_sk_callback_avx+0x1b4>
+  DB  196,98,125,24,13,249,75,0,0         ; vbroadcastss  0x4bf9(%rip),%ymm9        # 5e18 <_sk_callback_avx+0x1b4>
   DB  197,52,94,223                       ; vdivps        %ymm7,%ymm9,%ymm11
   DB  196,67,37,74,208,160                ; vblendvps     %ymm10,%ymm8,%ymm11,%ymm10
   DB  197,44,89,220                       ; vmulps        %ymm4,%ymm10,%ymm11
   DB  197,44,89,229                       ; vmulps        %ymm5,%ymm10,%ymm12
   DB  197,44,89,214                       ; vmulps        %ymm6,%ymm10,%ymm10
-  DB  196,98,125,24,45,186,75,0,0         ; vbroadcastss  0x4bba(%rip),%ymm13        # 5df8 <_sk_callback_avx+0x1b8>
+  DB  196,98,125,24,45,222,75,0,0         ; vbroadcastss  0x4bde(%rip),%ymm13        # 5e1c <_sk_callback_avx+0x1b8>
   DB  196,193,124,89,197                  ; vmulps        %ymm13,%ymm0,%ymm0
-  DB  196,98,125,24,53,176,75,0,0         ; vbroadcastss  0x4bb0(%rip),%ymm14        # 5dfc <_sk_callback_avx+0x1bc>
+  DB  196,98,125,24,53,212,75,0,0         ; vbroadcastss  0x4bd4(%rip),%ymm14        # 5e20 <_sk_callback_avx+0x1bc>
   DB  196,193,116,89,206                  ; vmulps        %ymm14,%ymm1,%ymm1
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,162,75,0,0        ; vbroadcastss  0x4ba2(%rip),%ymm1        # 5e00 <_sk_callback_avx+0x1c0>
+  DB  196,226,125,24,13,198,75,0,0        ; vbroadcastss  0x4bc6(%rip),%ymm1        # 5e24 <_sk_callback_avx+0x1c0>
   DB  197,236,89,209                      ; vmulps        %ymm1,%ymm2,%ymm2
   DB  197,252,88,194                      ; vaddps        %ymm2,%ymm0,%ymm0
   DB  196,193,36,89,213                   ; vmulps        %ymm13,%ymm11,%ymm2
@@ -6003,7 +6010,7 @@ _sk_clamp_0_avx LABEL PROC
 
 PUBLIC _sk_clamp_1_avx
 _sk_clamp_1_avx LABEL PROC
-  DB  196,98,125,24,5,75,74,0,0           ; vbroadcastss  0x4a4b(%rip),%ymm8        # 5e04 <_sk_callback_avx+0x1c4>
+  DB  196,98,125,24,5,111,74,0,0          ; vbroadcastss  0x4a6f(%rip),%ymm8        # 5e28 <_sk_callback_avx+0x1c4>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
@@ -6013,7 +6020,7 @@ _sk_clamp_1_avx LABEL PROC
 
 PUBLIC _sk_clamp_a_avx
 _sk_clamp_a_avx LABEL PROC
-  DB  196,98,125,24,5,46,74,0,0           ; vbroadcastss  0x4a2e(%rip),%ymm8        # 5e08 <_sk_callback_avx+0x1c8>
+  DB  196,98,125,24,5,82,74,0,0           ; vbroadcastss  0x4a52(%rip),%ymm8        # 5e2c <_sk_callback_avx+0x1c8>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  197,252,93,195                      ; vminps        %ymm3,%ymm0,%ymm0
   DB  197,244,93,203                      ; vminps        %ymm3,%ymm1,%ymm1
@@ -6085,7 +6092,7 @@ PUBLIC _sk_unpremul_avx
 _sk_unpremul_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,200,0                ; vcmpeqps      %ymm8,%ymm3,%ymm9
-  DB  196,98,125,24,21,118,73,0,0         ; vbroadcastss  0x4976(%rip),%ymm10        # 5e0c <_sk_callback_avx+0x1cc>
+  DB  196,98,125,24,21,154,73,0,0         ; vbroadcastss  0x499a(%rip),%ymm10        # 5e30 <_sk_callback_avx+0x1cc>
   DB  197,44,94,211                       ; vdivps        %ymm3,%ymm10,%ymm10
   DB  196,67,45,74,192,144                ; vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
@@ -6096,17 +6103,17 @@ _sk_unpremul_avx LABEL PROC
 
 PUBLIC _sk_from_srgb_avx
 _sk_from_srgb_avx LABEL PROC
-  DB  196,98,125,24,5,87,73,0,0           ; vbroadcastss  0x4957(%rip),%ymm8        # 5e10 <_sk_callback_avx+0x1d0>
+  DB  196,98,125,24,5,123,73,0,0          ; vbroadcastss  0x497b(%rip),%ymm8        # 5e34 <_sk_callback_avx+0x1d0>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  197,124,89,208                      ; vmulps        %ymm0,%ymm0,%ymm10
-  DB  196,98,125,24,29,73,73,0,0          ; vbroadcastss  0x4949(%rip),%ymm11        # 5e14 <_sk_callback_avx+0x1d4>
+  DB  196,98,125,24,29,109,73,0,0         ; vbroadcastss  0x496d(%rip),%ymm11        # 5e38 <_sk_callback_avx+0x1d4>
   DB  196,65,124,89,227                   ; vmulps        %ymm11,%ymm0,%ymm12
-  DB  196,98,125,24,45,63,73,0,0          ; vbroadcastss  0x493f(%rip),%ymm13        # 5e18 <_sk_callback_avx+0x1d8>
+  DB  196,98,125,24,45,99,73,0,0          ; vbroadcastss  0x4963(%rip),%ymm13        # 5e3c <_sk_callback_avx+0x1d8>
   DB  196,65,28,88,229                    ; vaddps        %ymm13,%ymm12,%ymm12
   DB  196,65,44,89,212                    ; vmulps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,48,73,0,0          ; vbroadcastss  0x4930(%rip),%ymm12        # 5e1c <_sk_callback_avx+0x1dc>
+  DB  196,98,125,24,37,84,73,0,0          ; vbroadcastss  0x4954(%rip),%ymm12        # 5e40 <_sk_callback_avx+0x1dc>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,53,38,73,0,0          ; vbroadcastss  0x4926(%rip),%ymm14        # 5e20 <_sk_callback_avx+0x1e0>
+  DB  196,98,125,24,53,74,73,0,0          ; vbroadcastss  0x494a(%rip),%ymm14        # 5e44 <_sk_callback_avx+0x1e0>
   DB  196,193,124,194,198,1               ; vcmpltps      %ymm14,%ymm0,%ymm0
   DB  196,195,45,74,193,0                 ; vblendvps     %ymm0,%ymm9,%ymm10,%ymm0
   DB  196,65,116,89,200                   ; vmulps        %ymm8,%ymm1,%ymm9
@@ -6133,18 +6140,18 @@ _sk_to_srgb_avx LABEL PROC
   DB  197,124,82,192                      ; vrsqrtps      %ymm0,%ymm8
   DB  196,65,124,83,200                   ; vrcpps        %ymm8,%ymm9
   DB  196,65,124,82,208                   ; vrsqrtps      %ymm8,%ymm10
-  DB  196,98,125,24,5,177,72,0,0          ; vbroadcastss  0x48b1(%rip),%ymm8        # 5e24 <_sk_callback_avx+0x1e4>
+  DB  196,98,125,24,5,213,72,0,0          ; vbroadcastss  0x48d5(%rip),%ymm8        # 5e48 <_sk_callback_avx+0x1e4>
   DB  196,65,124,89,216                   ; vmulps        %ymm8,%ymm0,%ymm11
-  DB  196,98,125,24,37,167,72,0,0         ; vbroadcastss  0x48a7(%rip),%ymm12        # 5e28 <_sk_callback_avx+0x1e8>
+  DB  196,98,125,24,37,203,72,0,0         ; vbroadcastss  0x48cb(%rip),%ymm12        # 5e4c <_sk_callback_avx+0x1e8>
   DB  196,65,52,89,204                    ; vmulps        %ymm12,%ymm9,%ymm9
-  DB  196,98,125,24,45,157,72,0,0         ; vbroadcastss  0x489d(%rip),%ymm13        # 5e2c <_sk_callback_avx+0x1ec>
+  DB  196,98,125,24,45,193,72,0,0         ; vbroadcastss  0x48c1(%rip),%ymm13        # 5e50 <_sk_callback_avx+0x1ec>
   DB  196,65,52,88,205                    ; vaddps        %ymm13,%ymm9,%ymm9
-  DB  196,98,125,24,53,147,72,0,0         ; vbroadcastss  0x4893(%rip),%ymm14        # 5e30 <_sk_callback_avx+0x1f0>
+  DB  196,98,125,24,53,183,72,0,0         ; vbroadcastss  0x48b7(%rip),%ymm14        # 5e54 <_sk_callback_avx+0x1f0>
   DB  196,65,44,89,214                    ; vmulps        %ymm14,%ymm10,%ymm10
   DB  196,65,44,88,201                    ; vaddps        %ymm9,%ymm10,%ymm9
-  DB  196,98,125,24,21,132,72,0,0         ; vbroadcastss  0x4884(%rip),%ymm10        # 5e34 <_sk_callback_avx+0x1f4>
+  DB  196,98,125,24,21,168,72,0,0         ; vbroadcastss  0x48a8(%rip),%ymm10        # 5e58 <_sk_callback_avx+0x1f4>
   DB  196,65,44,93,201                    ; vminps        %ymm9,%ymm10,%ymm9
-  DB  196,98,125,24,61,122,72,0,0         ; vbroadcastss  0x487a(%rip),%ymm15        # 5e38 <_sk_callback_avx+0x1f8>
+  DB  196,98,125,24,61,158,72,0,0         ; vbroadcastss  0x489e(%rip),%ymm15        # 5e5c <_sk_callback_avx+0x1f8>
   DB  196,193,124,194,199,1               ; vcmpltps      %ymm15,%ymm0,%ymm0
   DB  196,195,53,74,195,0                 ; vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   DB  197,124,82,201                      ; vrsqrtps      %ymm1,%ymm9
@@ -6179,7 +6186,7 @@ _sk_rgb_to_hsl_avx LABEL PROC
   DB  197,124,93,201                      ; vminps        %ymm1,%ymm0,%ymm9
   DB  197,52,93,202                       ; vminps        %ymm2,%ymm9,%ymm9
   DB  196,65,60,92,209                    ; vsubps        %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,29,224,71,0,0         ; vbroadcastss  0x47e0(%rip),%ymm11        # 5e3c <_sk_callback_avx+0x1fc>
+  DB  196,98,125,24,29,4,72,0,0           ; vbroadcastss  0x4804(%rip),%ymm11        # 5e60 <_sk_callback_avx+0x1fc>
   DB  196,65,36,94,218                    ; vdivps        %ymm10,%ymm11,%ymm11
   DB  197,116,92,226                      ; vsubps        %ymm2,%ymm1,%ymm12
   DB  196,65,28,89,227                    ; vmulps        %ymm11,%ymm12,%ymm12
@@ -6189,19 +6196,19 @@ _sk_rgb_to_hsl_avx LABEL PROC
   DB  196,193,108,89,211                  ; vmulps        %ymm11,%ymm2,%ymm2
   DB  197,252,92,201                      ; vsubps        %ymm1,%ymm0,%ymm1
   DB  196,193,116,89,203                  ; vmulps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,185,71,0,0         ; vbroadcastss  0x47b9(%rip),%ymm11        # 5e48 <_sk_callback_avx+0x208>
+  DB  196,98,125,24,29,221,71,0,0         ; vbroadcastss  0x47dd(%rip),%ymm11        # 5e6c <_sk_callback_avx+0x208>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,167,71,0,0         ; vbroadcastss  0x47a7(%rip),%ymm11        # 5e44 <_sk_callback_avx+0x204>
+  DB  196,98,125,24,29,203,71,0,0         ; vbroadcastss  0x47cb(%rip),%ymm11        # 5e68 <_sk_callback_avx+0x204>
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
   DB  196,227,117,74,202,224              ; vblendvps     %ymm14,%ymm2,%ymm1,%ymm1
-  DB  196,226,125,24,21,143,71,0,0        ; vbroadcastss  0x478f(%rip),%ymm2        # 5e40 <_sk_callback_avx+0x200>
+  DB  196,226,125,24,21,179,71,0,0        ; vbroadcastss  0x47b3(%rip),%ymm2        # 5e64 <_sk_callback_avx+0x200>
   DB  196,65,12,87,246                    ; vxorps        %ymm14,%ymm14,%ymm14
   DB  196,227,13,74,210,208               ; vblendvps     %ymm13,%ymm2,%ymm14,%ymm2
   DB  197,188,194,192,0                   ; vcmpeqps      %ymm0,%ymm8,%ymm0
   DB  196,193,108,88,212                  ; vaddps        %ymm12,%ymm2,%ymm2
   DB  196,227,117,74,194,0                ; vblendvps     %ymm0,%ymm2,%ymm1,%ymm0
   DB  196,193,60,88,201                   ; vaddps        %ymm9,%ymm8,%ymm1
-  DB  196,98,125,24,37,118,71,0,0         ; vbroadcastss  0x4776(%rip),%ymm12        # 5e50 <_sk_callback_avx+0x210>
+  DB  196,98,125,24,37,154,71,0,0         ; vbroadcastss  0x479a(%rip),%ymm12        # 5e74 <_sk_callback_avx+0x210>
   DB  196,193,116,89,212                  ; vmulps        %ymm12,%ymm1,%ymm2
   DB  197,28,194,226,1                    ; vcmpltps      %ymm2,%ymm12,%ymm12
   DB  196,65,36,92,216                    ; vsubps        %ymm8,%ymm11,%ymm11
@@ -6211,7 +6218,7 @@ _sk_rgb_to_hsl_avx LABEL PROC
   DB  197,172,94,201                      ; vdivps        %ymm1,%ymm10,%ymm1
   DB  196,195,125,74,198,128              ; vblendvps     %ymm8,%ymm14,%ymm0,%ymm0
   DB  196,195,117,74,206,128              ; vblendvps     %ymm8,%ymm14,%ymm1,%ymm1
-  DB  196,98,125,24,5,57,71,0,0           ; vbroadcastss  0x4739(%rip),%ymm8        # 5e4c <_sk_callback_avx+0x20c>
+  DB  196,98,125,24,5,93,71,0,0           ; vbroadcastss  0x475d(%rip),%ymm8        # 5e70 <_sk_callback_avx+0x20c>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -6226,7 +6233,7 @@ _sk_hsl_to_rgb_avx LABEL PROC
   DB  197,252,17,28,36                    ; vmovups       %ymm3,(%rsp)
   DB  197,252,40,225                      ; vmovaps       %ymm1,%ymm4
   DB  197,252,40,216                      ; vmovaps       %ymm0,%ymm3
-  DB  196,98,125,24,5,0,71,0,0            ; vbroadcastss  0x4700(%rip),%ymm8        # 5e54 <_sk_callback_avx+0x214>
+  DB  196,98,125,24,5,36,71,0,0           ; vbroadcastss  0x4724(%rip),%ymm8        # 5e78 <_sk_callback_avx+0x214>
   DB  197,60,194,202,2                    ; vcmpleps      %ymm2,%ymm8,%ymm9
   DB  197,92,89,210                       ; vmulps        %ymm2,%ymm4,%ymm10
   DB  196,65,92,92,218                    ; vsubps        %ymm10,%ymm4,%ymm11
@@ -6234,23 +6241,23 @@ _sk_hsl_to_rgb_avx LABEL PROC
   DB  197,52,88,210                       ; vaddps        %ymm2,%ymm9,%ymm10
   DB  197,108,88,202                      ; vaddps        %ymm2,%ymm2,%ymm9
   DB  196,65,52,92,202                    ; vsubps        %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,29,218,70,0,0         ; vbroadcastss  0x46da(%rip),%ymm11        # 5e58 <_sk_callback_avx+0x218>
+  DB  196,98,125,24,29,254,70,0,0         ; vbroadcastss  0x46fe(%rip),%ymm11        # 5e7c <_sk_callback_avx+0x218>
   DB  196,65,100,88,219                   ; vaddps        %ymm11,%ymm3,%ymm11
   DB  196,67,125,8,227,1                  ; vroundps      $0x1,%ymm11,%ymm12
   DB  196,65,36,92,252                    ; vsubps        %ymm12,%ymm11,%ymm15
   DB  196,65,44,92,217                    ; vsubps        %ymm9,%ymm10,%ymm11
-  DB  196,98,125,24,37,196,70,0,0         ; vbroadcastss  0x46c4(%rip),%ymm12        # 5e60 <_sk_callback_avx+0x220>
+  DB  196,98,125,24,37,232,70,0,0         ; vbroadcastss  0x46e8(%rip),%ymm12        # 5e84 <_sk_callback_avx+0x220>
   DB  196,193,4,89,196                    ; vmulps        %ymm12,%ymm15,%ymm0
-  DB  196,98,125,24,45,186,70,0,0         ; vbroadcastss  0x46ba(%rip),%ymm13        # 5e64 <_sk_callback_avx+0x224>
+  DB  196,98,125,24,45,222,70,0,0         ; vbroadcastss  0x46de(%rip),%ymm13        # 5e88 <_sk_callback_avx+0x224>
   DB  197,20,92,240                       ; vsubps        %ymm0,%ymm13,%ymm14
   DB  196,65,36,89,246                    ; vmulps        %ymm14,%ymm11,%ymm14
   DB  196,65,52,88,246                    ; vaddps        %ymm14,%ymm9,%ymm14
-  DB  196,226,125,24,13,155,70,0,0        ; vbroadcastss  0x469b(%rip),%ymm1        # 5e5c <_sk_callback_avx+0x21c>
+  DB  196,226,125,24,13,191,70,0,0        ; vbroadcastss  0x46bf(%rip),%ymm1        # 5e80 <_sk_callback_avx+0x21c>
   DB  196,193,116,194,255,2               ; vcmpleps      %ymm15,%ymm1,%ymm7
   DB  196,195,13,74,249,112               ; vblendvps     %ymm7,%ymm9,%ymm14,%ymm7
   DB  196,65,60,194,247,2                 ; vcmpleps      %ymm15,%ymm8,%ymm14
   DB  196,227,45,74,255,224               ; vblendvps     %ymm14,%ymm7,%ymm10,%ymm7
-  DB  196,98,125,24,53,134,70,0,0         ; vbroadcastss  0x4686(%rip),%ymm14        # 5e68 <_sk_callback_avx+0x228>
+  DB  196,98,125,24,53,170,70,0,0         ; vbroadcastss  0x46aa(%rip),%ymm14        # 5e8c <_sk_callback_avx+0x228>
   DB  196,65,12,194,255,2                 ; vcmpleps      %ymm15,%ymm14,%ymm15
   DB  196,193,124,89,195                  ; vmulps        %ymm11,%ymm0,%ymm0
   DB  197,180,88,192                      ; vaddps        %ymm0,%ymm9,%ymm0
@@ -6269,7 +6276,7 @@ _sk_hsl_to_rgb_avx LABEL PROC
   DB  197,164,89,247                      ; vmulps        %ymm7,%ymm11,%ymm6
   DB  197,180,88,246                      ; vaddps        %ymm6,%ymm9,%ymm6
   DB  196,227,77,74,237,0                 ; vblendvps     %ymm0,%ymm5,%ymm6,%ymm5
-  DB  196,226,125,24,5,40,70,0,0          ; vbroadcastss  0x4628(%rip),%ymm0        # 5e6c <_sk_callback_avx+0x22c>
+  DB  196,226,125,24,5,76,70,0,0          ; vbroadcastss  0x464c(%rip),%ymm0        # 5e90 <_sk_callback_avx+0x22c>
   DB  197,228,88,192                      ; vaddps        %ymm0,%ymm3,%ymm0
   DB  196,227,125,8,216,1                 ; vroundps      $0x1,%ymm0,%ymm3
   DB  197,252,92,195                      ; vsubps        %ymm3,%ymm0,%ymm0
@@ -6324,7 +6331,7 @@ _sk_scale_u8_avx LABEL PROC
   DB  196,66,121,49,192                   ; vpmovzxbd     %xmm8,%xmm8
   DB  196,67,53,24,192,1                  ; vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,75,69,0,0          ; vbroadcastss  0x454b(%rip),%ymm9        # 5e70 <_sk_callback_avx+0x230>
+  DB  196,98,125,24,13,111,69,0,0         ; vbroadcastss  0x456f(%rip),%ymm9        # 5e94 <_sk_callback_avx+0x230>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -6379,7 +6386,7 @@ _sk_lerp_u8_avx LABEL PROC
   DB  196,66,121,49,192                   ; vpmovzxbd     %xmm8,%xmm8
   DB  196,67,53,24,192,1                  ; vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,151,68,0,0         ; vbroadcastss  0x4497(%rip),%ymm9        # 5e74 <_sk_callback_avx+0x234>
+  DB  196,98,125,24,13,187,68,0,0         ; vbroadcastss  0x44bb(%rip),%ymm9        # 5e98 <_sk_callback_avx+0x234>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
@@ -6414,80 +6421,86 @@ _sk_lerp_565_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,174,0,0,0                    ; jne           1b01 <_sk_lerp_565_avx+0xbc>
+  DB  15,133,208,0,0,0                    ; jne           1b23 <_sk_lerp_565_avx+0xde>
   DB  196,65,122,111,4,122                ; vmovdqu       (%r10,%rdi,2),%xmm8
-  DB  197,225,239,219                     ; vpxor         %xmm3,%xmm3,%xmm3
-  DB  197,185,105,219                     ; vpunpckhwd    %xmm3,%xmm8,%xmm3
+  DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
+  DB  196,65,57,105,201                   ; vpunpckhwd    %xmm9,%xmm8,%xmm9
   DB  196,66,121,51,192                   ; vpmovzxwd     %xmm8,%xmm8
-  DB  196,227,61,24,219,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm3
-  DB  196,98,125,24,5,3,68,0,0            ; vbroadcastss  0x4403(%rip),%ymm8        # 5e78 <_sk_callback_avx+0x238>
-  DB  196,65,100,84,192                   ; vandps        %ymm8,%ymm3,%ymm8
-  DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,244,67,0,0         ; vbroadcastss  0x43f4(%rip),%ymm9        # 5e7c <_sk_callback_avx+0x23c>
-  DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,234,67,0,0         ; vbroadcastss  0x43ea(%rip),%ymm9        # 5e80 <_sk_callback_avx+0x240>
-  DB  196,65,100,84,201                   ; vandps        %ymm9,%ymm3,%ymm9
+  DB  196,67,61,24,193,1                  ; vinsertf128   $0x1,%xmm9,%ymm8,%ymm8
+  DB  196,98,125,24,13,37,68,0,0          ; vbroadcastss  0x4425(%rip),%ymm9        # 5e9c <_sk_callback_avx+0x238>
+  DB  196,65,60,84,201                    ; vandps        %ymm9,%ymm8,%ymm9
   DB  196,65,124,91,201                   ; vcvtdq2ps     %ymm9,%ymm9
-  DB  196,98,125,24,21,219,67,0,0         ; vbroadcastss  0x43db(%rip),%ymm10        # 5e84 <_sk_callback_avx+0x244>
+  DB  196,98,125,24,21,22,68,0,0          ; vbroadcastss  0x4416(%rip),%ymm10        # 5ea0 <_sk_callback_avx+0x23c>
   DB  196,65,52,89,202                    ; vmulps        %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,21,209,67,0,0         ; vbroadcastss  0x43d1(%rip),%ymm10        # 5e88 <_sk_callback_avx+0x248>
-  DB  196,193,100,84,218                  ; vandps        %ymm10,%ymm3,%ymm3
-  DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,21,195,67,0,0         ; vbroadcastss  0x43c3(%rip),%ymm10        # 5e8c <_sk_callback_avx+0x24c>
-  DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
+  DB  196,98,125,24,21,12,68,0,0          ; vbroadcastss  0x440c(%rip),%ymm10        # 5ea4 <_sk_callback_avx+0x240>
+  DB  196,65,60,84,210                    ; vandps        %ymm10,%ymm8,%ymm10
+  DB  196,65,124,91,210                   ; vcvtdq2ps     %ymm10,%ymm10
+  DB  196,98,125,24,29,253,67,0,0         ; vbroadcastss  0x43fd(%rip),%ymm11        # 5ea8 <_sk_callback_avx+0x244>
+  DB  196,65,44,89,211                    ; vmulps        %ymm11,%ymm10,%ymm10
+  DB  196,98,125,24,29,243,67,0,0         ; vbroadcastss  0x43f3(%rip),%ymm11        # 5eac <_sk_callback_avx+0x248>
+  DB  196,65,60,84,195                    ; vandps        %ymm11,%ymm8,%ymm8
+  DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
+  DB  196,98,125,24,29,228,67,0,0         ; vbroadcastss  0x43e4(%rip),%ymm11        # 5eb0 <_sk_callback_avx+0x24c>
+  DB  196,65,60,89,195                    ; vmulps        %ymm11,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
-  DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
+  DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  197,252,88,196                      ; vaddps        %ymm4,%ymm0,%ymm0
   DB  197,244,92,205                      ; vsubps        %ymm5,%ymm1,%ymm1
-  DB  196,193,116,89,201                  ; vmulps        %ymm9,%ymm1,%ymm1
+  DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  197,244,88,205                      ; vaddps        %ymm5,%ymm1,%ymm1
   DB  197,236,92,214                      ; vsubps        %ymm6,%ymm2,%ymm2
-  DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
+  DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,236,88,214                      ; vaddps        %ymm6,%ymm2,%ymm2
+  DB  197,228,92,223                      ; vsubps        %ymm7,%ymm3,%ymm3
+  DB  196,65,100,89,201                   ; vmulps        %ymm9,%ymm3,%ymm9
+  DB  197,52,88,207                       ; vaddps        %ymm7,%ymm9,%ymm9
+  DB  196,65,100,89,210                   ; vmulps        %ymm10,%ymm3,%ymm10
+  DB  197,44,88,215                       ; vaddps        %ymm7,%ymm10,%ymm10
+  DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
+  DB  197,228,88,223                      ; vaddps        %ymm7,%ymm3,%ymm3
+  DB  197,172,95,219                      ; vmaxps        %ymm3,%ymm10,%ymm3
+  DB  197,180,95,219                      ; vmaxps        %ymm3,%ymm9,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,145,67,0,0        ; vbroadcastss  0x4391(%rip),%ymm3        # 5e90 <_sk_callback_avx+0x250>
   DB  255,224                             ; jmpq          *%rax
   DB  65,137,200                          ; mov           %ecx,%r8d
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  196,65,57,239,192                   ; vpxor         %xmm8,%xmm8,%xmm8
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,63,255,255,255               ; ja            1a59 <_sk_lerp_565_avx+0x14>
+  DB  15,135,29,255,255,255               ; ja            1a59 <_sk_lerp_565_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 1b70 <_sk_lerp_565_avx+0x12b>
+  DB  76,141,13,77,0,0,0                  ; lea           0x4d(%rip),%r9        # 1b94 <_sk_lerp_565_avx+0x14f>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
-  DB  197,225,239,219                     ; vpxor         %xmm3,%xmm3,%xmm3
-  DB  196,65,97,196,68,122,12,6           ; vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm3,%xmm8
+  DB  196,65,57,239,192                   ; vpxor         %xmm8,%xmm8,%xmm8
+  DB  196,65,57,196,68,122,12,6           ; vpinsrw       $0x6,0xc(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,10,5           ; vpinsrw       $0x5,0xa(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,8,4            ; vpinsrw       $0x4,0x8(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,6,3            ; vpinsrw       $0x3,0x6(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,4,2            ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,2,1            ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,4,122,0               ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
-  DB  233,235,254,255,255                 ; jmpq          1a59 <_sk_lerp_565_avx+0x14>
-  DB  102,144                             ; xchg          %ax,%ax
-  DB  242,255                             ; repnz         (bad)
+  DB  233,200,254,255,255                 ; jmpq          1a59 <_sk_lerp_565_avx+0x14>
+  DB  15,31,0                             ; nopl          (%rax)
+  DB  241                                 ; icebp
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  234                                 ; (bad)
   DB  255                                 ; (bad)
+  DB  233,255,255,255,225                 ; jmpq          ffffffffe2001b9c <_sk_callback_avx+0xffffffffe1ffbf38>
   DB  255                                 ; (bad)
-  DB  255,226                             ; jmpq          *%rdx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
+  DB  217,255                             ; fcos
   DB  255                                 ; (bad)
-  DB  218,255                             ; (bad)
-  DB  255                                 ; (bad)
-  DB  255,210                             ; callq         *%rdx
+  DB  255,209                             ; callq         *%rcx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,202                             ; dec           %edx
+  DB  255,201                             ; dec           %ecx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  190                                 ; .byte         0xbe
+  DB  188                                 ; .byte         0xbc
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
@@ -6497,7 +6510,7 @@ _sk_load_tables_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,26,2,0,0                     ; jne           1db4 <_sk_load_tables_avx+0x228>
+  DB  15,133,26,2,0,0                     ; jne           1dd8 <_sk_load_tables_avx+0x228>
   DB  196,65,124,16,4,184                 ; vmovups       (%r8,%rdi,4),%ymm8
   DB  85                                  ; push          %rbp
   DB  65,87                               ; push          %r15
@@ -6505,7 +6518,7 @@ _sk_load_tables_avx LABEL PROC
   DB  65,85                               ; push          %r13
   DB  65,84                               ; push          %r12
   DB  83                                  ; push          %rbx
-  DB  197,124,40,13,206,69,0,0            ; vmovaps       0x45ce(%rip),%ymm9        # 6180 <_sk_callback_avx+0x540>
+  DB  197,124,40,13,202,69,0,0            ; vmovaps       0x45ca(%rip),%ymm9        # 61a0 <_sk_callback_avx+0x53c>
   DB  196,193,60,84,193                   ; vandps        %ymm9,%ymm8,%ymm0
   DB  196,193,249,126,193                 ; vmovq         %xmm0,%r9
   DB  69,137,203                          ; mov           %r9d,%r11d
@@ -6597,7 +6610,7 @@ _sk_load_tables_avx LABEL PROC
   DB  196,193,97,114,210,24               ; vpsrld        $0x18,%xmm10,%xmm3
   DB  196,227,61,24,219,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,243,64,0,0          ; vbroadcastss  0x40f3(%rip),%ymm8        # 5e94 <_sk_callback_avx+0x254>
+  DB  196,98,125,24,5,239,64,0,0          ; vbroadcastss  0x40ef(%rip),%ymm8        # 5eb4 <_sk_callback_avx+0x250>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -6612,9 +6625,9 @@ _sk_load_tables_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  65,254,201                          ; dec           %r9b
   DB  65,128,249,6                        ; cmp           $0x6,%r9b
-  DB  15,135,211,253,255,255              ; ja            1ba0 <_sk_load_tables_avx+0x14>
+  DB  15,135,211,253,255,255              ; ja            1bc4 <_sk_load_tables_avx+0x14>
   DB  69,15,182,201                       ; movzbl        %r9b,%r9d
-  DB  76,141,21,140,0,0,0                 ; lea           0x8c(%rip),%r10        # 1e64 <_sk_load_tables_avx+0x2d8>
+  DB  76,141,21,140,0,0,0                 ; lea           0x8c(%rip),%r10        # 1e88 <_sk_load_tables_avx+0x2d8>
   DB  79,99,12,138                        ; movslq        (%r10,%r9,4),%r9
   DB  77,1,209                            ; add           %r10,%r9
   DB  65,255,225                          ; jmpq          *%r9
@@ -6637,7 +6650,7 @@ _sk_load_tables_avx LABEL PROC
   DB  196,99,61,12,192,15                 ; vblendps      $0xf,%ymm0,%ymm8,%ymm8
   DB  196,195,57,34,4,184,0               ; vpinsrd       $0x0,(%r8,%rdi,4),%xmm8,%xmm0
   DB  196,99,61,12,192,15                 ; vblendps      $0xf,%ymm0,%ymm8,%ymm8
-  DB  233,62,253,255,255                  ; jmpq          1ba0 <_sk_load_tables_avx+0x14>
+  DB  233,62,253,255,255                  ; jmpq          1bc4 <_sk_load_tables_avx+0x14>
   DB  102,144                             ; xchg          %ax,%ax
   DB  236                                 ; in            (%dx),%al
   DB  255                                 ; (bad)
@@ -6655,7 +6668,7 @@ _sk_load_tables_avx LABEL PROC
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  126,255                             ; jle           1e7d <_sk_load_tables_avx+0x2f1>
+  DB  126,255                             ; jle           1ea1 <_sk_load_tables_avx+0x2f1>
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
 
@@ -6665,7 +6678,7 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,113,2,0,0                    ; jne           2107 <_sk_load_tables_u16_be_avx+0x287>
+  DB  15,133,113,2,0,0                    ; jne           212b <_sk_load_tables_u16_be_avx+0x287>
   DB  196,1,121,16,4,72                   ; vmovupd       (%r8,%r9,2),%xmm8
   DB  196,129,121,16,84,72,16             ; vmovupd       0x10(%r8,%r9,2),%xmm2
   DB  196,129,121,16,92,72,32             ; vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -6687,7 +6700,7 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  197,177,108,208                     ; vpunpcklqdq   %xmm0,%xmm9,%xmm2
   DB  197,177,109,200                     ; vpunpckhqdq   %xmm0,%xmm9,%xmm1
   DB  196,65,57,108,212                   ; vpunpcklqdq   %xmm12,%xmm8,%xmm10
-  DB  197,121,111,29,14,67,0,0            ; vmovdqa       0x430e(%rip),%xmm11        # 6200 <_sk_callback_avx+0x5c0>
+  DB  197,121,111,29,10,67,0,0            ; vmovdqa       0x430a(%rip),%xmm11        # 6220 <_sk_callback_avx+0x5bc>
   DB  196,193,105,219,195                 ; vpand         %xmm11,%xmm2,%xmm0
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  196,193,121,105,209                 ; vpunpckhwd    %xmm9,%xmm0,%xmm2
@@ -6786,7 +6799,7 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  196,226,121,51,219                  ; vpmovzxwd     %xmm3,%xmm3
   DB  196,195,101,24,216,1                ; vinsertf128   $0x1,%xmm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,164,61,0,0          ; vbroadcastss  0x3da4(%rip),%ymm8        # 5e98 <_sk_callback_avx+0x258>
+  DB  196,98,125,24,5,160,61,0,0          ; vbroadcastss  0x3da0(%rip),%ymm8        # 5eb8 <_sk_callback_avx+0x254>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -6799,29 +6812,29 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  196,1,123,16,4,72                   ; vmovsd        (%r8,%r9,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            216d <_sk_load_tables_u16_be_avx+0x2ed>
+  DB  116,85                              ; je            2191 <_sk_load_tables_u16_be_avx+0x2ed>
   DB  196,1,57,22,68,72,8                 ; vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            216d <_sk_load_tables_u16_be_avx+0x2ed>
+  DB  114,72                              ; jb            2191 <_sk_load_tables_u16_be_avx+0x2ed>
   DB  196,129,123,16,84,72,16             ; vmovsd        0x10(%r8,%r9,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            217a <_sk_load_tables_u16_be_avx+0x2fa>
+  DB  116,72                              ; je            219e <_sk_load_tables_u16_be_avx+0x2fa>
   DB  196,129,105,22,84,72,24             ; vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            217a <_sk_load_tables_u16_be_avx+0x2fa>
+  DB  114,59                              ; jb            219e <_sk_load_tables_u16_be_avx+0x2fa>
   DB  196,129,123,16,92,72,32             ; vmovsd        0x20(%r8,%r9,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,97,253,255,255               ; je            1eb1 <_sk_load_tables_u16_be_avx+0x31>
+  DB  15,132,97,253,255,255               ; je            1ed5 <_sk_load_tables_u16_be_avx+0x31>
   DB  196,129,97,22,92,72,40              ; vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,80,253,255,255               ; jb            1eb1 <_sk_load_tables_u16_be_avx+0x31>
+  DB  15,130,80,253,255,255               ; jb            1ed5 <_sk_load_tables_u16_be_avx+0x31>
   DB  196,1,122,126,76,72,48              ; vmovq         0x30(%r8,%r9,2),%xmm9
-  DB  233,68,253,255,255                  ; jmpq          1eb1 <_sk_load_tables_u16_be_avx+0x31>
+  DB  233,68,253,255,255                  ; jmpq          1ed5 <_sk_load_tables_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,55,253,255,255                  ; jmpq          1eb1 <_sk_load_tables_u16_be_avx+0x31>
+  DB  233,55,253,255,255                  ; jmpq          1ed5 <_sk_load_tables_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,46,253,255,255                  ; jmpq          1eb1 <_sk_load_tables_u16_be_avx+0x31>
+  DB  233,46,253,255,255                  ; jmpq          1ed5 <_sk_load_tables_u16_be_avx+0x31>
 
 PUBLIC _sk_load_tables_rgb_u16_be_avx
 _sk_load_tables_rgb_u16_be_avx LABEL PROC
@@ -6829,7 +6842,7 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,127                       ; lea           (%rdi,%rdi,2),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,93,2,0,0                     ; jne           23f2 <_sk_load_tables_rgb_u16_be_avx+0x26f>
+  DB  15,133,93,2,0,0                     ; jne           2416 <_sk_load_tables_rgb_u16_be_avx+0x26f>
   DB  196,129,122,111,4,72                ; vmovdqu       (%r8,%r9,2),%xmm0
   DB  196,129,122,111,84,72,12            ; vmovdqu       0xc(%r8,%r9,2),%xmm2
   DB  196,129,122,111,76,72,24            ; vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -6856,7 +6869,7 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  197,185,108,202                     ; vpunpcklqdq   %xmm2,%xmm8,%xmm1
   DB  197,185,109,210                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm2
   DB  197,121,108,195                     ; vpunpcklqdq   %xmm3,%xmm0,%xmm8
-  DB  197,121,111,13,7,64,0,0             ; vmovdqa       0x4007(%rip),%xmm9        # 6210 <_sk_callback_avx+0x5d0>
+  DB  197,121,111,13,3,64,0,0             ; vmovdqa       0x4003(%rip),%xmm9        # 6230 <_sk_callback_avx+0x5cc>
   DB  196,193,113,219,193                 ; vpand         %xmm9,%xmm1,%xmm0
   DB  196,65,41,239,210                   ; vpxor         %xmm10,%xmm10,%xmm10
   DB  196,193,121,105,202                 ; vpunpckhwd    %xmm10,%xmm0,%xmm1
@@ -6948,7 +6961,7 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  196,227,105,33,211,48               ; vinsertps     $0x30,%xmm3,%xmm2,%xmm2
   DB  196,195,109,24,208,1                ; vinsertf128   $0x1,%xmm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,182,58,0,0        ; vbroadcastss  0x3ab6(%rip),%ymm3        # 5e9c <_sk_callback_avx+0x25c>
+  DB  196,226,125,24,29,178,58,0,0        ; vbroadcastss  0x3ab2(%rip),%ymm3        # 5ebc <_sk_callback_avx+0x258>
   DB  91                                  ; pop           %rbx
   DB  65,92                               ; pop           %r12
   DB  65,93                               ; pop           %r13
@@ -6959,36 +6972,36 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  196,129,121,110,4,72                ; vmovd         (%r8,%r9,2),%xmm0
   DB  196,129,121,196,68,72,4,2           ; vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           240b <_sk_load_tables_rgb_u16_be_avx+0x288>
-  DB  233,190,253,255,255                 ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  117,5                               ; jne           242f <_sk_load_tables_rgb_u16_be_avx+0x288>
+  DB  233,190,253,255,255                 ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
   DB  196,129,121,110,76,72,6             ; vmovd         0x6(%r8,%r9,2),%xmm1
   DB  196,1,113,196,68,72,10,2            ; vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            243a <_sk_load_tables_rgb_u16_be_avx+0x2b7>
+  DB  114,26                              ; jb            245e <_sk_load_tables_rgb_u16_be_avx+0x2b7>
   DB  196,129,121,110,76,72,12            ; vmovd         0xc(%r8,%r9,2),%xmm1
   DB  196,129,113,196,84,72,16,2          ; vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           243f <_sk_load_tables_rgb_u16_be_avx+0x2bc>
-  DB  233,143,253,255,255                 ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  DB  233,138,253,255,255                 ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           2463 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
+  DB  233,143,253,255,255                 ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,138,253,255,255                 ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
   DB  196,129,121,110,76,72,18            ; vmovd         0x12(%r8,%r9,2),%xmm1
   DB  196,1,113,196,76,72,22,2            ; vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            246e <_sk_load_tables_rgb_u16_be_avx+0x2eb>
+  DB  114,26                              ; jb            2492 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
   DB  196,129,121,110,76,72,24            ; vmovd         0x18(%r8,%r9,2),%xmm1
   DB  196,129,113,196,76,72,28,2          ; vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           2473 <_sk_load_tables_rgb_u16_be_avx+0x2f0>
-  DB  233,91,253,255,255                  ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  DB  233,86,253,255,255                  ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           2497 <_sk_load_tables_rgb_u16_be_avx+0x2f0>
+  DB  233,91,253,255,255                  ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,86,253,255,255                  ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
   DB  196,129,121,110,92,72,30            ; vmovd         0x1e(%r8,%r9,2),%xmm3
   DB  196,1,97,196,92,72,34,2             ; vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            249c <_sk_load_tables_rgb_u16_be_avx+0x319>
+  DB  114,20                              ; jb            24c0 <_sk_load_tables_rgb_u16_be_avx+0x319>
   DB  196,129,121,110,92,72,36            ; vmovd         0x24(%r8,%r9,2),%xmm3
   DB  196,129,97,196,92,72,40,2           ; vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  DB  233,45,253,255,255                  ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  DB  233,40,253,255,255                  ; jmpq          21c9 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,45,253,255,255                  ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,40,253,255,255                  ; jmpq          21ed <_sk_load_tables_rgb_u16_be_avx+0x46>
 
 PUBLIC _sk_byte_tables_avx
 _sk_byte_tables_avx LABEL PROC
@@ -6999,7 +7012,7 @@ _sk_byte_tables_avx LABEL PROC
   DB  65,84                               ; push          %r12
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,234,57,0,0          ; vbroadcastss  0x39ea(%rip),%ymm8        # 5ea0 <_sk_callback_avx+0x260>
+  DB  196,98,125,24,5,230,57,0,0          ; vbroadcastss  0x39e6(%rip),%ymm8        # 5ec0 <_sk_callback_avx+0x25c>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,195,249,22,192,1                ; vpextrq       $0x1,%xmm0,%r8
@@ -7036,7 +7049,7 @@ _sk_byte_tables_avx LABEL PROC
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,53,24,192,1                 ; vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,56,57,0,0          ; vbroadcastss  0x3938(%rip),%ymm9        # 5ea4 <_sk_callback_avx+0x264>
+  DB  196,98,125,24,13,52,57,0,0          ; vbroadcastss  0x3934(%rip),%ymm9        # 5ec4 <_sk_callback_avx+0x260>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -7196,7 +7209,7 @@ _sk_byte_tables_rgb_avx LABEL PROC
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,53,24,192,1                 ; vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,94,54,0,0          ; vbroadcastss  0x365e(%rip),%ymm9        # 5ea8 <_sk_callback_avx+0x268>
+  DB  196,98,125,24,13,90,54,0,0          ; vbroadcastss  0x365a(%rip),%ymm9        # 5ec8 <_sk_callback_avx+0x264>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -7483,36 +7496,36 @@ _sk_parametric_r_avx LABEL PROC
   DB  196,193,124,88,195                  ; vaddps        %ymm11,%ymm0,%ymm0
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,216                      ; vcvtdq2ps     %ymm0,%ymm11
-  DB  196,98,125,24,37,188,49,0,0         ; vbroadcastss  0x31bc(%rip),%ymm12        # 5eac <_sk_callback_avx+0x26c>
+  DB  196,98,125,24,37,184,49,0,0         ; vbroadcastss  0x31b8(%rip),%ymm12        # 5ecc <_sk_callback_avx+0x268>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,178,49,0,0         ; vbroadcastss  0x31b2(%rip),%ymm12        # 5eb0 <_sk_callback_avx+0x270>
+  DB  196,98,125,24,37,174,49,0,0         ; vbroadcastss  0x31ae(%rip),%ymm12        # 5ed0 <_sk_callback_avx+0x26c>
   DB  196,193,124,84,196                  ; vandps        %ymm12,%ymm0,%ymm0
-  DB  196,98,125,24,37,168,49,0,0         ; vbroadcastss  0x31a8(%rip),%ymm12        # 5eb4 <_sk_callback_avx+0x274>
+  DB  196,98,125,24,37,164,49,0,0         ; vbroadcastss  0x31a4(%rip),%ymm12        # 5ed4 <_sk_callback_avx+0x270>
   DB  196,193,124,86,196                  ; vorps         %ymm12,%ymm0,%ymm0
-  DB  196,98,125,24,37,158,49,0,0         ; vbroadcastss  0x319e(%rip),%ymm12        # 5eb8 <_sk_callback_avx+0x278>
+  DB  196,98,125,24,37,154,49,0,0         ; vbroadcastss  0x319a(%rip),%ymm12        # 5ed8 <_sk_callback_avx+0x274>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,148,49,0,0         ; vbroadcastss  0x3194(%rip),%ymm12        # 5ebc <_sk_callback_avx+0x27c>
+  DB  196,98,125,24,37,144,49,0,0         ; vbroadcastss  0x3190(%rip),%ymm12        # 5edc <_sk_callback_avx+0x278>
   DB  196,65,124,89,228                   ; vmulps        %ymm12,%ymm0,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,133,49,0,0         ; vbroadcastss  0x3185(%rip),%ymm12        # 5ec0 <_sk_callback_avx+0x280>
+  DB  196,98,125,24,37,129,49,0,0         ; vbroadcastss  0x3181(%rip),%ymm12        # 5ee0 <_sk_callback_avx+0x27c>
   DB  196,193,124,88,196                  ; vaddps        %ymm12,%ymm0,%ymm0
-  DB  196,98,125,24,37,123,49,0,0         ; vbroadcastss  0x317b(%rip),%ymm12        # 5ec4 <_sk_callback_avx+0x284>
+  DB  196,98,125,24,37,119,49,0,0         ; vbroadcastss  0x3177(%rip),%ymm12        # 5ee4 <_sk_callback_avx+0x280>
   DB  197,156,94,192                      ; vdivps        %ymm0,%ymm12,%ymm0
   DB  197,164,92,192                      ; vsubps        %ymm0,%ymm11,%ymm0
   DB  197,172,89,192                      ; vmulps        %ymm0,%ymm10,%ymm0
   DB  196,99,125,8,208,1                  ; vroundps      $0x1,%ymm0,%ymm10
   DB  196,65,124,92,210                   ; vsubps        %ymm10,%ymm0,%ymm10
-  DB  196,98,125,24,29,95,49,0,0          ; vbroadcastss  0x315f(%rip),%ymm11        # 5ec8 <_sk_callback_avx+0x288>
+  DB  196,98,125,24,29,91,49,0,0          ; vbroadcastss  0x315b(%rip),%ymm11        # 5ee8 <_sk_callback_avx+0x284>
   DB  196,193,124,88,195                  ; vaddps        %ymm11,%ymm0,%ymm0
-  DB  196,98,125,24,29,85,49,0,0          ; vbroadcastss  0x3155(%rip),%ymm11        # 5ecc <_sk_callback_avx+0x28c>
+  DB  196,98,125,24,29,81,49,0,0          ; vbroadcastss  0x3151(%rip),%ymm11        # 5eec <_sk_callback_avx+0x288>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,124,92,195                  ; vsubps        %ymm11,%ymm0,%ymm0
-  DB  196,98,125,24,29,70,49,0,0          ; vbroadcastss  0x3146(%rip),%ymm11        # 5ed0 <_sk_callback_avx+0x290>
+  DB  196,98,125,24,29,66,49,0,0          ; vbroadcastss  0x3142(%rip),%ymm11        # 5ef0 <_sk_callback_avx+0x28c>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,60,49,0,0          ; vbroadcastss  0x313c(%rip),%ymm11        # 5ed4 <_sk_callback_avx+0x294>
+  DB  196,98,125,24,29,56,49,0,0          ; vbroadcastss  0x3138(%rip),%ymm11        # 5ef4 <_sk_callback_avx+0x290>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,124,88,194                  ; vaddps        %ymm10,%ymm0,%ymm0
-  DB  196,98,125,24,21,45,49,0,0          ; vbroadcastss  0x312d(%rip),%ymm10        # 5ed8 <_sk_callback_avx+0x298>
+  DB  196,98,125,24,21,41,49,0,0          ; vbroadcastss  0x3129(%rip),%ymm10        # 5ef8 <_sk_callback_avx+0x294>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7520,7 +7533,7 @@ _sk_parametric_r_avx LABEL PROC
   DB  196,195,125,74,193,128              ; vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,124,95,192                  ; vmaxps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,4,49,0,0            ; vbroadcastss  0x3104(%rip),%ymm8        # 5edc <_sk_callback_avx+0x29c>
+  DB  196,98,125,24,5,0,49,0,0            ; vbroadcastss  0x3100(%rip),%ymm8        # 5efc <_sk_callback_avx+0x298>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7540,36 +7553,36 @@ _sk_parametric_g_avx LABEL PROC
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,217                      ; vcvtdq2ps     %ymm1,%ymm11
-  DB  196,98,125,24,37,181,48,0,0         ; vbroadcastss  0x30b5(%rip),%ymm12        # 5ee0 <_sk_callback_avx+0x2a0>
+  DB  196,98,125,24,37,177,48,0,0         ; vbroadcastss  0x30b1(%rip),%ymm12        # 5f00 <_sk_callback_avx+0x29c>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,171,48,0,0         ; vbroadcastss  0x30ab(%rip),%ymm12        # 5ee4 <_sk_callback_avx+0x2a4>
+  DB  196,98,125,24,37,167,48,0,0         ; vbroadcastss  0x30a7(%rip),%ymm12        # 5f04 <_sk_callback_avx+0x2a0>
   DB  196,193,116,84,204                  ; vandps        %ymm12,%ymm1,%ymm1
-  DB  196,98,125,24,37,161,48,0,0         ; vbroadcastss  0x30a1(%rip),%ymm12        # 5ee8 <_sk_callback_avx+0x2a8>
+  DB  196,98,125,24,37,157,48,0,0         ; vbroadcastss  0x309d(%rip),%ymm12        # 5f08 <_sk_callback_avx+0x2a4>
   DB  196,193,116,86,204                  ; vorps         %ymm12,%ymm1,%ymm1
-  DB  196,98,125,24,37,151,48,0,0         ; vbroadcastss  0x3097(%rip),%ymm12        # 5eec <_sk_callback_avx+0x2ac>
+  DB  196,98,125,24,37,147,48,0,0         ; vbroadcastss  0x3093(%rip),%ymm12        # 5f0c <_sk_callback_avx+0x2a8>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,141,48,0,0         ; vbroadcastss  0x308d(%rip),%ymm12        # 5ef0 <_sk_callback_avx+0x2b0>
+  DB  196,98,125,24,37,137,48,0,0         ; vbroadcastss  0x3089(%rip),%ymm12        # 5f10 <_sk_callback_avx+0x2ac>
   DB  196,65,116,89,228                   ; vmulps        %ymm12,%ymm1,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,126,48,0,0         ; vbroadcastss  0x307e(%rip),%ymm12        # 5ef4 <_sk_callback_avx+0x2b4>
+  DB  196,98,125,24,37,122,48,0,0         ; vbroadcastss  0x307a(%rip),%ymm12        # 5f14 <_sk_callback_avx+0x2b0>
   DB  196,193,116,88,204                  ; vaddps        %ymm12,%ymm1,%ymm1
-  DB  196,98,125,24,37,116,48,0,0         ; vbroadcastss  0x3074(%rip),%ymm12        # 5ef8 <_sk_callback_avx+0x2b8>
+  DB  196,98,125,24,37,112,48,0,0         ; vbroadcastss  0x3070(%rip),%ymm12        # 5f18 <_sk_callback_avx+0x2b4>
   DB  197,156,94,201                      ; vdivps        %ymm1,%ymm12,%ymm1
   DB  197,164,92,201                      ; vsubps        %ymm1,%ymm11,%ymm1
   DB  197,172,89,201                      ; vmulps        %ymm1,%ymm10,%ymm1
   DB  196,99,125,8,209,1                  ; vroundps      $0x1,%ymm1,%ymm10
   DB  196,65,116,92,210                   ; vsubps        %ymm10,%ymm1,%ymm10
-  DB  196,98,125,24,29,88,48,0,0          ; vbroadcastss  0x3058(%rip),%ymm11        # 5efc <_sk_callback_avx+0x2bc>
+  DB  196,98,125,24,29,84,48,0,0          ; vbroadcastss  0x3054(%rip),%ymm11        # 5f1c <_sk_callback_avx+0x2b8>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,78,48,0,0          ; vbroadcastss  0x304e(%rip),%ymm11        # 5f00 <_sk_callback_avx+0x2c0>
+  DB  196,98,125,24,29,74,48,0,0          ; vbroadcastss  0x304a(%rip),%ymm11        # 5f20 <_sk_callback_avx+0x2bc>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,116,92,203                  ; vsubps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,63,48,0,0          ; vbroadcastss  0x303f(%rip),%ymm11        # 5f04 <_sk_callback_avx+0x2c4>
+  DB  196,98,125,24,29,59,48,0,0          ; vbroadcastss  0x303b(%rip),%ymm11        # 5f24 <_sk_callback_avx+0x2c0>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,53,48,0,0          ; vbroadcastss  0x3035(%rip),%ymm11        # 5f08 <_sk_callback_avx+0x2c8>
+  DB  196,98,125,24,29,49,48,0,0          ; vbroadcastss  0x3031(%rip),%ymm11        # 5f28 <_sk_callback_avx+0x2c4>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,116,88,202                  ; vaddps        %ymm10,%ymm1,%ymm1
-  DB  196,98,125,24,21,38,48,0,0          ; vbroadcastss  0x3026(%rip),%ymm10        # 5f0c <_sk_callback_avx+0x2cc>
+  DB  196,98,125,24,21,34,48,0,0          ; vbroadcastss  0x3022(%rip),%ymm10        # 5f2c <_sk_callback_avx+0x2c8>
   DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7577,7 +7590,7 @@ _sk_parametric_g_avx LABEL PROC
   DB  196,195,117,74,201,128              ; vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,116,95,200                  ; vmaxps        %ymm8,%ymm1,%ymm1
-  DB  196,98,125,24,5,253,47,0,0          ; vbroadcastss  0x2ffd(%rip),%ymm8        # 5f10 <_sk_callback_avx+0x2d0>
+  DB  196,98,125,24,5,249,47,0,0          ; vbroadcastss  0x2ff9(%rip),%ymm8        # 5f30 <_sk_callback_avx+0x2cc>
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7597,36 +7610,36 @@ _sk_parametric_b_avx LABEL PROC
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,218                      ; vcvtdq2ps     %ymm2,%ymm11
-  DB  196,98,125,24,37,174,47,0,0         ; vbroadcastss  0x2fae(%rip),%ymm12        # 5f14 <_sk_callback_avx+0x2d4>
+  DB  196,98,125,24,37,170,47,0,0         ; vbroadcastss  0x2faa(%rip),%ymm12        # 5f34 <_sk_callback_avx+0x2d0>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,164,47,0,0         ; vbroadcastss  0x2fa4(%rip),%ymm12        # 5f18 <_sk_callback_avx+0x2d8>
+  DB  196,98,125,24,37,160,47,0,0         ; vbroadcastss  0x2fa0(%rip),%ymm12        # 5f38 <_sk_callback_avx+0x2d4>
   DB  196,193,108,84,212                  ; vandps        %ymm12,%ymm2,%ymm2
-  DB  196,98,125,24,37,154,47,0,0         ; vbroadcastss  0x2f9a(%rip),%ymm12        # 5f1c <_sk_callback_avx+0x2dc>
+  DB  196,98,125,24,37,150,47,0,0         ; vbroadcastss  0x2f96(%rip),%ymm12        # 5f3c <_sk_callback_avx+0x2d8>
   DB  196,193,108,86,212                  ; vorps         %ymm12,%ymm2,%ymm2
-  DB  196,98,125,24,37,144,47,0,0         ; vbroadcastss  0x2f90(%rip),%ymm12        # 5f20 <_sk_callback_avx+0x2e0>
+  DB  196,98,125,24,37,140,47,0,0         ; vbroadcastss  0x2f8c(%rip),%ymm12        # 5f40 <_sk_callback_avx+0x2dc>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,134,47,0,0         ; vbroadcastss  0x2f86(%rip),%ymm12        # 5f24 <_sk_callback_avx+0x2e4>
+  DB  196,98,125,24,37,130,47,0,0         ; vbroadcastss  0x2f82(%rip),%ymm12        # 5f44 <_sk_callback_avx+0x2e0>
   DB  196,65,108,89,228                   ; vmulps        %ymm12,%ymm2,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,119,47,0,0         ; vbroadcastss  0x2f77(%rip),%ymm12        # 5f28 <_sk_callback_avx+0x2e8>
+  DB  196,98,125,24,37,115,47,0,0         ; vbroadcastss  0x2f73(%rip),%ymm12        # 5f48 <_sk_callback_avx+0x2e4>
   DB  196,193,108,88,212                  ; vaddps        %ymm12,%ymm2,%ymm2
-  DB  196,98,125,24,37,109,47,0,0         ; vbroadcastss  0x2f6d(%rip),%ymm12        # 5f2c <_sk_callback_avx+0x2ec>
+  DB  196,98,125,24,37,105,47,0,0         ; vbroadcastss  0x2f69(%rip),%ymm12        # 5f4c <_sk_callback_avx+0x2e8>
   DB  197,156,94,210                      ; vdivps        %ymm2,%ymm12,%ymm2
   DB  197,164,92,210                      ; vsubps        %ymm2,%ymm11,%ymm2
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  196,99,125,8,210,1                  ; vroundps      $0x1,%ymm2,%ymm10
   DB  196,65,108,92,210                   ; vsubps        %ymm10,%ymm2,%ymm10
-  DB  196,98,125,24,29,81,47,0,0          ; vbroadcastss  0x2f51(%rip),%ymm11        # 5f30 <_sk_callback_avx+0x2f0>
+  DB  196,98,125,24,29,77,47,0,0          ; vbroadcastss  0x2f4d(%rip),%ymm11        # 5f50 <_sk_callback_avx+0x2ec>
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
-  DB  196,98,125,24,29,71,47,0,0          ; vbroadcastss  0x2f47(%rip),%ymm11        # 5f34 <_sk_callback_avx+0x2f4>
+  DB  196,98,125,24,29,67,47,0,0          ; vbroadcastss  0x2f43(%rip),%ymm11        # 5f54 <_sk_callback_avx+0x2f0>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,108,92,211                  ; vsubps        %ymm11,%ymm2,%ymm2
-  DB  196,98,125,24,29,56,47,0,0          ; vbroadcastss  0x2f38(%rip),%ymm11        # 5f38 <_sk_callback_avx+0x2f8>
+  DB  196,98,125,24,29,52,47,0,0          ; vbroadcastss  0x2f34(%rip),%ymm11        # 5f58 <_sk_callback_avx+0x2f4>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,46,47,0,0          ; vbroadcastss  0x2f2e(%rip),%ymm11        # 5f3c <_sk_callback_avx+0x2fc>
+  DB  196,98,125,24,29,42,47,0,0          ; vbroadcastss  0x2f2a(%rip),%ymm11        # 5f5c <_sk_callback_avx+0x2f8>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,108,88,210                  ; vaddps        %ymm10,%ymm2,%ymm2
-  DB  196,98,125,24,21,31,47,0,0          ; vbroadcastss  0x2f1f(%rip),%ymm10        # 5f40 <_sk_callback_avx+0x300>
+  DB  196,98,125,24,21,27,47,0,0          ; vbroadcastss  0x2f1b(%rip),%ymm10        # 5f60 <_sk_callback_avx+0x2fc>
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  197,253,91,210                      ; vcvtps2dq     %ymm2,%ymm2
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7634,7 +7647,7 @@ _sk_parametric_b_avx LABEL PROC
   DB  196,195,109,74,209,128              ; vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,108,95,208                  ; vmaxps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,246,46,0,0          ; vbroadcastss  0x2ef6(%rip),%ymm8        # 5f44 <_sk_callback_avx+0x304>
+  DB  196,98,125,24,5,242,46,0,0          ; vbroadcastss  0x2ef2(%rip),%ymm8        # 5f64 <_sk_callback_avx+0x300>
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7654,36 +7667,36 @@ _sk_parametric_a_avx LABEL PROC
   DB  196,193,100,88,219                  ; vaddps        %ymm11,%ymm3,%ymm3
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,219                      ; vcvtdq2ps     %ymm3,%ymm11
-  DB  196,98,125,24,37,167,46,0,0         ; vbroadcastss  0x2ea7(%rip),%ymm12        # 5f48 <_sk_callback_avx+0x308>
+  DB  196,98,125,24,37,163,46,0,0         ; vbroadcastss  0x2ea3(%rip),%ymm12        # 5f68 <_sk_callback_avx+0x304>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,157,46,0,0         ; vbroadcastss  0x2e9d(%rip),%ymm12        # 5f4c <_sk_callback_avx+0x30c>
+  DB  196,98,125,24,37,153,46,0,0         ; vbroadcastss  0x2e99(%rip),%ymm12        # 5f6c <_sk_callback_avx+0x308>
   DB  196,193,100,84,220                  ; vandps        %ymm12,%ymm3,%ymm3
-  DB  196,98,125,24,37,147,46,0,0         ; vbroadcastss  0x2e93(%rip),%ymm12        # 5f50 <_sk_callback_avx+0x310>
+  DB  196,98,125,24,37,143,46,0,0         ; vbroadcastss  0x2e8f(%rip),%ymm12        # 5f70 <_sk_callback_avx+0x30c>
   DB  196,193,100,86,220                  ; vorps         %ymm12,%ymm3,%ymm3
-  DB  196,98,125,24,37,137,46,0,0         ; vbroadcastss  0x2e89(%rip),%ymm12        # 5f54 <_sk_callback_avx+0x314>
+  DB  196,98,125,24,37,133,46,0,0         ; vbroadcastss  0x2e85(%rip),%ymm12        # 5f74 <_sk_callback_avx+0x310>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,127,46,0,0         ; vbroadcastss  0x2e7f(%rip),%ymm12        # 5f58 <_sk_callback_avx+0x318>
+  DB  196,98,125,24,37,123,46,0,0         ; vbroadcastss  0x2e7b(%rip),%ymm12        # 5f78 <_sk_callback_avx+0x314>
   DB  196,65,100,89,228                   ; vmulps        %ymm12,%ymm3,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,112,46,0,0         ; vbroadcastss  0x2e70(%rip),%ymm12        # 5f5c <_sk_callback_avx+0x31c>
+  DB  196,98,125,24,37,108,46,0,0         ; vbroadcastss  0x2e6c(%rip),%ymm12        # 5f7c <_sk_callback_avx+0x318>
   DB  196,193,100,88,220                  ; vaddps        %ymm12,%ymm3,%ymm3
-  DB  196,98,125,24,37,102,46,0,0         ; vbroadcastss  0x2e66(%rip),%ymm12        # 5f60 <_sk_callback_avx+0x320>
+  DB  196,98,125,24,37,98,46,0,0          ; vbroadcastss  0x2e62(%rip),%ymm12        # 5f80 <_sk_callback_avx+0x31c>
   DB  197,156,94,219                      ; vdivps        %ymm3,%ymm12,%ymm3
   DB  197,164,92,219                      ; vsubps        %ymm3,%ymm11,%ymm3
   DB  197,172,89,219                      ; vmulps        %ymm3,%ymm10,%ymm3
   DB  196,99,125,8,211,1                  ; vroundps      $0x1,%ymm3,%ymm10
   DB  196,65,100,92,210                   ; vsubps        %ymm10,%ymm3,%ymm10
-  DB  196,98,125,24,29,74,46,0,0          ; vbroadcastss  0x2e4a(%rip),%ymm11        # 5f64 <_sk_callback_avx+0x324>
+  DB  196,98,125,24,29,70,46,0,0          ; vbroadcastss  0x2e46(%rip),%ymm11        # 5f84 <_sk_callback_avx+0x320>
   DB  196,193,100,88,219                  ; vaddps        %ymm11,%ymm3,%ymm3
-  DB  196,98,125,24,29,64,46,0,0          ; vbroadcastss  0x2e40(%rip),%ymm11        # 5f68 <_sk_callback_avx+0x328>
+  DB  196,98,125,24,29,60,46,0,0          ; vbroadcastss  0x2e3c(%rip),%ymm11        # 5f88 <_sk_callback_avx+0x324>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,100,92,219                  ; vsubps        %ymm11,%ymm3,%ymm3
-  DB  196,98,125,24,29,49,46,0,0          ; vbroadcastss  0x2e31(%rip),%ymm11        # 5f6c <_sk_callback_avx+0x32c>
+  DB  196,98,125,24,29,45,46,0,0          ; vbroadcastss  0x2e2d(%rip),%ymm11        # 5f8c <_sk_callback_avx+0x328>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,39,46,0,0          ; vbroadcastss  0x2e27(%rip),%ymm11        # 5f70 <_sk_callback_avx+0x330>
+  DB  196,98,125,24,29,35,46,0,0          ; vbroadcastss  0x2e23(%rip),%ymm11        # 5f90 <_sk_callback_avx+0x32c>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,100,88,218                  ; vaddps        %ymm10,%ymm3,%ymm3
-  DB  196,98,125,24,21,24,46,0,0          ; vbroadcastss  0x2e18(%rip),%ymm10        # 5f74 <_sk_callback_avx+0x334>
+  DB  196,98,125,24,21,20,46,0,0          ; vbroadcastss  0x2e14(%rip),%ymm10        # 5f94 <_sk_callback_avx+0x330>
   DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
   DB  197,253,91,219                      ; vcvtps2dq     %ymm3,%ymm3
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7691,38 +7704,38 @@ _sk_parametric_a_avx LABEL PROC
   DB  196,195,101,74,217,128              ; vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,100,95,216                  ; vmaxps        %ymm8,%ymm3,%ymm3
-  DB  196,98,125,24,5,239,45,0,0          ; vbroadcastss  0x2def(%rip),%ymm8        # 5f78 <_sk_callback_avx+0x338>
+  DB  196,98,125,24,5,235,45,0,0          ; vbroadcastss  0x2deb(%rip),%ymm8        # 5f98 <_sk_callback_avx+0x334>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_lab_to_xyz_avx
 _sk_lab_to_xyz_avx LABEL PROC
-  DB  196,98,125,24,5,225,45,0,0          ; vbroadcastss  0x2de1(%rip),%ymm8        # 5f7c <_sk_callback_avx+0x33c>
+  DB  196,98,125,24,5,221,45,0,0          ; vbroadcastss  0x2ddd(%rip),%ymm8        # 5f9c <_sk_callback_avx+0x338>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,215,45,0,0          ; vbroadcastss  0x2dd7(%rip),%ymm8        # 5f80 <_sk_callback_avx+0x340>
+  DB  196,98,125,24,5,211,45,0,0          ; vbroadcastss  0x2dd3(%rip),%ymm8        # 5fa0 <_sk_callback_avx+0x33c>
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,98,125,24,13,205,45,0,0         ; vbroadcastss  0x2dcd(%rip),%ymm9        # 5f84 <_sk_callback_avx+0x344>
+  DB  196,98,125,24,13,201,45,0,0         ; vbroadcastss  0x2dc9(%rip),%ymm9        # 5fa4 <_sk_callback_avx+0x340>
   DB  196,193,116,88,201                  ; vaddps        %ymm9,%ymm1,%ymm1
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  196,193,108,88,209                  ; vaddps        %ymm9,%ymm2,%ymm2
-  DB  196,98,125,24,5,185,45,0,0          ; vbroadcastss  0x2db9(%rip),%ymm8        # 5f88 <_sk_callback_avx+0x348>
+  DB  196,98,125,24,5,181,45,0,0          ; vbroadcastss  0x2db5(%rip),%ymm8        # 5fa8 <_sk_callback_avx+0x344>
   DB  196,193,124,88,192                  ; vaddps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,175,45,0,0          ; vbroadcastss  0x2daf(%rip),%ymm8        # 5f8c <_sk_callback_avx+0x34c>
+  DB  196,98,125,24,5,171,45,0,0          ; vbroadcastss  0x2dab(%rip),%ymm8        # 5fac <_sk_callback_avx+0x348>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,165,45,0,0          ; vbroadcastss  0x2da5(%rip),%ymm8        # 5f90 <_sk_callback_avx+0x350>
+  DB  196,98,125,24,5,161,45,0,0          ; vbroadcastss  0x2da1(%rip),%ymm8        # 5fb0 <_sk_callback_avx+0x34c>
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  197,252,88,201                      ; vaddps        %ymm1,%ymm0,%ymm1
-  DB  196,98,125,24,5,151,45,0,0          ; vbroadcastss  0x2d97(%rip),%ymm8        # 5f94 <_sk_callback_avx+0x354>
+  DB  196,98,125,24,5,147,45,0,0          ; vbroadcastss  0x2d93(%rip),%ymm8        # 5fb4 <_sk_callback_avx+0x350>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,252,92,210                      ; vsubps        %ymm2,%ymm0,%ymm2
   DB  197,116,89,193                      ; vmulps        %ymm1,%ymm1,%ymm8
   DB  196,65,116,89,192                   ; vmulps        %ymm8,%ymm1,%ymm8
-  DB  196,98,125,24,13,128,45,0,0         ; vbroadcastss  0x2d80(%rip),%ymm9        # 5f98 <_sk_callback_avx+0x358>
+  DB  196,98,125,24,13,124,45,0,0         ; vbroadcastss  0x2d7c(%rip),%ymm9        # 5fb8 <_sk_callback_avx+0x354>
   DB  196,65,52,194,208,1                 ; vcmpltps      %ymm8,%ymm9,%ymm10
-  DB  196,98,125,24,29,117,45,0,0         ; vbroadcastss  0x2d75(%rip),%ymm11        # 5f9c <_sk_callback_avx+0x35c>
+  DB  196,98,125,24,29,113,45,0,0         ; vbroadcastss  0x2d71(%rip),%ymm11        # 5fbc <_sk_callback_avx+0x358>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,37,107,45,0,0         ; vbroadcastss  0x2d6b(%rip),%ymm12        # 5fa0 <_sk_callback_avx+0x360>
+  DB  196,98,125,24,37,103,45,0,0         ; vbroadcastss  0x2d67(%rip),%ymm12        # 5fc0 <_sk_callback_avx+0x35c>
   DB  196,193,116,89,204                  ; vmulps        %ymm12,%ymm1,%ymm1
   DB  196,67,117,74,192,160               ; vblendvps     %ymm10,%ymm8,%ymm1,%ymm8
   DB  197,252,89,200                      ; vmulps        %ymm0,%ymm0,%ymm1
@@ -7737,9 +7750,9 @@ _sk_lab_to_xyz_avx LABEL PROC
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
   DB  196,193,108,89,212                  ; vmulps        %ymm12,%ymm2,%ymm2
   DB  196,227,109,74,208,144              ; vblendvps     %ymm9,%ymm0,%ymm2,%ymm2
-  DB  196,226,125,24,5,33,45,0,0          ; vbroadcastss  0x2d21(%rip),%ymm0        # 5fa4 <_sk_callback_avx+0x364>
+  DB  196,226,125,24,5,29,45,0,0          ; vbroadcastss  0x2d1d(%rip),%ymm0        # 5fc4 <_sk_callback_avx+0x360>
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
-  DB  196,98,125,24,5,24,45,0,0           ; vbroadcastss  0x2d18(%rip),%ymm8        # 5fa8 <_sk_callback_avx+0x368>
+  DB  196,98,125,24,5,20,45,0,0           ; vbroadcastss  0x2d14(%rip),%ymm8        # 5fc8 <_sk_callback_avx+0x364>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7751,14 +7764,14 @@ _sk_load_a8_avx LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,62                              ; jne           32e7 <_sk_load_a8_avx+0x4e>
+  DB  117,62                              ; jne           330b <_sk_load_a8_avx+0x4e>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,121,49,200                  ; vpmovzxbd     %xmm0,%xmm1
   DB  196,227,121,4,192,229               ; vpermilps     $0xe5,%xmm0,%xmm0
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,117,24,192,1                ; vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,220,44,0,0        ; vbroadcastss  0x2cdc(%rip),%ymm1        # 5fac <_sk_callback_avx+0x36c>
+  DB  196,226,125,24,13,216,44,0,0        ; vbroadcastss  0x2cd8(%rip),%ymm1        # 5fcc <_sk_callback_avx+0x368>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -7775,9 +7788,9 @@ _sk_load_a8_avx LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           32ef <_sk_load_a8_avx+0x56>
+  DB  117,234                             ; jne           3313 <_sk_load_a8_avx+0x56>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,161                             ; jmp           32ad <_sk_load_a8_avx+0x14>
+  DB  235,161                             ; jmp           32d1 <_sk_load_a8_avx+0x14>
 
 PUBLIC _sk_gather_a8_avx
 _sk_gather_a8_avx LABEL PROC
@@ -7825,7 +7838,7 @@ _sk_gather_a8_avx LABEL PROC
   DB  196,226,121,49,201                  ; vpmovzxbd     %xmm1,%xmm1
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,209,43,0,0        ; vbroadcastss  0x2bd1(%rip),%ymm1        # 5fb0 <_sk_callback_avx+0x370>
+  DB  196,226,125,24,13,205,43,0,0        ; vbroadcastss  0x2bcd(%rip),%ymm1        # 5fd0 <_sk_callback_avx+0x36c>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -7841,14 +7854,14 @@ PUBLIC _sk_store_a8_avx
 _sk_store_a8_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,172,43,0,0          ; vbroadcastss  0x2bac(%rip),%ymm8        # 5fb4 <_sk_callback_avx+0x374>
+  DB  196,98,125,24,5,168,43,0,0          ; vbroadcastss  0x2ba8(%rip),%ymm8        # 5fd4 <_sk_callback_avx+0x370>
   DB  196,65,100,89,192                   ; vmulps        %ymm8,%ymm3,%ymm8
   DB  196,65,125,91,192                   ; vcvtps2dq     %ymm8,%ymm8
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  196,65,57,103,192                   ; vpackuswb     %xmm8,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           3431 <_sk_store_a8_avx+0x37>
+  DB  117,10                              ; jne           3455 <_sk_store_a8_avx+0x37>
   DB  196,65,123,17,4,58                  ; vmovsd        %xmm8,(%r10,%rdi,1)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7856,10 +7869,10 @@ _sk_store_a8_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            342d <_sk_store_a8_avx+0x33>
+  DB  119,236                             ; ja            3451 <_sk_store_a8_avx+0x33>
   DB  196,66,121,48,192                   ; vpmovzxbw     %xmm8,%xmm8
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 3494 <_sk_store_a8_avx+0x9a>
+  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 34b8 <_sk_store_a8_avx+0x9a>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7870,7 +7883,7 @@ _sk_store_a8_avx LABEL PROC
   DB  196,67,121,20,68,58,2,4             ; vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   DB  196,67,121,20,68,58,1,2             ; vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   DB  196,67,121,20,4,58,0                ; vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  DB  235,154                             ; jmp           342d <_sk_store_a8_avx+0x33>
+  DB  235,154                             ; jmp           3451 <_sk_store_a8_avx+0x33>
   DB  144                                 ; nop
   DB  246,255                             ; idiv          %bh
   DB  255                                 ; (bad)
@@ -7902,17 +7915,17 @@ _sk_load_g8_avx LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,67                              ; jne           3503 <_sk_load_g8_avx+0x53>
+  DB  117,67                              ; jne           3527 <_sk_load_g8_avx+0x53>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,121,49,200                  ; vpmovzxbd     %xmm0,%xmm1
   DB  196,227,121,4,192,229               ; vpermilps     $0xe5,%xmm0,%xmm0
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,117,24,192,1                ; vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,209,42,0,0        ; vbroadcastss  0x2ad1(%rip),%ymm1        # 5fb8 <_sk_callback_avx+0x378>
+  DB  196,226,125,24,13,205,42,0,0        ; vbroadcastss  0x2acd(%rip),%ymm1        # 5fd8 <_sk_callback_avx+0x374>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,198,42,0,0        ; vbroadcastss  0x2ac6(%rip),%ymm3        # 5fbc <_sk_callback_avx+0x37c>
+  DB  196,226,125,24,29,194,42,0,0        ; vbroadcastss  0x2ac2(%rip),%ymm3        # 5fdc <_sk_callback_avx+0x378>
   DB  76,137,193                          ; mov           %r8,%rcx
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
@@ -7926,9 +7939,9 @@ _sk_load_g8_avx LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           350b <_sk_load_g8_avx+0x5b>
+  DB  117,234                             ; jne           352f <_sk_load_g8_avx+0x5b>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,156                             ; jmp           34c4 <_sk_load_g8_avx+0x14>
+  DB  235,156                             ; jmp           34e8 <_sk_load_g8_avx+0x14>
 
 PUBLIC _sk_gather_g8_avx
 _sk_gather_g8_avx LABEL PROC
@@ -7976,10 +7989,10 @@ _sk_gather_g8_avx LABEL PROC
   DB  196,226,121,49,201                  ; vpmovzxbd     %xmm1,%xmm1
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,197,41,0,0        ; vbroadcastss  0x29c5(%rip),%ymm1        # 5fc0 <_sk_callback_avx+0x380>
+  DB  196,226,125,24,13,193,41,0,0        ; vbroadcastss  0x29c1(%rip),%ymm1        # 5fe0 <_sk_callback_avx+0x37c>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,186,41,0,0        ; vbroadcastss  0x29ba(%rip),%ymm3        # 5fc4 <_sk_callback_avx+0x384>
+  DB  196,226,125,24,29,182,41,0,0        ; vbroadcastss  0x29b6(%rip),%ymm3        # 5fe4 <_sk_callback_avx+0x380>
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
   DB  91                                  ; pop           %rbx
@@ -7993,9 +8006,9 @@ _sk_gather_i8_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            362a <_sk_gather_i8_avx+0xf>
+  DB  116,5                               ; je            364e <_sk_gather_i8_avx+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           362c <_sk_gather_i8_avx+0x11>
+  DB  235,2                               ; jmp           3650 <_sk_gather_i8_avx+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,87                               ; push          %r15
   DB  65,86                               ; push          %r14
@@ -8057,10 +8070,10 @@ _sk_gather_i8_avx LABEL PROC
   DB  196,163,121,34,4,163,2              ; vpinsrd       $0x2,(%rbx,%r12,4),%xmm0,%xmm0
   DB  196,163,121,34,28,19,3              ; vpinsrd       $0x3,(%rbx,%r10,1),%xmm0,%xmm3
   DB  196,227,61,24,195,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  DB  197,124,40,21,74,42,0,0             ; vmovaps       0x2a4a(%rip),%ymm10        # 61a0 <_sk_callback_avx+0x560>
+  DB  197,124,40,21,70,42,0,0             ; vmovaps       0x2a46(%rip),%ymm10        # 61c0 <_sk_callback_avx+0x55c>
   DB  196,193,124,84,194                  ; vandps        %ymm10,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,96,40,0,0          ; vbroadcastss  0x2860(%rip),%ymm9        # 5fc8 <_sk_callback_avx+0x388>
+  DB  196,98,125,24,13,92,40,0,0          ; vbroadcastss  0x285c(%rip),%ymm9        # 5fe8 <_sk_callback_avx+0x384>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,113,114,208,8               ; vpsrld        $0x8,%xmm8,%xmm1
   DB  197,233,114,211,8                   ; vpsrld        $0x8,%xmm3,%xmm2
@@ -8092,38 +8105,38 @@ _sk_load_565_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,128,0,0,0                    ; jne           3860 <_sk_load_565_avx+0x8e>
+  DB  15,133,128,0,0,0                    ; jne           3884 <_sk_load_565_avx+0x8e>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  197,241,239,201                     ; vpxor         %xmm1,%xmm1,%xmm1
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,209,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  DB  196,226,125,24,5,202,39,0,0         ; vbroadcastss  0x27ca(%rip),%ymm0        # 5fcc <_sk_callback_avx+0x38c>
+  DB  196,226,125,24,5,198,39,0,0         ; vbroadcastss  0x27c6(%rip),%ymm0        # 5fec <_sk_callback_avx+0x388>
   DB  197,236,84,192                      ; vandps        %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,189,39,0,0        ; vbroadcastss  0x27bd(%rip),%ymm1        # 5fd0 <_sk_callback_avx+0x390>
+  DB  196,226,125,24,13,185,39,0,0        ; vbroadcastss  0x27b9(%rip),%ymm1        # 5ff0 <_sk_callback_avx+0x38c>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,180,39,0,0        ; vbroadcastss  0x27b4(%rip),%ymm1        # 5fd4 <_sk_callback_avx+0x394>
+  DB  196,226,125,24,13,176,39,0,0        ; vbroadcastss  0x27b0(%rip),%ymm1        # 5ff4 <_sk_callback_avx+0x390>
   DB  197,236,84,201                      ; vandps        %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,167,39,0,0        ; vbroadcastss  0x27a7(%rip),%ymm3        # 5fd8 <_sk_callback_avx+0x398>
+  DB  196,226,125,24,29,163,39,0,0        ; vbroadcastss  0x27a3(%rip),%ymm3        # 5ff8 <_sk_callback_avx+0x394>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,24,29,158,39,0,0        ; vbroadcastss  0x279e(%rip),%ymm3        # 5fdc <_sk_callback_avx+0x39c>
+  DB  196,226,125,24,29,154,39,0,0        ; vbroadcastss  0x279a(%rip),%ymm3        # 5ffc <_sk_callback_avx+0x398>
   DB  197,236,84,211                      ; vandps        %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,145,39,0,0        ; vbroadcastss  0x2791(%rip),%ymm3        # 5fe0 <_sk_callback_avx+0x3a0>
+  DB  196,226,125,24,29,141,39,0,0        ; vbroadcastss  0x278d(%rip),%ymm3        # 6000 <_sk_callback_avx+0x39c>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,134,39,0,0        ; vbroadcastss  0x2786(%rip),%ymm3        # 5fe4 <_sk_callback_avx+0x3a4>
+  DB  196,226,125,24,29,130,39,0,0        ; vbroadcastss  0x2782(%rip),%ymm3        # 6004 <_sk_callback_avx+0x3a0>
   DB  255,224                             ; jmpq          *%rax
   DB  65,137,200                          ; mov           %ecx,%r8d
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,110,255,255,255              ; ja            37e6 <_sk_load_565_avx+0x14>
+  DB  15,135,110,255,255,255              ; ja            380a <_sk_load_565_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 38cc <_sk_load_565_avx+0xfa>
+  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 38f0 <_sk_load_565_avx+0xfa>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8135,7 +8148,7 @@ _sk_load_565_avx LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,26,255,255,255                  ; jmpq          37e6 <_sk_load_565_avx+0x14>
+  DB  233,26,255,255,255                  ; jmpq          380a <_sk_load_565_avx+0x14>
   DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -8211,23 +8224,23 @@ _sk_gather_565_avx LABEL PROC
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,209,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  DB  196,226,125,24,5,38,38,0,0          ; vbroadcastss  0x2626(%rip),%ymm0        # 5fe8 <_sk_callback_avx+0x3a8>
+  DB  196,226,125,24,5,34,38,0,0          ; vbroadcastss  0x2622(%rip),%ymm0        # 6008 <_sk_callback_avx+0x3a4>
   DB  197,236,84,192                      ; vandps        %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,25,38,0,0         ; vbroadcastss  0x2619(%rip),%ymm1        # 5fec <_sk_callback_avx+0x3ac>
+  DB  196,226,125,24,13,21,38,0,0         ; vbroadcastss  0x2615(%rip),%ymm1        # 600c <_sk_callback_avx+0x3a8>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,16,38,0,0         ; vbroadcastss  0x2610(%rip),%ymm1        # 5ff0 <_sk_callback_avx+0x3b0>
+  DB  196,226,125,24,13,12,38,0,0         ; vbroadcastss  0x260c(%rip),%ymm1        # 6010 <_sk_callback_avx+0x3ac>
   DB  197,236,84,201                      ; vandps        %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,3,38,0,0          ; vbroadcastss  0x2603(%rip),%ymm3        # 5ff4 <_sk_callback_avx+0x3b4>
+  DB  196,226,125,24,29,255,37,0,0        ; vbroadcastss  0x25ff(%rip),%ymm3        # 6014 <_sk_callback_avx+0x3b0>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,24,29,250,37,0,0        ; vbroadcastss  0x25fa(%rip),%ymm3        # 5ff8 <_sk_callback_avx+0x3b8>
+  DB  196,226,125,24,29,246,37,0,0        ; vbroadcastss  0x25f6(%rip),%ymm3        # 6018 <_sk_callback_avx+0x3b4>
   DB  197,236,84,211                      ; vandps        %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,237,37,0,0        ; vbroadcastss  0x25ed(%rip),%ymm3        # 5ffc <_sk_callback_avx+0x3bc>
+  DB  196,226,125,24,29,233,37,0,0        ; vbroadcastss  0x25e9(%rip),%ymm3        # 601c <_sk_callback_avx+0x3b8>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,226,37,0,0        ; vbroadcastss  0x25e2(%rip),%ymm3        # 6000 <_sk_callback_avx+0x3c0>
+  DB  196,226,125,24,29,222,37,0,0        ; vbroadcastss  0x25de(%rip),%ymm3        # 6020 <_sk_callback_avx+0x3bc>
   DB  91                                  ; pop           %rbx
   DB  65,92                               ; pop           %r12
   DB  65,94                               ; pop           %r14
@@ -8239,14 +8252,14 @@ PUBLIC _sk_store_565_avx
 _sk_store_565_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,206,37,0,0          ; vbroadcastss  0x25ce(%rip),%ymm8        # 6004 <_sk_callback_avx+0x3c4>
+  DB  196,98,125,24,5,202,37,0,0          ; vbroadcastss  0x25ca(%rip),%ymm8        # 6024 <_sk_callback_avx+0x3c0>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,41,114,241,11               ; vpslld        $0xb,%xmm9,%xmm10
   DB  196,67,125,25,201,1                 ; vextractf128  $0x1,%ymm9,%xmm9
   DB  196,193,49,114,241,11               ; vpslld        $0xb,%xmm9,%xmm9
   DB  196,67,45,24,201,1                  ; vinsertf128   $0x1,%xmm9,%ymm10,%ymm9
-  DB  196,98,125,24,21,167,37,0,0         ; vbroadcastss  0x25a7(%rip),%ymm10        # 6008 <_sk_callback_avx+0x3c8>
+  DB  196,98,125,24,21,163,37,0,0         ; vbroadcastss  0x25a3(%rip),%ymm10        # 6028 <_sk_callback_avx+0x3c4>
   DB  196,65,116,89,210                   ; vmulps        %ymm10,%ymm1,%ymm10
   DB  196,65,125,91,210                   ; vcvtps2dq     %ymm10,%ymm10
   DB  196,193,33,114,242,5                ; vpslld        $0x5,%xmm10,%xmm11
@@ -8260,7 +8273,7 @@ _sk_store_565_avx LABEL PROC
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           3ab1 <_sk_store_565_avx+0x89>
+  DB  117,10                              ; jne           3ad5 <_sk_store_565_avx+0x89>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8268,9 +8281,9 @@ _sk_store_565_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            3aad <_sk_store_565_avx+0x85>
+  DB  119,236                             ; ja            3ad1 <_sk_store_565_avx+0x85>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3b10 <_sk_store_565_avx+0xe8>
+  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3b34 <_sk_store_565_avx+0xe8>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8281,7 +8294,7 @@ _sk_store_565_avx LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           3aad <_sk_store_565_avx+0x85>
+  DB  235,159                             ; jmp           3ad1 <_sk_store_565_avx+0x85>
   DB  102,144                             ; xchg          %ax,%ax
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -8312,31 +8325,31 @@ _sk_load_4444_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,152,0,0,0                    ; jne           3bd2 <_sk_load_4444_avx+0xa6>
+  DB  15,133,152,0,0,0                    ; jne           3bf6 <_sk_load_4444_avx+0xa6>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  197,241,239,201                     ; vpxor         %xmm1,%xmm1,%xmm1
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,217,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  DB  196,226,125,24,5,176,36,0,0         ; vbroadcastss  0x24b0(%rip),%ymm0        # 600c <_sk_callback_avx+0x3cc>
+  DB  196,226,125,24,5,172,36,0,0         ; vbroadcastss  0x24ac(%rip),%ymm0        # 602c <_sk_callback_avx+0x3c8>
   DB  197,228,84,192                      ; vandps        %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,163,36,0,0        ; vbroadcastss  0x24a3(%rip),%ymm1        # 6010 <_sk_callback_avx+0x3d0>
+  DB  196,226,125,24,13,159,36,0,0        ; vbroadcastss  0x249f(%rip),%ymm1        # 6030 <_sk_callback_avx+0x3cc>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,154,36,0,0        ; vbroadcastss  0x249a(%rip),%ymm1        # 6014 <_sk_callback_avx+0x3d4>
+  DB  196,226,125,24,13,150,36,0,0        ; vbroadcastss  0x2496(%rip),%ymm1        # 6034 <_sk_callback_avx+0x3d0>
   DB  197,228,84,201                      ; vandps        %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,141,36,0,0        ; vbroadcastss  0x248d(%rip),%ymm2        # 6018 <_sk_callback_avx+0x3d8>
+  DB  196,226,125,24,21,137,36,0,0        ; vbroadcastss  0x2489(%rip),%ymm2        # 6038 <_sk_callback_avx+0x3d4>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,24,21,132,36,0,0        ; vbroadcastss  0x2484(%rip),%ymm2        # 601c <_sk_callback_avx+0x3dc>
+  DB  196,226,125,24,21,128,36,0,0        ; vbroadcastss  0x2480(%rip),%ymm2        # 603c <_sk_callback_avx+0x3d8>
   DB  197,228,84,210                      ; vandps        %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,119,36,0,0          ; vbroadcastss  0x2477(%rip),%ymm8        # 6020 <_sk_callback_avx+0x3e0>
+  DB  196,98,125,24,5,115,36,0,0          ; vbroadcastss  0x2473(%rip),%ymm8        # 6040 <_sk_callback_avx+0x3dc>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,109,36,0,0          ; vbroadcastss  0x246d(%rip),%ymm8        # 6024 <_sk_callback_avx+0x3e4>
+  DB  196,98,125,24,5,105,36,0,0          ; vbroadcastss  0x2469(%rip),%ymm8        # 6044 <_sk_callback_avx+0x3e0>
   DB  196,193,100,84,216                  ; vandps        %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,95,36,0,0           ; vbroadcastss  0x245f(%rip),%ymm8        # 6028 <_sk_callback_avx+0x3e8>
+  DB  196,98,125,24,5,91,36,0,0           ; vbroadcastss  0x245b(%rip),%ymm8        # 6048 <_sk_callback_avx+0x3e4>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8345,9 +8358,9 @@ _sk_load_4444_avx LABEL PROC
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,86,255,255,255               ; ja            3b40 <_sk_load_4444_avx+0x14>
+  DB  15,135,86,255,255,255               ; ja            3b64 <_sk_load_4444_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 3c40 <_sk_load_4444_avx+0x114>
+  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 3c64 <_sk_load_4444_avx+0x114>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8359,7 +8372,7 @@ _sk_load_4444_avx LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,2,255,255,255                   ; jmpq          3b40 <_sk_load_4444_avx+0x14>
+  DB  233,2,255,255,255                   ; jmpq          3b64 <_sk_load_4444_avx+0x14>
   DB  102,144                             ; xchg          %ax,%ax
   DB  242,255                             ; repnz         (bad)
   DB  255                                 ; (bad)
@@ -8436,25 +8449,25 @@ _sk_gather_4444_avx LABEL PROC
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,217,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  DB  196,226,125,24,5,246,34,0,0         ; vbroadcastss  0x22f6(%rip),%ymm0        # 602c <_sk_callback_avx+0x3ec>
+  DB  196,226,125,24,5,242,34,0,0         ; vbroadcastss  0x22f2(%rip),%ymm0        # 604c <_sk_callback_avx+0x3e8>
   DB  197,228,84,192                      ; vandps        %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,233,34,0,0        ; vbroadcastss  0x22e9(%rip),%ymm1        # 6030 <_sk_callback_avx+0x3f0>
+  DB  196,226,125,24,13,229,34,0,0        ; vbroadcastss  0x22e5(%rip),%ymm1        # 6050 <_sk_callback_avx+0x3ec>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,224,34,0,0        ; vbroadcastss  0x22e0(%rip),%ymm1        # 6034 <_sk_callback_avx+0x3f4>
+  DB  196,226,125,24,13,220,34,0,0        ; vbroadcastss  0x22dc(%rip),%ymm1        # 6054 <_sk_callback_avx+0x3f0>
   DB  197,228,84,201                      ; vandps        %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,211,34,0,0        ; vbroadcastss  0x22d3(%rip),%ymm2        # 6038 <_sk_callback_avx+0x3f8>
+  DB  196,226,125,24,21,207,34,0,0        ; vbroadcastss  0x22cf(%rip),%ymm2        # 6058 <_sk_callback_avx+0x3f4>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,24,21,202,34,0,0        ; vbroadcastss  0x22ca(%rip),%ymm2        # 603c <_sk_callback_avx+0x3fc>
+  DB  196,226,125,24,21,198,34,0,0        ; vbroadcastss  0x22c6(%rip),%ymm2        # 605c <_sk_callback_avx+0x3f8>
   DB  197,228,84,210                      ; vandps        %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,189,34,0,0          ; vbroadcastss  0x22bd(%rip),%ymm8        # 6040 <_sk_callback_avx+0x400>
+  DB  196,98,125,24,5,185,34,0,0          ; vbroadcastss  0x22b9(%rip),%ymm8        # 6060 <_sk_callback_avx+0x3fc>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,179,34,0,0          ; vbroadcastss  0x22b3(%rip),%ymm8        # 6044 <_sk_callback_avx+0x404>
+  DB  196,98,125,24,5,175,34,0,0          ; vbroadcastss  0x22af(%rip),%ymm8        # 6064 <_sk_callback_avx+0x400>
   DB  196,193,100,84,216                  ; vandps        %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,165,34,0,0          ; vbroadcastss  0x22a5(%rip),%ymm8        # 6048 <_sk_callback_avx+0x408>
+  DB  196,98,125,24,5,161,34,0,0          ; vbroadcastss  0x22a1(%rip),%ymm8        # 6068 <_sk_callback_avx+0x404>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -8468,7 +8481,7 @@ PUBLIC _sk_store_4444_avx
 _sk_store_4444_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,138,34,0,0          ; vbroadcastss  0x228a(%rip),%ymm8        # 604c <_sk_callback_avx+0x40c>
+  DB  196,98,125,24,5,134,34,0,0          ; vbroadcastss  0x2286(%rip),%ymm8        # 606c <_sk_callback_avx+0x408>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,41,114,241,12               ; vpslld        $0xc,%xmm9,%xmm10
@@ -8495,7 +8508,7 @@ _sk_store_4444_avx LABEL PROC
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           3e5b <_sk_store_4444_avx+0xa7>
+  DB  117,10                              ; jne           3e7f <_sk_store_4444_avx+0xa7>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8503,9 +8516,9 @@ _sk_store_4444_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            3e57 <_sk_store_4444_avx+0xa3>
+  DB  119,236                             ; ja            3e7b <_sk_store_4444_avx+0xa3>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,66,0,0,0                  ; lea           0x42(%rip),%r9        # 3eb8 <_sk_store_4444_avx+0x104>
+  DB  76,141,13,66,0,0,0                  ; lea           0x42(%rip),%r9        # 3edc <_sk_store_4444_avx+0x104>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8516,7 +8529,7 @@ _sk_store_4444_avx LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           3e57 <_sk_store_4444_avx+0xa3>
+  DB  235,159                             ; jmp           3e7b <_sk_store_4444_avx+0xa3>
   DB  247,255                             ; idiv          %edi
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -8545,12 +8558,12 @@ _sk_load_8888_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,135,0,0,0                    ; jne           3f69 <_sk_load_8888_avx+0x95>
+  DB  15,133,135,0,0,0                    ; jne           3f8d <_sk_load_8888_avx+0x95>
   DB  196,65,124,16,12,186                ; vmovups       (%r10,%rdi,4),%ymm9
-  DB  197,124,40,21,208,34,0,0            ; vmovaps       0x22d0(%rip),%ymm10        # 61c0 <_sk_callback_avx+0x580>
+  DB  197,124,40,21,204,34,0,0            ; vmovaps       0x22cc(%rip),%ymm10        # 61e0 <_sk_callback_avx+0x57c>
   DB  196,193,52,84,194                   ; vandps        %ymm10,%ymm9,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,78,33,0,0           ; vbroadcastss  0x214e(%rip),%ymm8        # 6050 <_sk_callback_avx+0x410>
+  DB  196,98,125,24,5,74,33,0,0           ; vbroadcastss  0x214a(%rip),%ymm8        # 6070 <_sk_callback_avx+0x40c>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  196,193,113,114,209,8               ; vpsrld        $0x8,%xmm9,%xmm1
   DB  196,99,125,25,203,1                 ; vextractf128  $0x1,%ymm9,%xmm3
@@ -8577,9 +8590,9 @@ _sk_load_8888_avx LABEL PROC
   DB  196,65,52,87,201                    ; vxorps        %ymm9,%ymm9,%ymm9
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,102,255,255,255              ; ja            3ee8 <_sk_load_8888_avx+0x14>
+  DB  15,135,102,255,255,255              ; ja            3f0c <_sk_load_8888_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,139,0,0,0                 ; lea           0x8b(%rip),%r9        # 4018 <_sk_load_8888_avx+0x144>
+  DB  76,141,13,139,0,0,0                 ; lea           0x8b(%rip),%r9        # 403c <_sk_load_8888_avx+0x144>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8602,7 +8615,7 @@ _sk_load_8888_avx LABEL PROC
   DB  196,99,53,12,200,15                 ; vblendps      $0xf,%ymm0,%ymm9,%ymm9
   DB  196,195,49,34,4,186,0               ; vpinsrd       $0x0,(%r10,%rdi,4),%xmm9,%xmm0
   DB  196,99,53,12,200,15                 ; vblendps      $0xf,%ymm0,%ymm9,%ymm9
-  DB  233,210,254,255,255                 ; jmpq          3ee8 <_sk_load_8888_avx+0x14>
+  DB  233,210,254,255,255                 ; jmpq          3f0c <_sk_load_8888_avx+0x14>
   DB  102,144                             ; xchg          %ax,%ax
   DB  236                                 ; in            (%dx),%al
   DB  255                                 ; (bad)
@@ -8620,7 +8633,7 @@ _sk_load_8888_avx LABEL PROC
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  126,255                             ; jle           4031 <_sk_load_8888_avx+0x15d>
+  DB  126,255                             ; jle           4055 <_sk_load_8888_avx+0x15d>
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
 
@@ -8663,10 +8676,10 @@ _sk_gather_8888_avx LABEL PROC
   DB  196,131,121,34,4,152,2              ; vpinsrd       $0x2,(%r8,%r11,4),%xmm0,%xmm0
   DB  196,131,121,34,28,144,3             ; vpinsrd       $0x3,(%r8,%r10,4),%xmm0,%xmm3
   DB  196,227,61,24,195,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  DB  197,124,40,21,250,32,0,0            ; vmovaps       0x20fa(%rip),%ymm10        # 61e0 <_sk_callback_avx+0x5a0>
+  DB  197,124,40,21,246,32,0,0            ; vmovaps       0x20f6(%rip),%ymm10        # 6200 <_sk_callback_avx+0x59c>
   DB  196,193,124,84,194                  ; vandps        %ymm10,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,92,31,0,0          ; vbroadcastss  0x1f5c(%rip),%ymm9        # 6054 <_sk_callback_avx+0x414>
+  DB  196,98,125,24,13,88,31,0,0          ; vbroadcastss  0x1f58(%rip),%ymm9        # 6074 <_sk_callback_avx+0x410>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,113,114,208,8               ; vpsrld        $0x8,%xmm8,%xmm1
   DB  197,233,114,211,8                   ; vpsrld        $0x8,%xmm3,%xmm2
@@ -8696,7 +8709,7 @@ PUBLIC _sk_store_8888_avx
 _sk_store_8888_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,234,30,0,0          ; vbroadcastss  0x1eea(%rip),%ymm8        # 6058 <_sk_callback_avx+0x418>
+  DB  196,98,125,24,5,230,30,0,0          ; vbroadcastss  0x1ee6(%rip),%ymm8        # 6078 <_sk_callback_avx+0x414>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,65,116,89,208                   ; vmulps        %ymm8,%ymm1,%ymm10
@@ -8721,7 +8734,7 @@ _sk_store_8888_avx LABEL PROC
   DB  196,65,45,86,192                    ; vorpd         %ymm8,%ymm10,%ymm8
   DB  196,65,53,86,192                    ; vorpd         %ymm8,%ymm9,%ymm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           41fc <_sk_store_8888_avx+0x9c>
+  DB  117,10                              ; jne           4220 <_sk_store_8888_avx+0x9c>
   DB  196,65,124,17,4,186                 ; vmovups       %ymm8,(%r10,%rdi,4)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8729,9 +8742,9 @@ _sk_store_8888_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            41f8 <_sk_store_8888_avx+0x98>
+  DB  119,236                             ; ja            421c <_sk_store_8888_avx+0x98>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,85,0,0,0                  ; lea           0x55(%rip),%r9        # 426c <_sk_store_8888_avx+0x10c>
+  DB  76,141,13,85,0,0,0                  ; lea           0x55(%rip),%r9        # 4290 <_sk_store_8888_avx+0x10c>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8745,7 +8758,7 @@ _sk_store_8888_avx LABEL PROC
   DB  196,67,121,22,68,186,8,2            ; vpextrd       $0x2,%xmm8,0x8(%r10,%rdi,4)
   DB  196,67,121,22,68,186,4,1            ; vpextrd       $0x1,%xmm8,0x4(%r10,%rdi,4)
   DB  196,65,121,126,4,186                ; vmovd         %xmm8,(%r10,%rdi,4)
-  DB  235,143                             ; jmp           41f8 <_sk_store_8888_avx+0x98>
+  DB  235,143                             ; jmp           421c <_sk_store_8888_avx+0x98>
   DB  15,31,0                             ; nopl          (%rax)
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -8781,7 +8794,7 @@ _sk_load_f16_avx LABEL PROC
   DB  197,252,17,116,36,64                ; vmovups       %ymm6,0x40(%rsp)
   DB  197,252,17,108,36,32                ; vmovups       %ymm5,0x20(%rsp)
   DB  197,254,127,36,36                   ; vmovdqu       %ymm4,(%rsp)
-  DB  15,133,143,2,0,0                    ; jne           4543 <_sk_load_f16_avx+0x2bb>
+  DB  15,133,143,2,0,0                    ; jne           4567 <_sk_load_f16_avx+0x2bb>
   DB  197,121,16,4,248                    ; vmovupd       (%rax,%rdi,8),%xmm8
   DB  197,249,16,84,248,16                ; vmovupd       0x10(%rax,%rdi,8),%xmm2
   DB  197,249,16,76,248,32                ; vmovupd       0x20(%rax,%rdi,8),%xmm1
@@ -8799,13 +8812,13 @@ _sk_load_f16_avx LABEL PROC
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
-  DB  196,98,125,24,37,79,29,0,0          ; vbroadcastss  0x1d4f(%rip),%ymm12        # 605c <_sk_callback_avx+0x41c>
+  DB  196,98,125,24,37,75,29,0,0          ; vbroadcastss  0x1d4b(%rip),%ymm12        # 607c <_sk_callback_avx+0x418>
   DB  196,193,124,84,204                  ; vandps        %ymm12,%ymm0,%ymm1
   DB  197,252,87,193                      ; vxorps        %ymm1,%ymm0,%ymm0
   DB  196,195,125,25,198,1                ; vextractf128  $0x1,%ymm0,%xmm14
-  DB  196,98,121,24,29,59,29,0,0          ; vbroadcastss  0x1d3b(%rip),%xmm11        # 6060 <_sk_callback_avx+0x420>
+  DB  196,98,121,24,29,55,29,0,0          ; vbroadcastss  0x1d37(%rip),%xmm11        # 6080 <_sk_callback_avx+0x41c>
   DB  196,193,8,87,219                    ; vxorps        %xmm11,%xmm14,%xmm3
-  DB  196,98,121,24,45,49,29,0,0          ; vbroadcastss  0x1d31(%rip),%xmm13        # 6064 <_sk_callback_avx+0x424>
+  DB  196,98,121,24,45,45,29,0,0          ; vbroadcastss  0x1d2d(%rip),%xmm13        # 6084 <_sk_callback_avx+0x420>
   DB  197,145,102,219                     ; vpcmpgtd      %xmm3,%xmm13,%xmm3
   DB  196,65,120,87,211                   ; vxorps        %xmm11,%xmm0,%xmm10
   DB  196,65,17,102,210                   ; vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -8819,7 +8832,7 @@ _sk_load_f16_avx LABEL PROC
   DB  196,227,125,24,195,1                ; vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   DB  197,252,86,193                      ; vorps         %ymm1,%ymm0,%ymm0
   DB  196,227,125,25,193,1                ; vextractf128  $0x1,%ymm0,%xmm1
-  DB  196,226,121,24,29,231,28,0,0        ; vbroadcastss  0x1ce7(%rip),%xmm3        # 6068 <_sk_callback_avx+0x428>
+  DB  196,226,121,24,29,227,28,0,0        ; vbroadcastss  0x1ce3(%rip),%xmm3        # 6088 <_sk_callback_avx+0x424>
   DB  197,241,254,203                     ; vpaddd        %xmm3,%xmm1,%xmm1
   DB  197,249,254,195                     ; vpaddd        %xmm3,%xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
@@ -8912,29 +8925,29 @@ _sk_load_f16_avx LABEL PROC
   DB  197,123,16,4,248                    ; vmovsd        (%rax,%rdi,8),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,79                              ; je            45a2 <_sk_load_f16_avx+0x31a>
+  DB  116,79                              ; je            45c6 <_sk_load_f16_avx+0x31a>
   DB  197,57,22,68,248,8                  ; vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,67                              ; jb            45a2 <_sk_load_f16_avx+0x31a>
+  DB  114,67                              ; jb            45c6 <_sk_load_f16_avx+0x31a>
   DB  197,251,16,84,248,16                ; vmovsd        0x10(%rax,%rdi,8),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,68                              ; je            45af <_sk_load_f16_avx+0x327>
+  DB  116,68                              ; je            45d3 <_sk_load_f16_avx+0x327>
   DB  197,233,22,84,248,24                ; vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,56                              ; jb            45af <_sk_load_f16_avx+0x327>
+  DB  114,56                              ; jb            45d3 <_sk_load_f16_avx+0x327>
   DB  197,251,16,76,248,32                ; vmovsd        0x20(%rax,%rdi,8),%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,68,253,255,255               ; je            42cb <_sk_load_f16_avx+0x43>
+  DB  15,132,68,253,255,255               ; je            42ef <_sk_load_f16_avx+0x43>
   DB  197,241,22,76,248,40                ; vmovhpd       0x28(%rax,%rdi,8),%xmm1,%xmm1
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,52,253,255,255               ; jb            42cb <_sk_load_f16_avx+0x43>
+  DB  15,130,52,253,255,255               ; jb            42ef <_sk_load_f16_avx+0x43>
   DB  197,122,126,76,248,48               ; vmovq         0x30(%rax,%rdi,8),%xmm9
-  DB  233,41,253,255,255                  ; jmpq          42cb <_sk_load_f16_avx+0x43>
+  DB  233,41,253,255,255                  ; jmpq          42ef <_sk_load_f16_avx+0x43>
   DB  197,241,87,201                      ; vxorpd        %xmm1,%xmm1,%xmm1
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,28,253,255,255                  ; jmpq          42cb <_sk_load_f16_avx+0x43>
+  DB  233,28,253,255,255                  ; jmpq          42ef <_sk_load_f16_avx+0x43>
   DB  197,241,87,201                      ; vxorpd        %xmm1,%xmm1,%xmm1
-  DB  233,19,253,255,255                  ; jmpq          42cb <_sk_load_f16_avx+0x43>
+  DB  233,19,253,255,255                  ; jmpq          42ef <_sk_load_f16_avx+0x43>
 
 PUBLIC _sk_gather_f16_avx
 _sk_gather_f16_avx LABEL PROC
@@ -8996,13 +9009,13 @@ _sk_gather_f16_avx LABEL PROC
   DB  197,249,105,210                     ; vpunpckhwd    %xmm2,%xmm0,%xmm2
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,194,1                ; vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
-  DB  196,98,125,24,37,167,25,0,0         ; vbroadcastss  0x19a7(%rip),%ymm12        # 606c <_sk_callback_avx+0x42c>
+  DB  196,98,125,24,37,163,25,0,0         ; vbroadcastss  0x19a3(%rip),%ymm12        # 608c <_sk_callback_avx+0x428>
   DB  196,193,124,84,212                  ; vandps        %ymm12,%ymm0,%ymm2
   DB  197,252,87,194                      ; vxorps        %ymm2,%ymm0,%ymm0
   DB  196,195,125,25,198,1                ; vextractf128  $0x1,%ymm0,%xmm14
-  DB  196,98,121,24,29,147,25,0,0         ; vbroadcastss  0x1993(%rip),%xmm11        # 6070 <_sk_callback_avx+0x430>
+  DB  196,98,121,24,29,143,25,0,0         ; vbroadcastss  0x198f(%rip),%xmm11        # 6090 <_sk_callback_avx+0x42c>
   DB  196,193,8,87,219                    ; vxorps        %xmm11,%xmm14,%xmm3
-  DB  196,98,121,24,45,137,25,0,0         ; vbroadcastss  0x1989(%rip),%xmm13        # 6074 <_sk_callback_avx+0x434>
+  DB  196,98,121,24,45,133,25,0,0         ; vbroadcastss  0x1985(%rip),%xmm13        # 6094 <_sk_callback_avx+0x430>
   DB  197,145,102,219                     ; vpcmpgtd      %xmm3,%xmm13,%xmm3
   DB  196,65,120,87,211                   ; vxorps        %xmm11,%xmm0,%xmm10
   DB  196,65,17,102,210                   ; vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -9016,7 +9029,7 @@ _sk_gather_f16_avx LABEL PROC
   DB  196,227,125,24,195,1                ; vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   DB  197,252,86,194                      ; vorps         %ymm2,%ymm0,%ymm0
   DB  196,227,125,25,194,1                ; vextractf128  $0x1,%ymm0,%xmm2
-  DB  196,226,121,24,29,63,25,0,0         ; vbroadcastss  0x193f(%rip),%xmm3        # 6078 <_sk_callback_avx+0x438>
+  DB  196,226,121,24,29,59,25,0,0         ; vbroadcastss  0x193b(%rip),%xmm3        # 6098 <_sk_callback_avx+0x434>
   DB  197,233,254,211                     ; vpaddd        %xmm3,%xmm2,%xmm2
   DB  197,249,254,195                     ; vpaddd        %xmm3,%xmm0,%xmm0
   DB  196,227,125,24,194,1                ; vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
@@ -9118,12 +9131,12 @@ _sk_store_f16_avx LABEL PROC
   DB  197,252,17,180,36,128,0,0,0         ; vmovups       %ymm6,0x80(%rsp)
   DB  197,252,17,108,36,96                ; vmovups       %ymm5,0x60(%rsp)
   DB  197,252,17,100,36,64                ; vmovups       %ymm4,0x40(%rsp)
-  DB  196,98,125,24,13,76,23,0,0          ; vbroadcastss  0x174c(%rip),%ymm9        # 607c <_sk_callback_avx+0x43c>
+  DB  196,98,125,24,13,72,23,0,0          ; vbroadcastss  0x1748(%rip),%ymm9        # 609c <_sk_callback_avx+0x438>
   DB  196,65,124,84,209                   ; vandps        %ymm9,%ymm0,%ymm10
   DB  197,252,17,4,36                     ; vmovups       %ymm0,(%rsp)
   DB  196,65,124,87,218                   ; vxorps        %ymm10,%ymm0,%ymm11
   DB  196,67,125,25,220,1                 ; vextractf128  $0x1,%ymm11,%xmm12
-  DB  196,98,121,24,5,50,23,0,0           ; vbroadcastss  0x1732(%rip),%xmm8        # 6080 <_sk_callback_avx+0x440>
+  DB  196,98,121,24,5,46,23,0,0           ; vbroadcastss  0x172e(%rip),%xmm8        # 60a0 <_sk_callback_avx+0x43c>
   DB  196,65,57,102,236                   ; vpcmpgtd      %xmm12,%xmm8,%xmm13
   DB  196,65,57,102,243                   ; vpcmpgtd      %xmm11,%xmm8,%xmm14
   DB  196,67,13,24,237,1                  ; vinsertf128   $0x1,%xmm13,%ymm14,%ymm13
@@ -9133,7 +9146,7 @@ _sk_store_f16_avx LABEL PROC
   DB  196,67,13,24,242,1                  ; vinsertf128   $0x1,%xmm10,%ymm14,%ymm14
   DB  196,193,33,114,211,13               ; vpsrld        $0xd,%xmm11,%xmm11
   DB  196,193,25,114,212,13               ; vpsrld        $0xd,%xmm12,%xmm12
-  DB  196,98,125,24,21,249,22,0,0         ; vbroadcastss  0x16f9(%rip),%ymm10        # 6084 <_sk_callback_avx+0x444>
+  DB  196,98,125,24,21,245,22,0,0         ; vbroadcastss  0x16f5(%rip),%ymm10        # 60a4 <_sk_callback_avx+0x440>
   DB  196,65,12,86,242                    ; vorps         %ymm10,%ymm14,%ymm14
   DB  196,67,125,25,247,1                 ; vextractf128  $0x1,%ymm14,%xmm15
   DB  196,65,1,254,228                    ; vpaddd        %xmm12,%xmm15,%xmm12
@@ -9215,7 +9228,7 @@ _sk_store_f16_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,75                              ; jne           4b72 <_sk_store_f16_avx+0x270>
+  DB  117,75                              ; jne           4b96 <_sk_store_f16_avx+0x270>
   DB  197,120,17,28,248                   ; vmovups       %xmm11,(%rax,%rdi,8)
   DB  197,120,17,84,248,16                ; vmovups       %xmm10,0x10(%rax,%rdi,8)
   DB  197,120,17,76,248,32                ; vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -9231,22 +9244,22 @@ _sk_store_f16_avx LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  197,121,214,28,248                  ; vmovq         %xmm11,(%rax,%rdi,8)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,193                             ; je            4b3e <_sk_store_f16_avx+0x23c>
+  DB  116,193                             ; je            4b62 <_sk_store_f16_avx+0x23c>
   DB  197,121,23,92,248,8                 ; vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,181                             ; jb            4b3e <_sk_store_f16_avx+0x23c>
+  DB  114,181                             ; jb            4b62 <_sk_store_f16_avx+0x23c>
   DB  197,121,214,84,248,16               ; vmovq         %xmm10,0x10(%rax,%rdi,8)
-  DB  116,173                             ; je            4b3e <_sk_store_f16_avx+0x23c>
+  DB  116,173                             ; je            4b62 <_sk_store_f16_avx+0x23c>
   DB  197,121,23,84,248,24                ; vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,161                             ; jb            4b3e <_sk_store_f16_avx+0x23c>
+  DB  114,161                             ; jb            4b62 <_sk_store_f16_avx+0x23c>
   DB  197,121,214,76,248,32               ; vmovq         %xmm9,0x20(%rax,%rdi,8)
-  DB  116,153                             ; je            4b3e <_sk_store_f16_avx+0x23c>
+  DB  116,153                             ; je            4b62 <_sk_store_f16_avx+0x23c>
   DB  197,121,23,76,248,40                ; vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,141                             ; jb            4b3e <_sk_store_f16_avx+0x23c>
+  DB  114,141                             ; jb            4b62 <_sk_store_f16_avx+0x23c>
   DB  197,121,214,68,248,48               ; vmovq         %xmm8,0x30(%rax,%rdi,8)
-  DB  235,133                             ; jmp           4b3e <_sk_store_f16_avx+0x23c>
+  DB  235,133                             ; jmp           4b62 <_sk_store_f16_avx+0x23c>
 
 PUBLIC _sk_load_u16_be_avx
 _sk_load_u16_be_avx LABEL PROC
@@ -9254,7 +9267,7 @@ _sk_load_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,253,0,0,0                    ; jne           4ccc <_sk_load_u16_be_avx+0x113>
+  DB  15,133,253,0,0,0                    ; jne           4cf0 <_sk_load_u16_be_avx+0x113>
   DB  196,65,121,16,4,64                  ; vmovupd       (%r8,%rax,2),%xmm8
   DB  196,193,121,16,84,64,16             ; vmovupd       0x10(%r8,%rax,2),%xmm2
   DB  196,193,121,16,92,64,32             ; vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -9276,7 +9289,7 @@ _sk_load_u16_be_avx LABEL PROC
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,29,72,20,0,0          ; vbroadcastss  0x1448(%rip),%ymm11        # 6088 <_sk_callback_avx+0x448>
+  DB  196,98,125,24,29,68,20,0,0          ; vbroadcastss  0x1444(%rip),%ymm11        # 60a8 <_sk_callback_avx+0x444>
   DB  196,193,124,89,195                  ; vmulps        %ymm11,%ymm0,%ymm0
   DB  197,177,109,202                     ; vpunpckhqdq   %xmm2,%xmm9,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -9310,29 +9323,29 @@ _sk_load_u16_be_avx LABEL PROC
   DB  196,65,123,16,4,64                  ; vmovsd        (%r8,%rax,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            4d32 <_sk_load_u16_be_avx+0x179>
+  DB  116,85                              ; je            4d56 <_sk_load_u16_be_avx+0x179>
   DB  196,65,57,22,68,64,8                ; vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            4d32 <_sk_load_u16_be_avx+0x179>
+  DB  114,72                              ; jb            4d56 <_sk_load_u16_be_avx+0x179>
   DB  196,193,123,16,84,64,16             ; vmovsd        0x10(%r8,%rax,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            4d3f <_sk_load_u16_be_avx+0x186>
+  DB  116,72                              ; je            4d63 <_sk_load_u16_be_avx+0x186>
   DB  196,193,105,22,84,64,24             ; vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            4d3f <_sk_load_u16_be_avx+0x186>
+  DB  114,59                              ; jb            4d63 <_sk_load_u16_be_avx+0x186>
   DB  196,193,123,16,92,64,32             ; vmovsd        0x20(%r8,%rax,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,213,254,255,255              ; je            4bea <_sk_load_u16_be_avx+0x31>
+  DB  15,132,213,254,255,255              ; je            4c0e <_sk_load_u16_be_avx+0x31>
   DB  196,193,97,22,92,64,40              ; vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,196,254,255,255              ; jb            4bea <_sk_load_u16_be_avx+0x31>
+  DB  15,130,196,254,255,255              ; jb            4c0e <_sk_load_u16_be_avx+0x31>
   DB  196,65,122,126,76,64,48             ; vmovq         0x30(%r8,%rax,2),%xmm9
-  DB  233,184,254,255,255                 ; jmpq          4bea <_sk_load_u16_be_avx+0x31>
+  DB  233,184,254,255,255                 ; jmpq          4c0e <_sk_load_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,171,254,255,255                 ; jmpq          4bea <_sk_load_u16_be_avx+0x31>
+  DB  233,171,254,255,255                 ; jmpq          4c0e <_sk_load_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,162,254,255,255                 ; jmpq          4bea <_sk_load_u16_be_avx+0x31>
+  DB  233,162,254,255,255                 ; jmpq          4c0e <_sk_load_u16_be_avx+0x31>
 
 PUBLIC _sk_load_rgb_u16_be_avx
 _sk_load_rgb_u16_be_avx LABEL PROC
@@ -9340,7 +9353,7 @@ _sk_load_rgb_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,127                        ; lea           (%rdi,%rdi,2),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,243,0,0,0                    ; jne           4e4d <_sk_load_rgb_u16_be_avx+0x105>
+  DB  15,133,243,0,0,0                    ; jne           4e71 <_sk_load_rgb_u16_be_avx+0x105>
   DB  196,193,122,111,4,64                ; vmovdqu       (%r8,%rax,2),%xmm0
   DB  196,193,122,111,84,64,12            ; vmovdqu       0xc(%r8,%rax,2),%xmm2
   DB  196,193,122,111,76,64,24            ; vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -9367,7 +9380,7 @@ _sk_load_rgb_u16_be_avx LABEL PROC
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,29,168,18,0,0         ; vbroadcastss  0x12a8(%rip),%ymm11        # 608c <_sk_callback_avx+0x44c>
+  DB  196,98,125,24,29,164,18,0,0         ; vbroadcastss  0x12a4(%rip),%ymm11        # 60ac <_sk_callback_avx+0x448>
   DB  196,193,124,89,195                  ; vmulps        %ymm11,%ymm0,%ymm0
   DB  197,185,109,202                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -9388,48 +9401,48 @@ _sk_load_rgb_u16_be_avx LABEL PROC
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,211                  ; vmulps        %ymm11,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,69,18,0,0         ; vbroadcastss  0x1245(%rip),%ymm3        # 6090 <_sk_callback_avx+0x450>
+  DB  196,226,125,24,29,65,18,0,0         ; vbroadcastss  0x1241(%rip),%ymm3        # 60b0 <_sk_callback_avx+0x44c>
   DB  255,224                             ; jmpq          *%rax
   DB  196,193,121,110,4,64                ; vmovd         (%r8,%rax,2),%xmm0
   DB  196,193,121,196,68,64,4,2           ; vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           4e66 <_sk_load_rgb_u16_be_avx+0x11e>
-  DB  233,40,255,255,255                  ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
+  DB  117,5                               ; jne           4e8a <_sk_load_rgb_u16_be_avx+0x11e>
+  DB  233,40,255,255,255                  ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
   DB  196,193,121,110,76,64,6             ; vmovd         0x6(%r8,%rax,2),%xmm1
   DB  196,65,113,196,68,64,10,2           ; vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            4e95 <_sk_load_rgb_u16_be_avx+0x14d>
+  DB  114,26                              ; jb            4eb9 <_sk_load_rgb_u16_be_avx+0x14d>
   DB  196,193,121,110,76,64,12            ; vmovd         0xc(%r8,%rax,2),%xmm1
   DB  196,193,113,196,84,64,16,2          ; vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           4e9a <_sk_load_rgb_u16_be_avx+0x152>
-  DB  233,249,254,255,255                 ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
-  DB  233,244,254,255,255                 ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           4ebe <_sk_load_rgb_u16_be_avx+0x152>
+  DB  233,249,254,255,255                 ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,244,254,255,255                 ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
   DB  196,193,121,110,76,64,18            ; vmovd         0x12(%r8,%rax,2),%xmm1
   DB  196,65,113,196,76,64,22,2           ; vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            4ec9 <_sk_load_rgb_u16_be_avx+0x181>
+  DB  114,26                              ; jb            4eed <_sk_load_rgb_u16_be_avx+0x181>
   DB  196,193,121,110,76,64,24            ; vmovd         0x18(%r8,%rax,2),%xmm1
   DB  196,193,113,196,76,64,28,2          ; vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           4ece <_sk_load_rgb_u16_be_avx+0x186>
-  DB  233,197,254,255,255                 ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
-  DB  233,192,254,255,255                 ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           4ef2 <_sk_load_rgb_u16_be_avx+0x186>
+  DB  233,197,254,255,255                 ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,192,254,255,255                 ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
   DB  196,193,121,110,92,64,30            ; vmovd         0x1e(%r8,%rax,2),%xmm3
   DB  196,65,97,196,92,64,34,2            ; vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            4ef7 <_sk_load_rgb_u16_be_avx+0x1af>
+  DB  114,20                              ; jb            4f1b <_sk_load_rgb_u16_be_avx+0x1af>
   DB  196,193,121,110,92,64,36            ; vmovd         0x24(%r8,%rax,2),%xmm3
   DB  196,193,97,196,92,64,40,2           ; vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  DB  233,151,254,255,255                 ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
-  DB  233,146,254,255,255                 ; jmpq          4d8e <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,151,254,255,255                 ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,146,254,255,255                 ; jmpq          4db2 <_sk_load_rgb_u16_be_avx+0x46>
 
 PUBLIC _sk_store_u16_be_avx
 _sk_store_u16_be_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
-  DB  196,98,125,24,5,130,17,0,0          ; vbroadcastss  0x1182(%rip),%ymm8        # 6094 <_sk_callback_avx+0x454>
+  DB  196,98,125,24,5,126,17,0,0          ; vbroadcastss  0x117e(%rip),%ymm8        # 60b4 <_sk_callback_avx+0x450>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,67,125,25,202,1                 ; vextractf128  $0x1,%ymm9,%xmm10
@@ -9467,7 +9480,7 @@ _sk_store_u16_be_avx LABEL PROC
   DB  196,65,17,98,200                    ; vpunpckldq    %xmm8,%xmm13,%xmm9
   DB  196,65,17,106,192                   ; vpunpckhdq    %xmm8,%xmm13,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,31                              ; jne           4ff6 <_sk_store_u16_be_avx+0xfa>
+  DB  117,31                              ; jne           501a <_sk_store_u16_be_avx+0xfa>
   DB  196,65,120,17,28,64                 ; vmovups       %xmm11,(%r8,%rax,2)
   DB  196,65,120,17,84,64,16              ; vmovups       %xmm10,0x10(%r8,%rax,2)
   DB  196,65,120,17,76,64,32              ; vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -9476,31 +9489,31 @@ _sk_store_u16_be_avx LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,214,28,64                ; vmovq         %xmm11,(%r8,%rax,2)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  116,240                             ; je            5016 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,23,92,64,8               ; vmovhpd       %xmm11,0x8(%r8,%rax,2)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  114,227                             ; jb            5016 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,214,84,64,16             ; vmovq         %xmm10,0x10(%r8,%rax,2)
-  DB  116,218                             ; je            4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  116,218                             ; je            5016 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,23,84,64,24              ; vmovhpd       %xmm10,0x18(%r8,%rax,2)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  114,205                             ; jb            5016 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,214,76,64,32             ; vmovq         %xmm9,0x20(%r8,%rax,2)
-  DB  116,196                             ; je            4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  116,196                             ; je            5016 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,23,76,64,40              ; vmovhpd       %xmm9,0x28(%r8,%rax,2)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,183                             ; jb            4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  114,183                             ; jb            5016 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,214,68,64,48             ; vmovq         %xmm8,0x30(%r8,%rax,2)
-  DB  235,174                             ; jmp           4ff2 <_sk_store_u16_be_avx+0xf6>
+  DB  235,174                             ; jmp           5016 <_sk_store_u16_be_avx+0xf6>
 
 PUBLIC _sk_load_f32_avx
 _sk_load_f32_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  119,110                             ; ja            50ba <_sk_load_f32_avx+0x76>
+  DB  119,110                             ; ja            50de <_sk_load_f32_avx+0x76>
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
-  DB  76,141,21,134,0,0,0                 ; lea           0x86(%rip),%r10        # 50e4 <_sk_load_f32_avx+0xa0>
+  DB  76,141,21,134,0,0,0                 ; lea           0x86(%rip),%r10        # 5108 <_sk_load_f32_avx+0xa0>
   DB  73,99,4,138                         ; movslq        (%r10,%rcx,4),%rax
   DB  76,1,208                            ; add           %r10,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -9557,7 +9570,7 @@ _sk_store_f32_avx LABEL PROC
   DB  196,65,37,20,196                    ; vunpcklpd     %ymm12,%ymm11,%ymm8
   DB  196,65,37,21,220                    ; vunpckhpd     %ymm12,%ymm11,%ymm11
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,55                              ; jne           5171 <_sk_store_f32_avx+0x6d>
+  DB  117,55                              ; jne           5195 <_sk_store_f32_avx+0x6d>
   DB  196,67,45,24,225,1                  ; vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   DB  196,67,61,24,235,1                  ; vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   DB  196,67,45,6,201,49                  ; vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -9570,22 +9583,22 @@ _sk_store_f32_avx LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,17,20,128                ; vmovupd       %xmm10,(%r8,%rax,4)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            516d <_sk_store_f32_avx+0x69>
+  DB  116,240                             ; je            5191 <_sk_store_f32_avx+0x69>
   DB  196,65,121,17,76,128,16             ; vmovupd       %xmm9,0x10(%r8,%rax,4)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            516d <_sk_store_f32_avx+0x69>
+  DB  114,227                             ; jb            5191 <_sk_store_f32_avx+0x69>
   DB  196,65,121,17,68,128,32             ; vmovupd       %xmm8,0x20(%r8,%rax,4)
-  DB  116,218                             ; je            516d <_sk_store_f32_avx+0x69>
+  DB  116,218                             ; je            5191 <_sk_store_f32_avx+0x69>
   DB  196,65,121,17,92,128,48             ; vmovupd       %xmm11,0x30(%r8,%rax,4)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            516d <_sk_store_f32_avx+0x69>
+  DB  114,205                             ; jb            5191 <_sk_store_f32_avx+0x69>
   DB  196,67,125,25,84,128,64,1           ; vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  DB  116,195                             ; je            516d <_sk_store_f32_avx+0x69>
+  DB  116,195                             ; je            5191 <_sk_store_f32_avx+0x69>
   DB  196,67,125,25,76,128,80,1           ; vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,181                             ; jb            516d <_sk_store_f32_avx+0x69>
+  DB  114,181                             ; jb            5191 <_sk_store_f32_avx+0x69>
   DB  196,67,125,25,68,128,96,1           ; vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  DB  235,171                             ; jmp           516d <_sk_store_f32_avx+0x69>
+  DB  235,171                             ; jmp           5191 <_sk_store_f32_avx+0x69>
 
 PUBLIC _sk_clamp_x_avx
 _sk_clamp_x_avx LABEL PROC
@@ -9707,12 +9720,12 @@ _sk_mirror_y_avx LABEL PROC
 
 PUBLIC _sk_luminance_to_alpha_avx
 _sk_luminance_to_alpha_avx LABEL PROC
-  DB  196,226,125,24,29,11,13,0,0         ; vbroadcastss  0xd0b(%rip),%ymm3        # 6098 <_sk_callback_avx+0x458>
+  DB  196,226,125,24,29,7,13,0,0          ; vbroadcastss  0xd07(%rip),%ymm3        # 60b8 <_sk_callback_avx+0x454>
   DB  197,252,89,195                      ; vmulps        %ymm3,%ymm0,%ymm0
-  DB  196,226,125,24,29,2,13,0,0          ; vbroadcastss  0xd02(%rip),%ymm3        # 609c <_sk_callback_avx+0x45c>
+  DB  196,226,125,24,29,254,12,0,0        ; vbroadcastss  0xcfe(%rip),%ymm3        # 60bc <_sk_callback_avx+0x458>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,245,12,0,0        ; vbroadcastss  0xcf5(%rip),%ymm1        # 60a0 <_sk_callback_avx+0x460>
+  DB  196,226,125,24,13,241,12,0,0        ; vbroadcastss  0xcf1(%rip),%ymm1        # 60c0 <_sk_callback_avx+0x45c>
   DB  197,236,89,201                      ; vmulps        %ymm1,%ymm2,%ymm1
   DB  197,252,88,217                      ; vaddps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -9880,7 +9893,7 @@ _sk_linear_gradient_avx LABEL PROC
   DB  196,226,125,24,88,28                ; vbroadcastss  0x1c(%rax),%ymm3
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  15,132,146,0,0,0                    ; je            5701 <_sk_linear_gradient_avx+0xb8>
+  DB  15,132,146,0,0,0                    ; je            5725 <_sk_linear_gradient_avx+0xb8>
   DB  72,139,64,8                         ; mov           0x8(%rax),%rax
   DB  72,131,192,32                       ; add           $0x20,%rax
   DB  196,65,28,87,228                    ; vxorps        %ymm12,%ymm12,%ymm12
@@ -9907,8 +9920,8 @@ _sk_linear_gradient_avx LABEL PROC
   DB  196,227,13,74,219,208               ; vblendvps     %ymm13,%ymm3,%ymm14,%ymm3
   DB  72,131,192,36                       ; add           $0x24,%rax
   DB  73,255,200                          ; dec           %r8
-  DB  117,140                             ; jne           568b <_sk_linear_gradient_avx+0x42>
-  DB  235,20                              ; jmp           5715 <_sk_linear_gradient_avx+0xcc>
+  DB  117,140                             ; jne           56af <_sk_linear_gradient_avx+0x42>
+  DB  235,20                              ; jmp           5739 <_sk_linear_gradient_avx+0xcc>
   DB  196,65,36,87,219                    ; vxorps        %ymm11,%ymm11,%ymm11
   DB  196,65,44,87,210                    ; vxorps        %ymm10,%ymm10,%ymm10
   DB  196,65,52,87,201                    ; vxorps        %ymm9,%ymm9,%ymm9
@@ -9959,27 +9972,27 @@ _sk_xy_to_polar_unit_avx LABEL PROC
   DB  196,65,52,95,226                    ; vmaxps        %ymm10,%ymm9,%ymm12
   DB  196,65,36,94,220                    ; vdivps        %ymm12,%ymm11,%ymm11
   DB  196,65,36,89,227                    ; vmulps        %ymm11,%ymm11,%ymm12
-  DB  196,98,125,24,45,218,8,0,0          ; vbroadcastss  0x8da(%rip),%ymm13        # 60a4 <_sk_callback_avx+0x464>
+  DB  196,98,125,24,45,214,8,0,0          ; vbroadcastss  0x8d6(%rip),%ymm13        # 60c4 <_sk_callback_avx+0x460>
   DB  196,65,28,89,237                    ; vmulps        %ymm13,%ymm12,%ymm13
-  DB  196,98,125,24,53,208,8,0,0          ; vbroadcastss  0x8d0(%rip),%ymm14        # 60a8 <_sk_callback_avx+0x468>
+  DB  196,98,125,24,53,204,8,0,0          ; vbroadcastss  0x8cc(%rip),%ymm14        # 60c8 <_sk_callback_avx+0x464>
   DB  196,65,20,88,238                    ; vaddps        %ymm14,%ymm13,%ymm13
   DB  196,65,28,89,237                    ; vmulps        %ymm13,%ymm12,%ymm13
-  DB  196,98,125,24,53,193,8,0,0          ; vbroadcastss  0x8c1(%rip),%ymm14        # 60ac <_sk_callback_avx+0x46c>
+  DB  196,98,125,24,53,189,8,0,0          ; vbroadcastss  0x8bd(%rip),%ymm14        # 60cc <_sk_callback_avx+0x468>
   DB  196,65,20,88,238                    ; vaddps        %ymm14,%ymm13,%ymm13
   DB  196,65,28,89,229                    ; vmulps        %ymm13,%ymm12,%ymm12
-  DB  196,98,125,24,45,178,8,0,0          ; vbroadcastss  0x8b2(%rip),%ymm13        # 60b0 <_sk_callback_avx+0x470>
+  DB  196,98,125,24,45,174,8,0,0          ; vbroadcastss  0x8ae(%rip),%ymm13        # 60d0 <_sk_callback_avx+0x46c>
   DB  196,65,28,88,229                    ; vaddps        %ymm13,%ymm12,%ymm12
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
   DB  196,65,52,194,202,1                 ; vcmpltps      %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,21,157,8,0,0          ; vbroadcastss  0x89d(%rip),%ymm10        # 60b4 <_sk_callback_avx+0x474>
+  DB  196,98,125,24,21,153,8,0,0          ; vbroadcastss  0x899(%rip),%ymm10        # 60d4 <_sk_callback_avx+0x470>
   DB  196,65,44,92,211                    ; vsubps        %ymm11,%ymm10,%ymm10
   DB  196,67,37,74,202,144                ; vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   DB  196,193,124,194,192,1               ; vcmpltps      %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,21,135,8,0,0          ; vbroadcastss  0x887(%rip),%ymm10        # 60b8 <_sk_callback_avx+0x478>
+  DB  196,98,125,24,21,131,8,0,0          ; vbroadcastss  0x883(%rip),%ymm10        # 60d8 <_sk_callback_avx+0x474>
   DB  196,65,44,92,209                    ; vsubps        %ymm9,%ymm10,%ymm10
   DB  196,195,53,74,194,0                 ; vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   DB  196,65,116,194,200,1                ; vcmpltps      %ymm8,%ymm1,%ymm9
-  DB  196,98,125,24,21,113,8,0,0          ; vbroadcastss  0x871(%rip),%ymm10        # 60bc <_sk_callback_avx+0x47c>
+  DB  196,98,125,24,21,109,8,0,0          ; vbroadcastss  0x86d(%rip),%ymm10        # 60dc <_sk_callback_avx+0x478>
   DB  197,44,92,208                       ; vsubps        %ymm0,%ymm10,%ymm10
   DB  196,195,125,74,194,144              ; vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   DB  196,65,124,194,200,3                ; vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -10000,7 +10013,7 @@ _sk_xy_to_radius_avx LABEL PROC
 PUBLIC _sk_save_xy_avx
 _sk_save_xy_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,55,8,0,0            ; vbroadcastss  0x837(%rip),%ymm8        # 60c0 <_sk_callback_avx+0x480>
+  DB  196,98,125,24,5,51,8,0,0            ; vbroadcastss  0x833(%rip),%ymm8        # 60e0 <_sk_callback_avx+0x47c>
   DB  196,65,124,88,200                   ; vaddps        %ymm8,%ymm0,%ymm9
   DB  196,67,125,8,209,1                  ; vroundps      $0x1,%ymm9,%ymm10
   DB  196,65,52,92,202                    ; vsubps        %ymm10,%ymm9,%ymm9
@@ -10033,9 +10046,9 @@ _sk_accumulate_avx LABEL PROC
 PUBLIC _sk_bilinear_nx_avx
 _sk_bilinear_nx_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,195,7,0,0          ; vbroadcastss  0x7c3(%rip),%ymm0        # 60c4 <_sk_callback_avx+0x484>
+  DB  196,226,125,24,5,191,7,0,0          ; vbroadcastss  0x7bf(%rip),%ymm0        # 60e4 <_sk_callback_avx+0x480>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,186,7,0,0           ; vbroadcastss  0x7ba(%rip),%ymm8        # 60c8 <_sk_callback_avx+0x488>
+  DB  196,98,125,24,5,182,7,0,0           ; vbroadcastss  0x7b6(%rip),%ymm8        # 60e8 <_sk_callback_avx+0x484>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10044,7 +10057,7 @@ _sk_bilinear_nx_avx LABEL PROC
 PUBLIC _sk_bilinear_px_avx
 _sk_bilinear_px_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,162,7,0,0          ; vbroadcastss  0x7a2(%rip),%ymm0        # 60cc <_sk_callback_avx+0x48c>
+  DB  196,226,125,24,5,158,7,0,0          ; vbroadcastss  0x79e(%rip),%ymm0        # 60ec <_sk_callback_avx+0x488>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -10054,9 +10067,9 @@ _sk_bilinear_px_avx LABEL PROC
 PUBLIC _sk_bilinear_ny_avx
 _sk_bilinear_ny_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,134,7,0,0         ; vbroadcastss  0x786(%rip),%ymm1        # 60d0 <_sk_callback_avx+0x490>
+  DB  196,226,125,24,13,130,7,0,0         ; vbroadcastss  0x782(%rip),%ymm1        # 60f0 <_sk_callback_avx+0x48c>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,124,7,0,0           ; vbroadcastss  0x77c(%rip),%ymm8        # 60d4 <_sk_callback_avx+0x494>
+  DB  196,98,125,24,5,120,7,0,0           ; vbroadcastss  0x778(%rip),%ymm8        # 60f4 <_sk_callback_avx+0x490>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10065,7 +10078,7 @@ _sk_bilinear_ny_avx LABEL PROC
 PUBLIC _sk_bilinear_py_avx
 _sk_bilinear_py_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,100,7,0,0         ; vbroadcastss  0x764(%rip),%ymm1        # 60d8 <_sk_callback_avx+0x498>
+  DB  196,226,125,24,13,96,7,0,0          ; vbroadcastss  0x760(%rip),%ymm1        # 60f8 <_sk_callback_avx+0x494>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -10075,14 +10088,14 @@ _sk_bilinear_py_avx LABEL PROC
 PUBLIC _sk_bicubic_n3x_avx
 _sk_bicubic_n3x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,71,7,0,0           ; vbroadcastss  0x747(%rip),%ymm0        # 60dc <_sk_callback_avx+0x49c>
+  DB  196,226,125,24,5,67,7,0,0           ; vbroadcastss  0x743(%rip),%ymm0        # 60fc <_sk_callback_avx+0x498>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,62,7,0,0            ; vbroadcastss  0x73e(%rip),%ymm8        # 60e0 <_sk_callback_avx+0x4a0>
+  DB  196,98,125,24,5,58,7,0,0            ; vbroadcastss  0x73a(%rip),%ymm8        # 6100 <_sk_callback_avx+0x49c>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,47,7,0,0           ; vbroadcastss  0x72f(%rip),%ymm10        # 60e4 <_sk_callback_avx+0x4a4>
+  DB  196,98,125,24,21,43,7,0,0           ; vbroadcastss  0x72b(%rip),%ymm10        # 6104 <_sk_callback_avx+0x4a0>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,37,7,0,0           ; vbroadcastss  0x725(%rip),%ymm10        # 60e8 <_sk_callback_avx+0x4a8>
+  DB  196,98,125,24,21,33,7,0,0           ; vbroadcastss  0x721(%rip),%ymm10        # 6108 <_sk_callback_avx+0x4a4>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -10092,19 +10105,19 @@ _sk_bicubic_n3x_avx LABEL PROC
 PUBLIC _sk_bicubic_n1x_avx
 _sk_bicubic_n1x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,8,7,0,0            ; vbroadcastss  0x708(%rip),%ymm0        # 60ec <_sk_callback_avx+0x4ac>
+  DB  196,226,125,24,5,4,7,0,0            ; vbroadcastss  0x704(%rip),%ymm0        # 610c <_sk_callback_avx+0x4a8>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,255,6,0,0           ; vbroadcastss  0x6ff(%rip),%ymm8        # 60f0 <_sk_callback_avx+0x4b0>
+  DB  196,98,125,24,5,251,6,0,0           ; vbroadcastss  0x6fb(%rip),%ymm8        # 6110 <_sk_callback_avx+0x4ac>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,245,6,0,0          ; vbroadcastss  0x6f5(%rip),%ymm9        # 60f4 <_sk_callback_avx+0x4b4>
+  DB  196,98,125,24,13,241,6,0,0          ; vbroadcastss  0x6f1(%rip),%ymm9        # 6114 <_sk_callback_avx+0x4b0>
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,235,6,0,0          ; vbroadcastss  0x6eb(%rip),%ymm10        # 60f8 <_sk_callback_avx+0x4b8>
+  DB  196,98,125,24,21,231,6,0,0          ; vbroadcastss  0x6e7(%rip),%ymm10        # 6118 <_sk_callback_avx+0x4b4>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,220,6,0,0          ; vbroadcastss  0x6dc(%rip),%ymm10        # 60fc <_sk_callback_avx+0x4bc>
+  DB  196,98,125,24,21,216,6,0,0          ; vbroadcastss  0x6d8(%rip),%ymm10        # 611c <_sk_callback_avx+0x4b8>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,205,6,0,0          ; vbroadcastss  0x6cd(%rip),%ymm9        # 6100 <_sk_callback_avx+0x4c0>
+  DB  196,98,125,24,13,201,6,0,0          ; vbroadcastss  0x6c9(%rip),%ymm9        # 6120 <_sk_callback_avx+0x4bc>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10113,17 +10126,17 @@ _sk_bicubic_n1x_avx LABEL PROC
 PUBLIC _sk_bicubic_p1x_avx
 _sk_bicubic_p1x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,181,6,0,0           ; vbroadcastss  0x6b5(%rip),%ymm8        # 6104 <_sk_callback_avx+0x4c4>
+  DB  196,98,125,24,5,177,6,0,0           ; vbroadcastss  0x6b1(%rip),%ymm8        # 6124 <_sk_callback_avx+0x4c0>
   DB  197,188,88,0                        ; vaddps        (%rax),%ymm8,%ymm0
   DB  197,124,16,72,64                    ; vmovups       0x40(%rax),%ymm9
-  DB  196,98,125,24,21,167,6,0,0          ; vbroadcastss  0x6a7(%rip),%ymm10        # 6108 <_sk_callback_avx+0x4c8>
+  DB  196,98,125,24,21,163,6,0,0          ; vbroadcastss  0x6a3(%rip),%ymm10        # 6128 <_sk_callback_avx+0x4c4>
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
-  DB  196,98,125,24,29,157,6,0,0          ; vbroadcastss  0x69d(%rip),%ymm11        # 610c <_sk_callback_avx+0x4cc>
+  DB  196,98,125,24,29,153,6,0,0          ; vbroadcastss  0x699(%rip),%ymm11        # 612c <_sk_callback_avx+0x4c8>
   DB  196,65,44,88,211                    ; vaddps        %ymm11,%ymm10,%ymm10
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
   DB  196,65,44,88,192                    ; vaddps        %ymm8,%ymm10,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
-  DB  196,98,125,24,13,132,6,0,0          ; vbroadcastss  0x684(%rip),%ymm9        # 6110 <_sk_callback_avx+0x4d0>
+  DB  196,98,125,24,13,128,6,0,0          ; vbroadcastss  0x680(%rip),%ymm9        # 6130 <_sk_callback_avx+0x4cc>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10132,13 +10145,13 @@ _sk_bicubic_p1x_avx LABEL PROC
 PUBLIC _sk_bicubic_p3x_avx
 _sk_bicubic_p3x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,108,6,0,0          ; vbroadcastss  0x66c(%rip),%ymm0        # 6114 <_sk_callback_avx+0x4d4>
+  DB  196,226,125,24,5,104,6,0,0          ; vbroadcastss  0x668(%rip),%ymm0        # 6134 <_sk_callback_avx+0x4d0>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,89,6,0,0           ; vbroadcastss  0x659(%rip),%ymm10        # 6118 <_sk_callback_avx+0x4d8>
+  DB  196,98,125,24,21,85,6,0,0           ; vbroadcastss  0x655(%rip),%ymm10        # 6138 <_sk_callback_avx+0x4d4>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,79,6,0,0           ; vbroadcastss  0x64f(%rip),%ymm10        # 611c <_sk_callback_avx+0x4dc>
+  DB  196,98,125,24,21,75,6,0,0           ; vbroadcastss  0x64b(%rip),%ymm10        # 613c <_sk_callback_avx+0x4d8>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -10148,14 +10161,14 @@ _sk_bicubic_p3x_avx LABEL PROC
 PUBLIC _sk_bicubic_n3y_avx
 _sk_bicubic_n3y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,50,6,0,0          ; vbroadcastss  0x632(%rip),%ymm1        # 6120 <_sk_callback_avx+0x4e0>
+  DB  196,226,125,24,13,46,6,0,0          ; vbroadcastss  0x62e(%rip),%ymm1        # 6140 <_sk_callback_avx+0x4dc>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,40,6,0,0            ; vbroadcastss  0x628(%rip),%ymm8        # 6124 <_sk_callback_avx+0x4e4>
+  DB  196,98,125,24,5,36,6,0,0            ; vbroadcastss  0x624(%rip),%ymm8        # 6144 <_sk_callback_avx+0x4e0>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,25,6,0,0           ; vbroadcastss  0x619(%rip),%ymm10        # 6128 <_sk_callback_avx+0x4e8>
+  DB  196,98,125,24,21,21,6,0,0           ; vbroadcastss  0x615(%rip),%ymm10        # 6148 <_sk_callback_avx+0x4e4>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,15,6,0,0           ; vbroadcastss  0x60f(%rip),%ymm10        # 612c <_sk_callback_avx+0x4ec>
+  DB  196,98,125,24,21,11,6,0,0           ; vbroadcastss  0x60b(%rip),%ymm10        # 614c <_sk_callback_avx+0x4e8>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -10165,19 +10178,19 @@ _sk_bicubic_n3y_avx LABEL PROC
 PUBLIC _sk_bicubic_n1y_avx
 _sk_bicubic_n1y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,242,5,0,0         ; vbroadcastss  0x5f2(%rip),%ymm1        # 6130 <_sk_callback_avx+0x4f0>
+  DB  196,226,125,24,13,238,5,0,0         ; vbroadcastss  0x5ee(%rip),%ymm1        # 6150 <_sk_callback_avx+0x4ec>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,232,5,0,0           ; vbroadcastss  0x5e8(%rip),%ymm8        # 6134 <_sk_callback_avx+0x4f4>
+  DB  196,98,125,24,5,228,5,0,0           ; vbroadcastss  0x5e4(%rip),%ymm8        # 6154 <_sk_callback_avx+0x4f0>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,222,5,0,0          ; vbroadcastss  0x5de(%rip),%ymm9        # 6138 <_sk_callback_avx+0x4f8>
+  DB  196,98,125,24,13,218,5,0,0          ; vbroadcastss  0x5da(%rip),%ymm9        # 6158 <_sk_callback_avx+0x4f4>
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,212,5,0,0          ; vbroadcastss  0x5d4(%rip),%ymm10        # 613c <_sk_callback_avx+0x4fc>
+  DB  196,98,125,24,21,208,5,0,0          ; vbroadcastss  0x5d0(%rip),%ymm10        # 615c <_sk_callback_avx+0x4f8>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,197,5,0,0          ; vbroadcastss  0x5c5(%rip),%ymm10        # 6140 <_sk_callback_avx+0x500>
+  DB  196,98,125,24,21,193,5,0,0          ; vbroadcastss  0x5c1(%rip),%ymm10        # 6160 <_sk_callback_avx+0x4fc>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,182,5,0,0          ; vbroadcastss  0x5b6(%rip),%ymm9        # 6144 <_sk_callback_avx+0x504>
+  DB  196,98,125,24,13,178,5,0,0          ; vbroadcastss  0x5b2(%rip),%ymm9        # 6164 <_sk_callback_avx+0x500>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10186,17 +10199,17 @@ _sk_bicubic_n1y_avx LABEL PROC
 PUBLIC _sk_bicubic_p1y_avx
 _sk_bicubic_p1y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,158,5,0,0           ; vbroadcastss  0x59e(%rip),%ymm8        # 6148 <_sk_callback_avx+0x508>
+  DB  196,98,125,24,5,154,5,0,0           ; vbroadcastss  0x59a(%rip),%ymm8        # 6168 <_sk_callback_avx+0x504>
   DB  197,188,88,72,32                    ; vaddps        0x20(%rax),%ymm8,%ymm1
   DB  197,124,16,72,96                    ; vmovups       0x60(%rax),%ymm9
-  DB  196,98,125,24,21,143,5,0,0          ; vbroadcastss  0x58f(%rip),%ymm10        # 614c <_sk_callback_avx+0x50c>
+  DB  196,98,125,24,21,139,5,0,0          ; vbroadcastss  0x58b(%rip),%ymm10        # 616c <_sk_callback_avx+0x508>
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
-  DB  196,98,125,24,29,133,5,0,0          ; vbroadcastss  0x585(%rip),%ymm11        # 6150 <_sk_callback_avx+0x510>
+  DB  196,98,125,24,29,129,5,0,0          ; vbroadcastss  0x581(%rip),%ymm11        # 6170 <_sk_callback_avx+0x50c>
   DB  196,65,44,88,211                    ; vaddps        %ymm11,%ymm10,%ymm10
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
   DB  196,65,44,88,192                    ; vaddps        %ymm8,%ymm10,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
-  DB  196,98,125,24,13,108,5,0,0          ; vbroadcastss  0x56c(%rip),%ymm9        # 6154 <_sk_callback_avx+0x514>
+  DB  196,98,125,24,13,104,5,0,0          ; vbroadcastss  0x568(%rip),%ymm9        # 6174 <_sk_callback_avx+0x510>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10205,13 +10218,13 @@ _sk_bicubic_p1y_avx LABEL PROC
 PUBLIC _sk_bicubic_p3y_avx
 _sk_bicubic_p3y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,84,5,0,0          ; vbroadcastss  0x554(%rip),%ymm1        # 6158 <_sk_callback_avx+0x518>
+  DB  196,226,125,24,13,80,5,0,0          ; vbroadcastss  0x550(%rip),%ymm1        # 6178 <_sk_callback_avx+0x514>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,64,5,0,0           ; vbroadcastss  0x540(%rip),%ymm10        # 615c <_sk_callback_avx+0x51c>
+  DB  196,98,125,24,21,60,5,0,0           ; vbroadcastss  0x53c(%rip),%ymm10        # 617c <_sk_callback_avx+0x518>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,54,5,0,0           ; vbroadcastss  0x536(%rip),%ymm10        # 6160 <_sk_callback_avx+0x520>
+  DB  196,98,125,24,21,50,5,0,0           ; vbroadcastss  0x532(%rip),%ymm10        # 6180 <_sk_callback_avx+0x51c>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -10326,25 +10339,25 @@ ALIGN 4
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 5e11 <.literal4+0xb5>
+  DB  71,225,61                           ; rex.RXB       loope 5e35 <.literal4+0xb5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 5e21 <.literal4+0xc5>
+  DB  71,225,61                           ; rex.RXB       loope 5e45 <.literal4+0xc5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 5e31 <.literal4+0xd5>
+  DB  71,225,61                           ; rex.RXB       loope 5e55 <.literal4+0xd5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 5e41 <.literal4+0xe5>
+  DB  71,225,61                           ; rex.RXB       loope 5e65 <.literal4+0xe5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -10393,24 +10406,26 @@ ALIGN 4
   DB  190,129,128,128,59                  ; mov           $0x3b808081,%esi
   DB  129,128,128,59,0,248,0,0,8,33       ; addl          $0x21080000,-0x7ffc480(%rax)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5e89 <.literal4+0x12d>
+  DB  224,7                               ; loopne        5ead <.literal4+0x12d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
   DB  31                                  ; (bad)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,8                                 ; add           %cl,(%rax)
-  DB  33,4,61,0,0,128,63                  ; and           %eax,0x3f800000(,%rdi,1)
-  DB  129,128,128,59,128,0,128,55,0,0     ; addl          $0x3780,0x803b80(%rax)
+  DB  33,4,61,129,128,128,59              ; and           %eax,0x3b808081(,%rdi,1)
+  DB  128,0,128                           ; addb          $0x80,(%rax)
+  DB  55                                  ; (bad)
+  DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
   DB  0,52,255                            ; add           %dh,(%rdi,%rdi,8)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5eb4 <.literal4+0x158>
+  DB  127,0                               ; jg            5ed4 <.literal4+0x154>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5f2d <.literal4+0x1d1>
+  DB  119,115                             ; ja            5f4d <.literal4+0x1cd>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10424,10 +10439,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5ee8 <.literal4+0x18c>
+  DB  127,0                               ; jg            5f08 <.literal4+0x188>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5f61 <.literal4+0x205>
+  DB  119,115                             ; ja            5f81 <.literal4+0x201>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10441,10 +10456,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5f1c <.literal4+0x1c0>
+  DB  127,0                               ; jg            5f3c <.literal4+0x1bc>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5f95 <.literal4+0x239>
+  DB  119,115                             ; ja            5fb5 <.literal4+0x235>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10458,10 +10473,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5f50 <.literal4+0x1f4>
+  DB  127,0                               ; jg            5f70 <.literal4+0x1f0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5fc9 <.literal4+0x26d>
+  DB  119,115                             ; ja            5fe9 <.literal4+0x269>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10474,7 +10489,7 @@ ALIGN 4
   DB  0,75,0                              ; add           %cl,0x0(%rbx)
   DB  0,128,63,0,0,200                    ; add           %al,-0x37ffffc1(%rax)
   DB  66,0,0                              ; rex.X         add %al,(%rax)
-  DB  127,67                              ; jg            5fc7 <.literal4+0x26b>
+  DB  127,67                              ; jg            5fe7 <.literal4+0x267>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -10486,10 +10501,10 @@ ALIGN 4
   DB  190,80,128,3,62                     ; mov           $0x3e038050,%esi
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5fe7 <.literal4+0x28b>
+  DB  118,63                              ; jbe           6007 <.literal4+0x287>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            5ffb <.literal4+0x29f>
+  DB  127,67                              ; jg            601b <.literal4+0x29b>
   DB  129,128,128,59,0,0,128,63,129,128   ; addl          $0x80813f80,0x3b80(%rax)
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,128,63,129,128,128                ; add           %al,-0x7f7f7ec1(%rax)
@@ -10498,7 +10513,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5fdd <.literal4+0x281>
+  DB  224,7                               ; loopne        5ffd <.literal4+0x27d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -10510,7 +10525,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5ff9 <.literal4+0x29d>
+  DB  224,7                               ; loopne        6019 <.literal4+0x299>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -10521,7 +10536,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            604e <.literal4+0x2f2>
+  DB  124,66                              ; jl            606e <.literal4+0x2ee>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,55,0,15                 ; mov           %ecx,0xf003788(%rax)
@@ -10539,9 +10554,9 @@ ALIGN 4
   DB  137,136,136,59,15,0                 ; mov           %ecx,0xf3b88(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,61,0,0                  ; mov           %ecx,0x3d88(%rax)
-  DB  112,65                              ; jo            6091 <.literal4+0x335>
+  DB  112,65                              ; jo            60b1 <.literal4+0x331>
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            609f <.literal4+0x343>
+  DB  127,67                              ; jg            60bf <.literal4+0x33f>
   DB  0,128,0,0,0,0                       ; add           %al,0x0(%rax)
   DB  0,128,0,4,0,128                     ; add           %al,-0x7ffffc00(%rax)
   DB  0,0                                 ; add           %al,(%rax)
@@ -10557,7 +10572,7 @@ ALIGN 4
   DB  0,128,55,0,0,128                    ; add           %al,-0x7fffffc9(%rax)
   DB  63                                  ; (bad)
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            60df <.literal4+0x383>
+  DB  127,71                              ; jg            60ff <.literal4+0x37f>
   DB  208                                 ; (bad)
   DB  179,89                              ; mov           $0x59,%bl
   DB  62,89                               ; ds            pop %rcx
@@ -10805,7 +10820,7 @@ _sk_seed_shader_sse41 LABEL PROC
   DB  102,15,110,199                      ; movd          %edi,%xmm0
   DB  102,15,112,192,0                    ; pshufd        $0x0,%xmm0,%xmm0
   DB  15,91,200                           ; cvtdq2ps      %xmm0,%xmm1
-  DB  15,40,21,209,66,0,0                 ; movaps        0x42d1(%rip),%xmm2        # 43e0 <_sk_callback_sse41+0xab>
+  DB  15,40,21,1,67,0,0                   ; movaps        0x4301(%rip),%xmm2        # 4410 <_sk_callback_sse41+0xb7>
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  15,16,2                             ; movups        (%rdx),%xmm0
   DB  15,88,193                           ; addps         %xmm1,%xmm0
@@ -10814,7 +10829,7 @@ _sk_seed_shader_sse41 LABEL PROC
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,21,192,66,0,0                 ; movaps        0x42c0(%rip),%xmm2        # 43f0 <_sk_callback_sse41+0xbb>
+  DB  15,40,21,240,66,0,0                 ; movaps        0x42f0(%rip),%xmm2        # 4420 <_sk_callback_sse41+0xc7>
   DB  15,87,219                           ; xorps         %xmm3,%xmm3
   DB  15,87,228                           ; xorps         %xmm4,%xmm4
   DB  15,87,237                           ; xorps         %xmm5,%xmm5
@@ -10835,14 +10850,14 @@ _sk_dither_sse41 LABEL PROC
   DB  102,68,15,110,1                     ; movd          (%rcx),%xmm8
   DB  102,69,15,112,192,0                 ; pshufd        $0x0,%xmm8,%xmm8
   DB  102,69,15,239,193                   ; pxor          %xmm9,%xmm8
-  DB  102,68,15,111,21,133,66,0,0         ; movdqa        0x4285(%rip),%xmm10        # 4400 <_sk_callback_sse41+0xcb>
+  DB  102,68,15,111,21,181,66,0,0         ; movdqa        0x42b5(%rip),%xmm10        # 4430 <_sk_callback_sse41+0xd7>
   DB  102,69,15,111,216                   ; movdqa        %xmm8,%xmm11
   DB  102,69,15,219,218                   ; pand          %xmm10,%xmm11
   DB  102,65,15,114,243,5                 ; pslld         $0x5,%xmm11
   DB  102,69,15,219,209                   ; pand          %xmm9,%xmm10
   DB  102,65,15,114,242,4                 ; pslld         $0x4,%xmm10
-  DB  102,68,15,111,37,113,66,0,0         ; movdqa        0x4271(%rip),%xmm12        # 4410 <_sk_callback_sse41+0xdb>
-  DB  102,68,15,111,45,120,66,0,0         ; movdqa        0x4278(%rip),%xmm13        # 4420 <_sk_callback_sse41+0xeb>
+  DB  102,68,15,111,37,161,66,0,0         ; movdqa        0x42a1(%rip),%xmm12        # 4440 <_sk_callback_sse41+0xe7>
+  DB  102,68,15,111,45,168,66,0,0         ; movdqa        0x42a8(%rip),%xmm13        # 4450 <_sk_callback_sse41+0xf7>
   DB  102,69,15,111,240                   ; movdqa        %xmm8,%xmm14
   DB  102,69,15,219,245                   ; pand          %xmm13,%xmm14
   DB  102,65,15,114,246,2                 ; pslld         $0x2,%xmm14
@@ -10858,8 +10873,8 @@ _sk_dither_sse41 LABEL PROC
   DB  102,69,15,235,245                   ; por           %xmm13,%xmm14
   DB  102,69,15,235,240                   ; por           %xmm8,%xmm14
   DB  69,15,91,198                        ; cvtdq2ps      %xmm14,%xmm8
-  DB  68,15,89,5,51,66,0,0                ; mulps         0x4233(%rip),%xmm8        # 4430 <_sk_callback_sse41+0xfb>
-  DB  68,15,88,5,59,66,0,0                ; addps         0x423b(%rip),%xmm8        # 4440 <_sk_callback_sse41+0x10b>
+  DB  68,15,89,5,99,66,0,0                ; mulps         0x4263(%rip),%xmm8        # 4460 <_sk_callback_sse41+0x107>
+  DB  68,15,88,5,107,66,0,0               ; addps         0x426b(%rip),%xmm8        # 4470 <_sk_callback_sse41+0x117>
   DB  243,68,15,16,72,8                   ; movss         0x8(%rax),%xmm9
   DB  69,15,198,201,0                     ; shufps        $0x0,%xmm9,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
@@ -10895,7 +10910,7 @@ _sk_clear_sse41 LABEL PROC
 PUBLIC _sk_srcatop_sse41
 _sk_srcatop_sse41 LABEL PROC
   DB  15,89,199                           ; mulps         %xmm7,%xmm0
-  DB  68,15,40,5,232,65,0,0               ; movaps        0x41e8(%rip),%xmm8        # 4450 <_sk_callback_sse41+0x11b>
+  DB  68,15,40,5,24,66,0,0                ; movaps        0x4218(%rip),%xmm8        # 4480 <_sk_callback_sse41+0x127>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -10918,7 +10933,7 @@ PUBLIC _sk_dstatop_sse41
 _sk_dstatop_sse41 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
   DB  68,15,89,196                        ; mulps         %xmm4,%xmm8
-  DB  68,15,40,13,171,65,0,0              ; movaps        0x41ab(%rip),%xmm9        # 4460 <_sk_callback_sse41+0x12b>
+  DB  68,15,40,13,219,65,0,0              ; movaps        0x41db(%rip),%xmm9        # 4490 <_sk_callback_sse41+0x137>
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
@@ -10959,7 +10974,7 @@ _sk_dstin_sse41 LABEL PROC
 
 PUBLIC _sk_srcout_sse41
 _sk_srcout_sse41 LABEL PROC
-  DB  68,15,40,5,79,65,0,0                ; movaps        0x414f(%rip),%xmm8        # 4470 <_sk_callback_sse41+0x13b>
+  DB  68,15,40,5,127,65,0,0               ; movaps        0x417f(%rip),%xmm8        # 44a0 <_sk_callback_sse41+0x147>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
@@ -10970,7 +10985,7 @@ _sk_srcout_sse41 LABEL PROC
 
 PUBLIC _sk_dstout_sse41
 _sk_dstout_sse41 LABEL PROC
-  DB  68,15,40,5,63,65,0,0                ; movaps        0x413f(%rip),%xmm8        # 4480 <_sk_callback_sse41+0x14b>
+  DB  68,15,40,5,111,65,0,0               ; movaps        0x416f(%rip),%xmm8        # 44b0 <_sk_callback_sse41+0x157>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  15,89,196                           ; mulps         %xmm4,%xmm0
@@ -10985,7 +11000,7 @@ _sk_dstout_sse41 LABEL PROC
 
 PUBLIC _sk_srcover_sse41
 _sk_srcover_sse41 LABEL PROC
-  DB  68,15,40,5,34,65,0,0                ; movaps        0x4122(%rip),%xmm8        # 4490 <_sk_callback_sse41+0x15b>
+  DB  68,15,40,5,82,65,0,0                ; movaps        0x4152(%rip),%xmm8        # 44c0 <_sk_callback_sse41+0x167>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -11003,7 +11018,7 @@ _sk_srcover_sse41 LABEL PROC
 
 PUBLIC _sk_dstover_sse41
 _sk_dstover_sse41 LABEL PROC
-  DB  68,15,40,5,246,64,0,0               ; movaps        0x40f6(%rip),%xmm8        # 44a0 <_sk_callback_sse41+0x16b>
+  DB  68,15,40,5,38,65,0,0                ; movaps        0x4126(%rip),%xmm8        # 44d0 <_sk_callback_sse41+0x177>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -11027,7 +11042,7 @@ _sk_modulate_sse41 LABEL PROC
 
 PUBLIC _sk_multiply_sse41
 _sk_multiply_sse41 LABEL PROC
-  DB  68,15,40,5,202,64,0,0               ; movaps        0x40ca(%rip),%xmm8        # 44b0 <_sk_callback_sse41+0x17b>
+  DB  68,15,40,5,250,64,0,0               ; movaps        0x40fa(%rip),%xmm8        # 44e0 <_sk_callback_sse41+0x187>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
@@ -11097,7 +11112,7 @@ _sk_screen_sse41 LABEL PROC
 PUBLIC _sk_xor__sse41
 _sk_xor__sse41 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
-  DB  15,40,29,251,63,0,0                 ; movaps        0x3ffb(%rip),%xmm3        # 44c0 <_sk_callback_sse41+0x18b>
+  DB  15,40,29,43,64,0,0                  ; movaps        0x402b(%rip),%xmm3        # 44f0 <_sk_callback_sse41+0x197>
   DB  68,15,40,203                        ; movaps        %xmm3,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
@@ -11143,7 +11158,7 @@ _sk_darken_sse41 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,95,209                        ; maxps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,102,63,0,0                 ; movaps        0x3f66(%rip),%xmm2        # 44d0 <_sk_callback_sse41+0x19b>
+  DB  15,40,21,150,63,0,0                 ; movaps        0x3f96(%rip),%xmm2        # 4500 <_sk_callback_sse41+0x1a7>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -11175,7 +11190,7 @@ _sk_lighten_sse41 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,11,63,0,0                  ; movaps        0x3f0b(%rip),%xmm2        # 44e0 <_sk_callback_sse41+0x1ab>
+  DB  15,40,21,59,63,0,0                  ; movaps        0x3f3b(%rip),%xmm2        # 4510 <_sk_callback_sse41+0x1b7>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -11210,7 +11225,7 @@ _sk_difference_sse41 LABEL PROC
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,165,62,0,0                 ; movaps        0x3ea5(%rip),%xmm2        # 44f0 <_sk_callback_sse41+0x1bb>
+  DB  15,40,21,213,62,0,0                 ; movaps        0x3ed5(%rip),%xmm2        # 4520 <_sk_callback_sse41+0x1c7>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -11235,7 +11250,7 @@ _sk_exclusion_sse41 LABEL PROC
   DB  15,89,214                           ; mulps         %xmm6,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,202                        ; subps         %xmm2,%xmm9
-  DB  15,40,13,102,62,0,0                 ; movaps        0x3e66(%rip),%xmm1        # 4500 <_sk_callback_sse41+0x1cb>
+  DB  15,40,13,150,62,0,0                 ; movaps        0x3e96(%rip),%xmm1        # 4530 <_sk_callback_sse41+0x1d7>
   DB  15,92,203                           ; subps         %xmm3,%xmm1
   DB  15,89,207                           ; mulps         %xmm7,%xmm1
   DB  15,88,217                           ; addps         %xmm1,%xmm3
@@ -11247,7 +11262,7 @@ _sk_exclusion_sse41 LABEL PROC
 PUBLIC _sk_colorburn_sse41
 _sk_colorburn_sse41 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,40,21,85,62,0,0               ; movaps        0x3e55(%rip),%xmm10        # 4510 <_sk_callback_sse41+0x1db>
+  DB  68,15,40,21,133,62,0,0              ; movaps        0x3e85(%rip),%xmm10        # 4540 <_sk_callback_sse41+0x1e7>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  69,15,40,203                        ; movaps        %xmm11,%xmm9
@@ -11327,7 +11342,7 @@ _sk_colorburn_sse41 LABEL PROC
 PUBLIC _sk_colordodge_sse41
 _sk_colordodge_sse41 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,40,21,51,61,0,0               ; movaps        0x3d33(%rip),%xmm10        # 4520 <_sk_callback_sse41+0x1eb>
+  DB  68,15,40,21,99,61,0,0               ; movaps        0x3d63(%rip),%xmm10        # 4550 <_sk_callback_sse41+0x1f7>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  69,15,40,227                        ; movaps        %xmm11,%xmm12
@@ -11408,7 +11423,7 @@ _sk_hardlight_sse41 LABEL PROC
   DB  15,40,244                           ; movaps        %xmm4,%xmm6
   DB  15,40,227                           ; movaps        %xmm3,%xmm4
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
-  DB  68,15,40,21,9,60,0,0                ; movaps        0x3c09(%rip),%xmm10        # 4530 <_sk_callback_sse41+0x1fb>
+  DB  68,15,40,21,57,60,0,0               ; movaps        0x3c39(%rip),%xmm10        # 4560 <_sk_callback_sse41+0x207>
   DB  65,15,40,234                        ; movaps        %xmm10,%xmm5
   DB  15,92,239                           ; subps         %xmm7,%xmm5
   DB  15,40,197                           ; movaps        %xmm5,%xmm0
@@ -11490,7 +11505,7 @@ PUBLIC _sk_overlay_sse41
 _sk_overlay_sse41 LABEL PROC
   DB  68,15,40,201                        ; movaps        %xmm1,%xmm9
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
-  DB  68,15,40,21,235,58,0,0              ; movaps        0x3aeb(%rip),%xmm10        # 4540 <_sk_callback_sse41+0x20b>
+  DB  68,15,40,21,27,59,0,0               ; movaps        0x3b1b(%rip),%xmm10        # 4570 <_sk_callback_sse41+0x217>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
@@ -11574,7 +11589,7 @@ _sk_softlight_sse41 LABEL PROC
   DB  15,40,198                           ; movaps        %xmm6,%xmm0
   DB  15,94,199                           ; divps         %xmm7,%xmm0
   DB  65,15,84,193                        ; andps         %xmm9,%xmm0
-  DB  15,40,13,190,57,0,0                 ; movaps        0x39be(%rip),%xmm1        # 4550 <_sk_callback_sse41+0x21b>
+  DB  15,40,13,238,57,0,0                 ; movaps        0x39ee(%rip),%xmm1        # 4580 <_sk_callback_sse41+0x227>
   DB  68,15,40,209                        ; movaps        %xmm1,%xmm10
   DB  68,15,92,208                        ; subps         %xmm0,%xmm10
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
@@ -11587,10 +11602,10 @@ _sk_softlight_sse41 LABEL PROC
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  15,89,210                           ; mulps         %xmm2,%xmm2
   DB  15,88,208                           ; addps         %xmm0,%xmm2
-  DB  68,15,40,45,156,57,0,0              ; movaps        0x399c(%rip),%xmm13        # 4560 <_sk_callback_sse41+0x22b>
+  DB  68,15,40,45,204,57,0,0              ; movaps        0x39cc(%rip),%xmm13        # 4590 <_sk_callback_sse41+0x237>
   DB  69,15,88,245                        ; addps         %xmm13,%xmm14
   DB  68,15,89,242                        ; mulps         %xmm2,%xmm14
-  DB  68,15,40,37,156,57,0,0              ; movaps        0x399c(%rip),%xmm12        # 4570 <_sk_callback_sse41+0x23b>
+  DB  68,15,40,37,204,57,0,0              ; movaps        0x39cc(%rip),%xmm12        # 45a0 <_sk_callback_sse41+0x247>
   DB  69,15,89,252                        ; mulps         %xmm12,%xmm15
   DB  69,15,88,254                        ; addps         %xmm14,%xmm15
   DB  15,40,198                           ; movaps        %xmm6,%xmm0
@@ -11733,7 +11748,7 @@ _sk_hue_sse41 LABEL PROC
   DB  15,40,243                           ; movaps        %xmm3,%xmm6
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,87,246                        ; xorps         %xmm14,%xmm14
-  DB  68,15,40,45,165,55,0,0              ; movaps        0x37a5(%rip),%xmm13        # 4580 <_sk_callback_sse41+0x24b>
+  DB  68,15,40,45,213,55,0,0              ; movaps        0x37d5(%rip),%xmm13        # 45b0 <_sk_callback_sse41+0x257>
   DB  65,15,40,221                        ; movaps        %xmm13,%xmm3
   DB  15,94,222                           ; divps         %xmm6,%xmm3
   DB  15,40,198                           ; movaps        %xmm6,%xmm0
@@ -11777,12 +11792,12 @@ _sk_hue_sse41 LABEL PROC
   DB  68,15,84,194                        ; andps         %xmm2,%xmm8
   DB  15,84,202                           ; andps         %xmm2,%xmm1
   DB  15,84,194                           ; andps         %xmm2,%xmm0
-  DB  68,15,40,13,21,55,0,0               ; movaps        0x3715(%rip),%xmm9        # 4590 <_sk_callback_sse41+0x25b>
+  DB  68,15,40,13,69,55,0,0               ; movaps        0x3745(%rip),%xmm9        # 45c0 <_sk_callback_sse41+0x267>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  15,40,29,26,55,0,0                  ; movaps        0x371a(%rip),%xmm3        # 45a0 <_sk_callback_sse41+0x26b>
+  DB  15,40,29,74,55,0,0                  ; movaps        0x374a(%rip),%xmm3        # 45d0 <_sk_callback_sse41+0x277>
   DB  68,15,89,219                        ; mulps         %xmm3,%xmm11
   DB  69,15,88,218                        ; addps         %xmm10,%xmm11
-  DB  68,15,40,53,26,55,0,0               ; movaps        0x371a(%rip),%xmm14        # 45b0 <_sk_callback_sse41+0x27b>
+  DB  68,15,40,53,74,55,0,0               ; movaps        0x374a(%rip),%xmm14        # 45e0 <_sk_callback_sse41+0x287>
   DB  68,15,40,253                        ; movaps        %xmm5,%xmm15
   DB  69,15,89,254                        ; mulps         %xmm14,%xmm15
   DB  69,15,88,251                        ; addps         %xmm11,%xmm15
@@ -11890,7 +11905,7 @@ _sk_saturation_sse41 LABEL PROC
   DB  68,15,40,220                        ; movaps        %xmm4,%xmm11
   DB  15,40,243                           ; movaps        %xmm3,%xmm6
   DB  69,15,87,246                        ; xorps         %xmm14,%xmm14
-  DB  68,15,40,37,140,53,0,0              ; movaps        0x358c(%rip),%xmm12        # 45c0 <_sk_callback_sse41+0x28b>
+  DB  68,15,40,37,188,53,0,0              ; movaps        0x35bc(%rip),%xmm12        # 45f0 <_sk_callback_sse41+0x297>
   DB  65,15,40,220                        ; movaps        %xmm12,%xmm3
   DB  15,94,223                           ; divps         %xmm7,%xmm3
   DB  68,15,40,199                        ; movaps        %xmm7,%xmm8
@@ -11932,14 +11947,14 @@ _sk_saturation_sse41 LABEL PROC
   DB  68,15,84,202                        ; andps         %xmm2,%xmm9
   DB  68,15,84,234                        ; andps         %xmm2,%xmm13
   DB  68,15,84,194                        ; andps         %xmm2,%xmm8
-  DB  15,40,13,248,52,0,0                 ; movaps        0x34f8(%rip),%xmm1        # 45d0 <_sk_callback_sse41+0x29b>
+  DB  15,40,13,40,53,0,0                  ; movaps        0x3528(%rip),%xmm1        # 4600 <_sk_callback_sse41+0x2a7>
   DB  65,15,40,211                        ; movaps        %xmm11,%xmm2
   DB  15,89,209                           ; mulps         %xmm1,%xmm2
-  DB  15,40,5,250,52,0,0                  ; movaps        0x34fa(%rip),%xmm0        # 45e0 <_sk_callback_sse41+0x2ab>
+  DB  15,40,5,42,53,0,0                   ; movaps        0x352a(%rip),%xmm0        # 4610 <_sk_callback_sse41+0x2b7>
   DB  15,40,221                           ; movaps        %xmm5,%xmm3
   DB  15,89,216                           ; mulps         %xmm0,%xmm3
   DB  15,88,218                           ; addps         %xmm2,%xmm3
-  DB  68,15,40,53,249,52,0,0              ; movaps        0x34f9(%rip),%xmm14        # 45f0 <_sk_callback_sse41+0x2bb>
+  DB  68,15,40,53,41,53,0,0               ; movaps        0x3529(%rip),%xmm14        # 4620 <_sk_callback_sse41+0x2c7>
   DB  69,15,40,250                        ; movaps        %xmm10,%xmm15
   DB  69,15,89,254                        ; mulps         %xmm14,%xmm15
   DB  68,15,88,251                        ; addps         %xmm3,%xmm15
@@ -12047,7 +12062,7 @@ _sk_color_sse41 LABEL PROC
   DB  15,40,227                           ; movaps        %xmm3,%xmm4
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,87,201                        ; xorps         %xmm9,%xmm9
-  DB  68,15,40,45,108,51,0,0              ; movaps        0x336c(%rip),%xmm13        # 4600 <_sk_callback_sse41+0x2cb>
+  DB  68,15,40,45,156,51,0,0              ; movaps        0x339c(%rip),%xmm13        # 4630 <_sk_callback_sse41+0x2d7>
   DB  65,15,40,197                        ; movaps        %xmm13,%xmm0
   DB  15,94,196                           ; divps         %xmm4,%xmm0
   DB  65,15,194,217,4                     ; cmpneqps      %xmm9,%xmm3
@@ -12055,13 +12070,13 @@ _sk_color_sse41 LABEL PROC
   DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
   DB  15,89,203                           ; mulps         %xmm3,%xmm1
   DB  15,89,218                           ; mulps         %xmm2,%xmm3
-  DB  68,15,40,13,91,51,0,0               ; movaps        0x335b(%rip),%xmm9        # 4610 <_sk_callback_sse41+0x2db>
+  DB  68,15,40,13,139,51,0,0              ; movaps        0x338b(%rip),%xmm9        # 4640 <_sk_callback_sse41+0x2e7>
   DB  15,40,213                           ; movaps        %xmm5,%xmm2
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
-  DB  68,15,40,21,92,51,0,0               ; movaps        0x335c(%rip),%xmm10        # 4620 <_sk_callback_sse41+0x2eb>
+  DB  68,15,40,21,140,51,0,0              ; movaps        0x338c(%rip),%xmm10        # 4650 <_sk_callback_sse41+0x2f7>
   DB  69,15,89,218                        ; mulps         %xmm10,%xmm11
   DB  68,15,88,218                        ; addps         %xmm2,%xmm11
-  DB  68,15,40,53,92,51,0,0               ; movaps        0x335c(%rip),%xmm14        # 4630 <_sk_callback_sse41+0x2fb>
+  DB  68,15,40,53,140,51,0,0              ; movaps        0x338c(%rip),%xmm14        # 4660 <_sk_callback_sse41+0x307>
   DB  68,15,40,254                        ; movaps        %xmm6,%xmm15
   DB  69,15,89,254                        ; mulps         %xmm14,%xmm15
   DB  69,15,88,251                        ; addps         %xmm11,%xmm15
@@ -12170,7 +12185,7 @@ _sk_luminosity_sse41 LABEL PROC
   DB  15,40,244                           ; movaps        %xmm4,%xmm6
   DB  15,40,235                           ; movaps        %xmm3,%xmm5
   DB  69,15,87,228                        ; xorps         %xmm12,%xmm12
-  DB  68,15,40,45,198,49,0,0              ; movaps        0x31c6(%rip),%xmm13        # 4640 <_sk_callback_sse41+0x30b>
+  DB  68,15,40,45,246,49,0,0              ; movaps        0x31f6(%rip),%xmm13        # 4670 <_sk_callback_sse41+0x317>
   DB  69,15,40,197                        ; movaps        %xmm13,%xmm8
   DB  68,15,94,199                        ; divps         %xmm7,%xmm8
   DB  15,40,223                           ; movaps        %xmm7,%xmm3
@@ -12181,12 +12196,12 @@ _sk_luminosity_sse41 LABEL PROC
   DB  68,15,40,219                        ; movaps        %xmm3,%xmm11
   DB  69,15,89,222                        ; mulps         %xmm14,%xmm11
   DB  65,15,89,217                        ; mulps         %xmm9,%xmm3
-  DB  68,15,40,5,166,49,0,0               ; movaps        0x31a6(%rip),%xmm8        # 4650 <_sk_callback_sse41+0x31b>
+  DB  68,15,40,5,214,49,0,0               ; movaps        0x31d6(%rip),%xmm8        # 4680 <_sk_callback_sse41+0x327>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
-  DB  68,15,40,13,170,49,0,0              ; movaps        0x31aa(%rip),%xmm9        # 4660 <_sk_callback_sse41+0x32b>
+  DB  68,15,40,13,218,49,0,0              ; movaps        0x31da(%rip),%xmm9        # 4690 <_sk_callback_sse41+0x337>
   DB  65,15,89,201                        ; mulps         %xmm9,%xmm1
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  68,15,40,53,171,49,0,0              ; movaps        0x31ab(%rip),%xmm14        # 4670 <_sk_callback_sse41+0x33b>
+  DB  68,15,40,53,219,49,0,0              ; movaps        0x31db(%rip),%xmm14        # 46a0 <_sk_callback_sse41+0x347>
   DB  65,15,89,214                        ; mulps         %xmm14,%xmm2
   DB  15,88,209                           ; addps         %xmm1,%xmm2
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
@@ -12296,7 +12311,7 @@ _sk_clamp_0_sse41 LABEL PROC
 
 PUBLIC _sk_clamp_1_sse41
 _sk_clamp_1_sse41 LABEL PROC
-  DB  68,15,40,5,34,48,0,0                ; movaps        0x3022(%rip),%xmm8        # 4680 <_sk_callback_sse41+0x34b>
+  DB  68,15,40,5,82,48,0,0                ; movaps        0x3052(%rip),%xmm8        # 46b0 <_sk_callback_sse41+0x357>
   DB  65,15,93,192                        ; minps         %xmm8,%xmm0
   DB  65,15,93,200                        ; minps         %xmm8,%xmm1
   DB  65,15,93,208                        ; minps         %xmm8,%xmm2
@@ -12306,7 +12321,7 @@ _sk_clamp_1_sse41 LABEL PROC
 
 PUBLIC _sk_clamp_a_sse41
 _sk_clamp_a_sse41 LABEL PROC
-  DB  15,93,29,23,48,0,0                  ; minps         0x3017(%rip),%xmm3        # 4690 <_sk_callback_sse41+0x35b>
+  DB  15,93,29,71,48,0,0                  ; minps         0x3047(%rip),%xmm3        # 46c0 <_sk_callback_sse41+0x367>
   DB  15,93,195                           ; minps         %xmm3,%xmm0
   DB  15,93,203                           ; minps         %xmm3,%xmm1
   DB  15,93,211                           ; minps         %xmm3,%xmm2
@@ -12379,7 +12394,7 @@ _sk_premul_sse41 LABEL PROC
 PUBLIC _sk_unpremul_sse41
 _sk_unpremul_sse41 LABEL PROC
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
-  DB  68,15,40,13,130,47,0,0              ; movaps        0x2f82(%rip),%xmm9        # 46a0 <_sk_callback_sse41+0x36b>
+  DB  68,15,40,13,178,47,0,0              ; movaps        0x2fb2(%rip),%xmm9        # 46d0 <_sk_callback_sse41+0x377>
   DB  68,15,94,203                        ; divps         %xmm3,%xmm9
   DB  68,15,194,195,4                     ; cmpneqps      %xmm3,%xmm8
   DB  69,15,84,193                        ; andps         %xmm9,%xmm8
@@ -12391,20 +12406,20 @@ _sk_unpremul_sse41 LABEL PROC
 
 PUBLIC _sk_from_srgb_sse41
 _sk_from_srgb_sse41 LABEL PROC
-  DB  68,15,40,29,109,47,0,0              ; movaps        0x2f6d(%rip),%xmm11        # 46b0 <_sk_callback_sse41+0x37b>
+  DB  68,15,40,29,157,47,0,0              ; movaps        0x2f9d(%rip),%xmm11        # 46e0 <_sk_callback_sse41+0x387>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
   DB  68,15,40,208                        ; movaps        %xmm0,%xmm10
   DB  69,15,89,210                        ; mulps         %xmm10,%xmm10
-  DB  68,15,40,37,101,47,0,0              ; movaps        0x2f65(%rip),%xmm12        # 46c0 <_sk_callback_sse41+0x38b>
+  DB  68,15,40,37,149,47,0,0              ; movaps        0x2f95(%rip),%xmm12        # 46f0 <_sk_callback_sse41+0x397>
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,89,196                        ; mulps         %xmm12,%xmm8
-  DB  68,15,40,45,101,47,0,0              ; movaps        0x2f65(%rip),%xmm13        # 46d0 <_sk_callback_sse41+0x39b>
+  DB  68,15,40,45,149,47,0,0              ; movaps        0x2f95(%rip),%xmm13        # 4700 <_sk_callback_sse41+0x3a7>
   DB  69,15,88,197                        ; addps         %xmm13,%xmm8
   DB  69,15,89,194                        ; mulps         %xmm10,%xmm8
-  DB  68,15,40,53,101,47,0,0              ; movaps        0x2f65(%rip),%xmm14        # 46e0 <_sk_callback_sse41+0x3ab>
+  DB  68,15,40,53,149,47,0,0              ; movaps        0x2f95(%rip),%xmm14        # 4710 <_sk_callback_sse41+0x3b7>
   DB  69,15,88,198                        ; addps         %xmm14,%xmm8
-  DB  68,15,40,61,105,47,0,0              ; movaps        0x2f69(%rip),%xmm15        # 46f0 <_sk_callback_sse41+0x3bb>
+  DB  68,15,40,61,153,47,0,0              ; movaps        0x2f99(%rip),%xmm15        # 4720 <_sk_callback_sse41+0x3c7>
   DB  65,15,194,199,1                     ; cmpltps       %xmm15,%xmm0
   DB  102,69,15,56,20,193                 ; blendvps      %xmm0,%xmm9,%xmm8
   DB  68,15,40,209                        ; movaps        %xmm1,%xmm10
@@ -12448,20 +12463,20 @@ _sk_to_srgb_sse41 LABEL PROC
   DB  68,15,82,192                        ; rsqrtps       %xmm0,%xmm8
   DB  69,15,83,200                        ; rcpps         %xmm8,%xmm9
   DB  69,15,82,208                        ; rsqrtps       %xmm8,%xmm10
-  DB  68,15,40,29,214,46,0,0              ; movaps        0x2ed6(%rip),%xmm11        # 4700 <_sk_callback_sse41+0x3cb>
+  DB  68,15,40,29,6,47,0,0                ; movaps        0x2f06(%rip),%xmm11        # 4730 <_sk_callback_sse41+0x3d7>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
-  DB  68,15,40,37,215,46,0,0              ; movaps        0x2ed7(%rip),%xmm12        # 4710 <_sk_callback_sse41+0x3db>
+  DB  68,15,40,37,7,47,0,0                ; movaps        0x2f07(%rip),%xmm12        # 4740 <_sk_callback_sse41+0x3e7>
   DB  69,15,89,204                        ; mulps         %xmm12,%xmm9
-  DB  68,15,40,45,219,46,0,0              ; movaps        0x2edb(%rip),%xmm13        # 4720 <_sk_callback_sse41+0x3eb>
+  DB  68,15,40,45,11,47,0,0               ; movaps        0x2f0b(%rip),%xmm13        # 4750 <_sk_callback_sse41+0x3f7>
   DB  69,15,88,205                        ; addps         %xmm13,%xmm9
-  DB  68,15,40,53,223,46,0,0              ; movaps        0x2edf(%rip),%xmm14        # 4730 <_sk_callback_sse41+0x3fb>
+  DB  68,15,40,53,15,47,0,0               ; movaps        0x2f0f(%rip),%xmm14        # 4760 <_sk_callback_sse41+0x407>
   DB  69,15,89,214                        ; mulps         %xmm14,%xmm10
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
-  DB  68,15,40,5,223,46,0,0               ; movaps        0x2edf(%rip),%xmm8        # 4740 <_sk_callback_sse41+0x40b>
+  DB  68,15,40,5,15,47,0,0                ; movaps        0x2f0f(%rip),%xmm8        # 4770 <_sk_callback_sse41+0x417>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,93,202                        ; minps         %xmm10,%xmm9
-  DB  68,15,40,61,223,46,0,0              ; movaps        0x2edf(%rip),%xmm15        # 4750 <_sk_callback_sse41+0x41b>
+  DB  68,15,40,61,15,47,0,0               ; movaps        0x2f0f(%rip),%xmm15        # 4780 <_sk_callback_sse41+0x427>
   DB  65,15,194,199,1                     ; cmpltps       %xmm15,%xmm0
   DB  102,68,15,56,20,201                 ; blendvps      %xmm0,%xmm1,%xmm9
   DB  15,82,194                           ; rsqrtps       %xmm2,%xmm0
@@ -12514,7 +12529,7 @@ _sk_rgb_to_hsl_sse41 LABEL PROC
   DB  68,15,93,226                        ; minps         %xmm2,%xmm12
   DB  65,15,40,203                        ; movaps        %xmm11,%xmm1
   DB  65,15,92,204                        ; subps         %xmm12,%xmm1
-  DB  68,15,40,53,45,46,0,0               ; movaps        0x2e2d(%rip),%xmm14        # 4760 <_sk_callback_sse41+0x42b>
+  DB  68,15,40,53,93,46,0,0               ; movaps        0x2e5d(%rip),%xmm14        # 4790 <_sk_callback_sse41+0x437>
   DB  68,15,94,241                        ; divps         %xmm1,%xmm14
   DB  69,15,40,211                        ; movaps        %xmm11,%xmm10
   DB  69,15,194,208,0                     ; cmpeqps       %xmm8,%xmm10
@@ -12523,27 +12538,27 @@ _sk_rgb_to_hsl_sse41 LABEL PROC
   DB  65,15,89,198                        ; mulps         %xmm14,%xmm0
   DB  69,15,40,249                        ; movaps        %xmm9,%xmm15
   DB  68,15,194,250,1                     ; cmpltps       %xmm2,%xmm15
-  DB  68,15,84,61,20,46,0,0               ; andps         0x2e14(%rip),%xmm15        # 4770 <_sk_callback_sse41+0x43b>
+  DB  68,15,84,61,68,46,0,0               ; andps         0x2e44(%rip),%xmm15        # 47a0 <_sk_callback_sse41+0x447>
   DB  68,15,88,248                        ; addps         %xmm0,%xmm15
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  65,15,194,193,0                     ; cmpeqps       %xmm9,%xmm0
   DB  65,15,92,208                        ; subps         %xmm8,%xmm2
   DB  65,15,89,214                        ; mulps         %xmm14,%xmm2
-  DB  68,15,40,45,7,46,0,0                ; movaps        0x2e07(%rip),%xmm13        # 4780 <_sk_callback_sse41+0x44b>
+  DB  68,15,40,45,55,46,0,0               ; movaps        0x2e37(%rip),%xmm13        # 47b0 <_sk_callback_sse41+0x457>
   DB  65,15,88,213                        ; addps         %xmm13,%xmm2
   DB  69,15,92,193                        ; subps         %xmm9,%xmm8
   DB  69,15,89,198                        ; mulps         %xmm14,%xmm8
-  DB  68,15,88,5,3,46,0,0                 ; addps         0x2e03(%rip),%xmm8        # 4790 <_sk_callback_sse41+0x45b>
+  DB  68,15,88,5,51,46,0,0                ; addps         0x2e33(%rip),%xmm8        # 47c0 <_sk_callback_sse41+0x467>
   DB  102,68,15,56,20,194                 ; blendvps      %xmm0,%xmm2,%xmm8
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  102,69,15,56,20,199                 ; blendvps      %xmm0,%xmm15,%xmm8
-  DB  68,15,89,5,251,45,0,0               ; mulps         0x2dfb(%rip),%xmm8        # 47a0 <_sk_callback_sse41+0x46b>
+  DB  68,15,89,5,43,46,0,0                ; mulps         0x2e2b(%rip),%xmm8        # 47d0 <_sk_callback_sse41+0x477>
   DB  69,15,40,203                        ; movaps        %xmm11,%xmm9
   DB  69,15,194,204,4                     ; cmpneqps      %xmm12,%xmm9
   DB  69,15,84,193                        ; andps         %xmm9,%xmm8
   DB  69,15,92,235                        ; subps         %xmm11,%xmm13
   DB  69,15,88,220                        ; addps         %xmm12,%xmm11
-  DB  15,40,5,239,45,0,0                  ; movaps        0x2def(%rip),%xmm0        # 47b0 <_sk_callback_sse41+0x47b>
+  DB  15,40,5,31,46,0,0                   ; movaps        0x2e1f(%rip),%xmm0        # 47e0 <_sk_callback_sse41+0x487>
   DB  65,15,40,211                        ; movaps        %xmm11,%xmm2
   DB  15,89,208                           ; mulps         %xmm0,%xmm2
   DB  15,194,194,1                        ; cmpltps       %xmm2,%xmm0
@@ -12564,7 +12579,7 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  15,41,100,36,32                     ; movaps        %xmm4,0x20(%rsp)
   DB  15,41,92,36,16                      ; movaps        %xmm3,0x10(%rsp)
   DB  68,15,40,208                        ; movaps        %xmm0,%xmm10
-  DB  68,15,40,13,177,45,0,0              ; movaps        0x2db1(%rip),%xmm9        # 47c0 <_sk_callback_sse41+0x48b>
+  DB  68,15,40,13,225,45,0,0              ; movaps        0x2de1(%rip),%xmm9        # 47f0 <_sk_callback_sse41+0x497>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  15,194,194,2                        ; cmpleps       %xmm2,%xmm0
   DB  15,40,217                           ; movaps        %xmm1,%xmm3
@@ -12577,19 +12592,19 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  15,41,20,36                         ; movaps        %xmm2,(%rsp)
   DB  69,15,88,192                        ; addps         %xmm8,%xmm8
   DB  68,15,92,197                        ; subps         %xmm5,%xmm8
-  DB  68,15,40,53,141,45,0,0              ; movaps        0x2d8d(%rip),%xmm14        # 47d0 <_sk_callback_sse41+0x49b>
+  DB  68,15,40,53,189,45,0,0              ; movaps        0x2dbd(%rip),%xmm14        # 4800 <_sk_callback_sse41+0x4a7>
   DB  69,15,88,242                        ; addps         %xmm10,%xmm14
   DB  102,65,15,58,8,198,1                ; roundps       $0x1,%xmm14,%xmm0
   DB  68,15,92,240                        ; subps         %xmm0,%xmm14
-  DB  68,15,40,29,134,45,0,0              ; movaps        0x2d86(%rip),%xmm11        # 47e0 <_sk_callback_sse41+0x4ab>
+  DB  68,15,40,29,182,45,0,0              ; movaps        0x2db6(%rip),%xmm11        # 4810 <_sk_callback_sse41+0x4b7>
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  65,15,194,198,2                     ; cmpleps       %xmm14,%xmm0
   DB  15,40,245                           ; movaps        %xmm5,%xmm6
   DB  65,15,92,240                        ; subps         %xmm8,%xmm6
-  DB  15,40,61,127,45,0,0                 ; movaps        0x2d7f(%rip),%xmm7        # 47f0 <_sk_callback_sse41+0x4bb>
+  DB  15,40,61,175,45,0,0                 ; movaps        0x2daf(%rip),%xmm7        # 4820 <_sk_callback_sse41+0x4c7>
   DB  69,15,40,238                        ; movaps        %xmm14,%xmm13
   DB  68,15,89,239                        ; mulps         %xmm7,%xmm13
-  DB  15,40,29,128,45,0,0                 ; movaps        0x2d80(%rip),%xmm3        # 4800 <_sk_callback_sse41+0x4cb>
+  DB  15,40,29,176,45,0,0                 ; movaps        0x2db0(%rip),%xmm3        # 4830 <_sk_callback_sse41+0x4d7>
   DB  68,15,40,227                        ; movaps        %xmm3,%xmm12
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  68,15,89,230                        ; mulps         %xmm6,%xmm12
@@ -12599,7 +12614,7 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  65,15,194,198,2                     ; cmpleps       %xmm14,%xmm0
   DB  68,15,40,253                        ; movaps        %xmm5,%xmm15
   DB  102,69,15,56,20,252                 ; blendvps      %xmm0,%xmm12,%xmm15
-  DB  68,15,40,37,95,45,0,0               ; movaps        0x2d5f(%rip),%xmm12        # 4810 <_sk_callback_sse41+0x4db>
+  DB  68,15,40,37,143,45,0,0              ; movaps        0x2d8f(%rip),%xmm12        # 4840 <_sk_callback_sse41+0x4e7>
   DB  65,15,40,196                        ; movaps        %xmm12,%xmm0
   DB  65,15,194,198,2                     ; cmpleps       %xmm14,%xmm0
   DB  68,15,89,238                        ; mulps         %xmm6,%xmm13
@@ -12633,7 +12648,7 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  65,15,40,198                        ; movaps        %xmm14,%xmm0
   DB  15,40,20,36                         ; movaps        (%rsp),%xmm2
   DB  102,15,56,20,202                    ; blendvps      %xmm0,%xmm2,%xmm1
-  DB  68,15,88,21,216,44,0,0              ; addps         0x2cd8(%rip),%xmm10        # 4820 <_sk_callback_sse41+0x4eb>
+  DB  68,15,88,21,8,45,0,0                ; addps         0x2d08(%rip),%xmm10        # 4850 <_sk_callback_sse41+0x4f7>
   DB  102,65,15,58,8,194,1                ; roundps       $0x1,%xmm10,%xmm0
   DB  68,15,92,208                        ; subps         %xmm0,%xmm10
   DB  69,15,194,218,2                     ; cmpleps       %xmm10,%xmm11
@@ -12682,7 +12697,7 @@ _sk_scale_u8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,68,15,56,49,4,56                ; pmovzxbd      (%rax,%rdi,1),%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,49,44,0,0                ; mulps         0x2c31(%rip),%xmm8        # 4830 <_sk_callback_sse41+0x4fb>
+  DB  68,15,89,5,97,44,0,0                ; mulps         0x2c61(%rip),%xmm8        # 4860 <_sk_callback_sse41+0x507>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
@@ -12716,7 +12731,7 @@ _sk_lerp_u8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,68,15,56,49,4,56                ; pmovzxbd      (%rax,%rdi,1),%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,221,43,0,0               ; mulps         0x2bdd(%rip),%xmm8        # 4840 <_sk_callback_sse41+0x50b>
+  DB  68,15,89,5,13,44,0,0                ; mulps         0x2c0d(%rip),%xmm8        # 4870 <_sk_callback_sse41+0x517>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -12736,29 +12751,38 @@ PUBLIC _sk_lerp_565_sse41
 _sk_lerp_565_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  102,68,15,56,51,4,120               ; pmovzxwd      (%rax,%rdi,2),%xmm8
-  DB  102,15,111,29,173,43,0,0            ; movdqa        0x2bad(%rip),%xmm3        # 4850 <_sk_callback_sse41+0x51b>
-  DB  102,65,15,219,216                   ; pand          %xmm8,%xmm3
-  DB  68,15,91,203                        ; cvtdq2ps      %xmm3,%xmm9
-  DB  68,15,89,13,172,43,0,0              ; mulps         0x2bac(%rip),%xmm9        # 4860 <_sk_callback_sse41+0x52b>
-  DB  102,15,111,29,180,43,0,0            ; movdqa        0x2bb4(%rip),%xmm3        # 4870 <_sk_callback_sse41+0x53b>
-  DB  102,65,15,219,216                   ; pand          %xmm8,%xmm3
-  DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,181,43,0,0                 ; mulps         0x2bb5(%rip),%xmm3        # 4880 <_sk_callback_sse41+0x54b>
-  DB  102,68,15,219,5,188,43,0,0          ; pand          0x2bbc(%rip),%xmm8        # 4890 <_sk_callback_sse41+0x55b>
+  DB  102,68,15,56,51,20,120              ; pmovzxwd      (%rax,%rdi,2),%xmm10
+  DB  102,68,15,111,5,220,43,0,0          ; movdqa        0x2bdc(%rip),%xmm8        # 4880 <_sk_callback_sse41+0x527>
+  DB  102,69,15,219,194                   ; pand          %xmm10,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,192,43,0,0               ; mulps         0x2bc0(%rip),%xmm8        # 48a0 <_sk_callback_sse41+0x56b>
+  DB  68,15,89,5,219,43,0,0               ; mulps         0x2bdb(%rip),%xmm8        # 4890 <_sk_callback_sse41+0x537>
+  DB  102,68,15,111,13,226,43,0,0         ; movdqa        0x2be2(%rip),%xmm9        # 48a0 <_sk_callback_sse41+0x547>
+  DB  102,69,15,219,202                   ; pand          %xmm10,%xmm9
+  DB  69,15,91,201                        ; cvtdq2ps      %xmm9,%xmm9
+  DB  68,15,89,13,225,43,0,0              ; mulps         0x2be1(%rip),%xmm9        # 48b0 <_sk_callback_sse41+0x557>
+  DB  102,68,15,219,21,232,43,0,0         ; pand          0x2be8(%rip),%xmm10        # 48c0 <_sk_callback_sse41+0x567>
+  DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
+  DB  68,15,89,21,236,43,0,0              ; mulps         0x2bec(%rip),%xmm10        # 48d0 <_sk_callback_sse41+0x577>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
-  DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
+  DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
   DB  15,92,205                           ; subps         %xmm5,%xmm1
-  DB  15,89,203                           ; mulps         %xmm3,%xmm1
+  DB  65,15,89,201                        ; mulps         %xmm9,%xmm1
   DB  15,88,205                           ; addps         %xmm5,%xmm1
   DB  15,92,214                           ; subps         %xmm6,%xmm2
-  DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
+  DB  65,15,89,210                        ; mulps         %xmm10,%xmm2
   DB  15,88,214                           ; addps         %xmm6,%xmm2
+  DB  15,92,223                           ; subps         %xmm7,%xmm3
+  DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
+  DB  68,15,88,199                        ; addps         %xmm7,%xmm8
+  DB  68,15,89,203                        ; mulps         %xmm3,%xmm9
+  DB  68,15,88,207                        ; addps         %xmm7,%xmm9
+  DB  65,15,89,218                        ; mulps         %xmm10,%xmm3
+  DB  15,88,223                           ; addps         %xmm7,%xmm3
+  DB  68,15,95,203                        ; maxps         %xmm3,%xmm9
+  DB  69,15,95,193                        ; maxps         %xmm9,%xmm8
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,170,43,0,0                 ; movaps        0x2baa(%rip),%xmm3        # 48b0 <_sk_callback_sse41+0x57b>
+  DB  65,15,40,216                        ; movaps        %xmm8,%xmm3
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_load_tables_sse41
@@ -12767,7 +12791,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,139,72,8                         ; mov           0x8(%rax),%r9
   DB  243,69,15,111,4,184                 ; movdqu        (%r8,%rdi,4),%xmm8
-  DB  102,15,111,5,161,43,0,0             ; movdqa        0x2ba1(%rip),%xmm0        # 48c0 <_sk_callback_sse41+0x58b>
+  DB  102,15,111,5,157,43,0,0             ; movdqa        0x2b9d(%rip),%xmm0        # 48e0 <_sk_callback_sse41+0x587>
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,73,15,58,22,192,1               ; pextrq        $0x1,%xmm0,%r8
   DB  102,72,15,126,193                   ; movq          %xmm0,%rcx
@@ -12782,7 +12806,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  102,15,58,33,193,48                 ; insertps      $0x30,%xmm1,%xmm0
   DB  76,139,64,16                        ; mov           0x10(%rax),%r8
   DB  102,65,15,111,200                   ; movdqa        %xmm8,%xmm1
-  DB  102,15,56,0,13,92,43,0,0            ; pshufb        0x2b5c(%rip),%xmm1        # 48d0 <_sk_callback_sse41+0x59b>
+  DB  102,15,56,0,13,88,43,0,0            ; pshufb        0x2b58(%rip),%xmm1        # 48f0 <_sk_callback_sse41+0x597>
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
   DB  102,72,15,126,201                   ; movq          %xmm1,%rcx
   DB  68,15,182,209                       ; movzbl        %cl,%r10d
@@ -12797,7 +12821,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  102,15,58,33,202,48                 ; insertps      $0x30,%xmm2,%xmm1
   DB  76,139,64,24                        ; mov           0x18(%rax),%r8
   DB  102,65,15,111,208                   ; movdqa        %xmm8,%xmm2
-  DB  102,15,56,0,21,24,43,0,0            ; pshufb        0x2b18(%rip),%xmm2        # 48e0 <_sk_callback_sse41+0x5ab>
+  DB  102,15,56,0,21,20,43,0,0            ; pshufb        0x2b14(%rip),%xmm2        # 4900 <_sk_callback_sse41+0x5a7>
   DB  102,72,15,58,22,209,1               ; pextrq        $0x1,%xmm2,%rcx
   DB  102,72,15,126,208                   ; movq          %xmm2,%rax
   DB  68,15,182,200                       ; movzbl        %al,%r9d
@@ -12812,7 +12836,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  102,15,58,33,211,48                 ; insertps      $0x30,%xmm3,%xmm2
   DB  102,65,15,114,208,24                ; psrld         $0x18,%xmm8
   DB  65,15,91,216                        ; cvtdq2ps      %xmm8,%xmm3
-  DB  15,89,29,213,42,0,0                 ; mulps         0x2ad5(%rip),%xmm3        # 48f0 <_sk_callback_sse41+0x5bb>
+  DB  15,89,29,209,42,0,0                 ; mulps         0x2ad1(%rip),%xmm3        # 4910 <_sk_callback_sse41+0x5b7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -12829,7 +12853,7 @@ _sk_load_tables_u16_be_sse41 LABEL PROC
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,97,200                       ; punpcklwd     %xmm0,%xmm1
   DB  102,68,15,105,200                   ; punpckhwd     %xmm0,%xmm9
-  DB  102,68,15,111,5,168,42,0,0          ; movdqa        0x2aa8(%rip),%xmm8        # 4900 <_sk_callback_sse41+0x5cb>
+  DB  102,68,15,111,5,164,42,0,0          ; movdqa        0x2aa4(%rip),%xmm8        # 4920 <_sk_callback_sse41+0x5c7>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,15,56,51,192                    ; pmovzxwd      %xmm0,%xmm0
@@ -12846,7 +12870,7 @@ _sk_load_tables_u16_be_sse41 LABEL PROC
   DB  243,67,15,16,20,8                   ; movss         (%r8,%r9,1),%xmm2
   DB  102,15,58,33,194,48                 ; insertps      $0x30,%xmm2,%xmm0
   DB  76,139,64,16                        ; mov           0x10(%rax),%r8
-  DB  102,15,56,0,13,91,42,0,0            ; pshufb        0x2a5b(%rip),%xmm1        # 4910 <_sk_callback_sse41+0x5db>
+  DB  102,15,56,0,13,87,42,0,0            ; pshufb        0x2a57(%rip),%xmm1        # 4930 <_sk_callback_sse41+0x5d7>
   DB  102,15,56,51,201                    ; pmovzxwd      %xmm1,%xmm1
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
   DB  102,72,15,126,201                   ; movq          %xmm1,%rcx
@@ -12882,7 +12906,7 @@ _sk_load_tables_u16_be_sse41 LABEL PROC
   DB  102,65,15,235,216                   ; por           %xmm8,%xmm3
   DB  102,15,56,51,219                    ; pmovzxwd      %xmm3,%xmm3
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,169,41,0,0                 ; mulps         0x29a9(%rip),%xmm3        # 4920 <_sk_callback_sse41+0x5eb>
+  DB  15,89,29,165,41,0,0                 ; mulps         0x29a5(%rip),%xmm3        # 4940 <_sk_callback_sse41+0x5e7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -12902,7 +12926,7 @@ _sk_load_tables_rgb_u16_be_sse41 LABEL PROC
   DB  102,68,15,97,200                    ; punpcklwd     %xmm0,%xmm9
   DB  102,15,111,202                      ; movdqa        %xmm2,%xmm1
   DB  102,65,15,97,201                    ; punpcklwd     %xmm9,%xmm1
-  DB  102,68,15,111,5,107,41,0,0          ; movdqa        0x296b(%rip),%xmm8        # 4930 <_sk_callback_sse41+0x5fb>
+  DB  102,68,15,111,5,103,41,0,0          ; movdqa        0x2967(%rip),%xmm8        # 4950 <_sk_callback_sse41+0x5f7>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,15,56,51,192                    ; pmovzxwd      %xmm0,%xmm0
@@ -12919,7 +12943,7 @@ _sk_load_tables_rgb_u16_be_sse41 LABEL PROC
   DB  243,67,15,16,28,8                   ; movss         (%r8,%r9,1),%xmm3
   DB  102,15,58,33,195,48                 ; insertps      $0x30,%xmm3,%xmm0
   DB  76,139,64,16                        ; mov           0x10(%rax),%r8
-  DB  102,15,56,0,13,30,41,0,0            ; pshufb        0x291e(%rip),%xmm1        # 4940 <_sk_callback_sse41+0x60b>
+  DB  102,15,56,0,13,26,41,0,0            ; pshufb        0x291a(%rip),%xmm1        # 4960 <_sk_callback_sse41+0x607>
   DB  102,15,56,51,201                    ; pmovzxwd      %xmm1,%xmm1
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
   DB  102,72,15,126,201                   ; movq          %xmm1,%rcx
@@ -12950,7 +12974,7 @@ _sk_load_tables_rgb_u16_be_sse41 LABEL PROC
   DB  243,65,15,16,28,8                   ; movss         (%r8,%rcx,1),%xmm3
   DB  102,15,58,33,211,48                 ; insertps      $0x30,%xmm3,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,137,40,0,0                 ; movaps        0x2889(%rip),%xmm3        # 4950 <_sk_callback_sse41+0x61b>
+  DB  15,40,29,133,40,0,0                 ; movaps        0x2885(%rip),%xmm3        # 4970 <_sk_callback_sse41+0x617>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_byte_tables_sse41
@@ -12958,7 +12982,7 @@ _sk_byte_tables_sse41 LABEL PROC
   DB  65,86                               ; push          %r14
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,138,40,0,0               ; movaps        0x288a(%rip),%xmm8        # 4960 <_sk_callback_sse41+0x62b>
+  DB  68,15,40,5,134,40,0,0               ; movaps        0x2886(%rip),%xmm8        # 4980 <_sk_callback_sse41+0x627>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,91,192                       ; cvtps2dq      %xmm0,%xmm0
   DB  102,72,15,58,22,193,1               ; pextrq        $0x1,%xmm0,%rcx
@@ -12977,7 +13001,7 @@ _sk_byte_tables_sse41 LABEL PROC
   DB  102,15,58,32,193,3                  ; pinsrb        $0x3,%ecx,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,13,59,40,0,0               ; movaps        0x283b(%rip),%xmm9        # 4970 <_sk_callback_sse41+0x63b>
+  DB  68,15,40,13,55,40,0,0               ; movaps        0x2837(%rip),%xmm9        # 4990 <_sk_callback_sse41+0x637>
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -13066,7 +13090,7 @@ _sk_byte_tables_rgb_sse41 LABEL PROC
   DB  102,15,58,32,193,3                  ; pinsrb        $0x3,%ecx,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,13,195,38,0,0              ; movaps        0x26c3(%rip),%xmm9        # 4980 <_sk_callback_sse41+0x64b>
+  DB  68,15,40,13,191,38,0,0              ; movaps        0x26bf(%rip),%xmm9        # 49a0 <_sk_callback_sse41+0x647>
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -13233,31 +13257,31 @@ _sk_parametric_r_sse41 LABEL PROC
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,194                        ; cvtdq2ps      %xmm10,%xmm8
-  DB  68,15,89,5,26,36,0,0                ; mulps         0x241a(%rip),%xmm8        # 4990 <_sk_callback_sse41+0x65b>
-  DB  68,15,84,21,34,36,0,0               ; andps         0x2422(%rip),%xmm10        # 49a0 <_sk_callback_sse41+0x66b>
-  DB  68,15,86,21,42,36,0,0               ; orps          0x242a(%rip),%xmm10        # 49b0 <_sk_callback_sse41+0x67b>
-  DB  68,15,88,5,50,36,0,0                ; addps         0x2432(%rip),%xmm8        # 49c0 <_sk_callback_sse41+0x68b>
-  DB  68,15,40,37,58,36,0,0               ; movaps        0x243a(%rip),%xmm12        # 49d0 <_sk_callback_sse41+0x69b>
+  DB  68,15,89,5,22,36,0,0                ; mulps         0x2416(%rip),%xmm8        # 49b0 <_sk_callback_sse41+0x657>
+  DB  68,15,84,21,30,36,0,0               ; andps         0x241e(%rip),%xmm10        # 49c0 <_sk_callback_sse41+0x667>
+  DB  68,15,86,21,38,36,0,0               ; orps          0x2426(%rip),%xmm10        # 49d0 <_sk_callback_sse41+0x677>
+  DB  68,15,88,5,46,36,0,0                ; addps         0x242e(%rip),%xmm8        # 49e0 <_sk_callback_sse41+0x687>
+  DB  68,15,40,37,54,36,0,0               ; movaps        0x2436(%rip),%xmm12        # 49f0 <_sk_callback_sse41+0x697>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,196                        ; subps         %xmm12,%xmm8
-  DB  68,15,88,21,58,36,0,0               ; addps         0x243a(%rip),%xmm10        # 49e0 <_sk_callback_sse41+0x6ab>
-  DB  68,15,40,37,66,36,0,0               ; movaps        0x2442(%rip),%xmm12        # 49f0 <_sk_callback_sse41+0x6bb>
+  DB  68,15,88,21,54,36,0,0               ; addps         0x2436(%rip),%xmm10        # 4a00 <_sk_callback_sse41+0x6a7>
+  DB  68,15,40,37,62,36,0,0               ; movaps        0x243e(%rip),%xmm12        # 4a10 <_sk_callback_sse41+0x6b7>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,196                        ; subps         %xmm12,%xmm8
   DB  69,15,89,195                        ; mulps         %xmm11,%xmm8
   DB  102,69,15,58,8,208,1                ; roundps       $0x1,%xmm8,%xmm10
   DB  69,15,40,216                        ; movaps        %xmm8,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,5,47,36,0,0                ; addps         0x242f(%rip),%xmm8        # 4a00 <_sk_callback_sse41+0x6cb>
-  DB  68,15,40,21,55,36,0,0               ; movaps        0x2437(%rip),%xmm10        # 4a10 <_sk_callback_sse41+0x6db>
+  DB  68,15,88,5,43,36,0,0                ; addps         0x242b(%rip),%xmm8        # 4a20 <_sk_callback_sse41+0x6c7>
+  DB  68,15,40,21,51,36,0,0               ; movaps        0x2433(%rip),%xmm10        # 4a30 <_sk_callback_sse41+0x6d7>
   DB  69,15,89,211                        ; mulps         %xmm11,%xmm10
   DB  69,15,92,194                        ; subps         %xmm10,%xmm8
-  DB  68,15,40,21,55,36,0,0               ; movaps        0x2437(%rip),%xmm10        # 4a20 <_sk_callback_sse41+0x6eb>
+  DB  68,15,40,21,51,36,0,0               ; movaps        0x2433(%rip),%xmm10        # 4a40 <_sk_callback_sse41+0x6e7>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  68,15,40,29,59,36,0,0               ; movaps        0x243b(%rip),%xmm11        # 4a30 <_sk_callback_sse41+0x6fb>
+  DB  68,15,40,29,55,36,0,0               ; movaps        0x2437(%rip),%xmm11        # 4a50 <_sk_callback_sse41+0x6f7>
   DB  69,15,94,218                        ; divps         %xmm10,%xmm11
   DB  69,15,88,216                        ; addps         %xmm8,%xmm11
-  DB  68,15,89,29,59,36,0,0               ; mulps         0x243b(%rip),%xmm11        # 4a40 <_sk_callback_sse41+0x70b>
+  DB  68,15,89,29,55,36,0,0               ; mulps         0x2437(%rip),%xmm11        # 4a60 <_sk_callback_sse41+0x707>
   DB  102,69,15,91,211                    ; cvtps2dq      %xmm11,%xmm10
   DB  243,68,15,16,64,20                  ; movss         0x14(%rax),%xmm8
   DB  69,15,198,192,0                     ; shufps        $0x0,%xmm8,%xmm8
@@ -13265,7 +13289,7 @@ _sk_parametric_r_sse41 LABEL PROC
   DB  102,69,15,56,20,193                 ; blendvps      %xmm0,%xmm9,%xmm8
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  68,15,95,192                        ; maxps         %xmm0,%xmm8
-  DB  68,15,93,5,34,36,0,0                ; minps         0x2422(%rip),%xmm8        # 4a50 <_sk_callback_sse41+0x71b>
+  DB  68,15,93,5,30,36,0,0                ; minps         0x241e(%rip),%xmm8        # 4a70 <_sk_callback_sse41+0x717>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -13293,31 +13317,31 @@ _sk_parametric_g_sse41 LABEL PROC
   DB  68,15,88,217                        ; addps         %xmm1,%xmm11
   DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
   DB  69,15,91,227                        ; cvtdq2ps      %xmm11,%xmm12
-  DB  68,15,89,37,195,35,0,0              ; mulps         0x23c3(%rip),%xmm12        # 4a60 <_sk_callback_sse41+0x72b>
-  DB  68,15,84,29,203,35,0,0              ; andps         0x23cb(%rip),%xmm11        # 4a70 <_sk_callback_sse41+0x73b>
-  DB  68,15,86,29,211,35,0,0              ; orps          0x23d3(%rip),%xmm11        # 4a80 <_sk_callback_sse41+0x74b>
-  DB  68,15,88,37,219,35,0,0              ; addps         0x23db(%rip),%xmm12        # 4a90 <_sk_callback_sse41+0x75b>
-  DB  15,40,13,228,35,0,0                 ; movaps        0x23e4(%rip),%xmm1        # 4aa0 <_sk_callback_sse41+0x76b>
+  DB  68,15,89,37,191,35,0,0              ; mulps         0x23bf(%rip),%xmm12        # 4a80 <_sk_callback_sse41+0x727>
+  DB  68,15,84,29,199,35,0,0              ; andps         0x23c7(%rip),%xmm11        # 4a90 <_sk_callback_sse41+0x737>
+  DB  68,15,86,29,207,35,0,0              ; orps          0x23cf(%rip),%xmm11        # 4aa0 <_sk_callback_sse41+0x747>
+  DB  68,15,88,37,215,35,0,0              ; addps         0x23d7(%rip),%xmm12        # 4ab0 <_sk_callback_sse41+0x757>
+  DB  15,40,13,224,35,0,0                 ; movaps        0x23e0(%rip),%xmm1        # 4ac0 <_sk_callback_sse41+0x767>
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
   DB  68,15,92,225                        ; subps         %xmm1,%xmm12
-  DB  68,15,88,29,228,35,0,0              ; addps         0x23e4(%rip),%xmm11        # 4ab0 <_sk_callback_sse41+0x77b>
-  DB  15,40,13,237,35,0,0                 ; movaps        0x23ed(%rip),%xmm1        # 4ac0 <_sk_callback_sse41+0x78b>
+  DB  68,15,88,29,224,35,0,0              ; addps         0x23e0(%rip),%xmm11        # 4ad0 <_sk_callback_sse41+0x777>
+  DB  15,40,13,233,35,0,0                 ; movaps        0x23e9(%rip),%xmm1        # 4ae0 <_sk_callback_sse41+0x787>
   DB  65,15,94,203                        ; divps         %xmm11,%xmm1
   DB  68,15,92,225                        ; subps         %xmm1,%xmm12
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  102,69,15,58,8,212,1                ; roundps       $0x1,%xmm12,%xmm10
   DB  69,15,40,220                        ; movaps        %xmm12,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,37,218,35,0,0              ; addps         0x23da(%rip),%xmm12        # 4ad0 <_sk_callback_sse41+0x79b>
-  DB  15,40,13,227,35,0,0                 ; movaps        0x23e3(%rip),%xmm1        # 4ae0 <_sk_callback_sse41+0x7ab>
+  DB  68,15,88,37,214,35,0,0              ; addps         0x23d6(%rip),%xmm12        # 4af0 <_sk_callback_sse41+0x797>
+  DB  15,40,13,223,35,0,0                 ; movaps        0x23df(%rip),%xmm1        # 4b00 <_sk_callback_sse41+0x7a7>
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
   DB  68,15,92,225                        ; subps         %xmm1,%xmm12
-  DB  68,15,40,21,227,35,0,0              ; movaps        0x23e3(%rip),%xmm10        # 4af0 <_sk_callback_sse41+0x7bb>
+  DB  68,15,40,21,223,35,0,0              ; movaps        0x23df(%rip),%xmm10        # 4b10 <_sk_callback_sse41+0x7b7>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  15,40,13,232,35,0,0                 ; movaps        0x23e8(%rip),%xmm1        # 4b00 <_sk_callback_sse41+0x7cb>
+  DB  15,40,13,228,35,0,0                 ; movaps        0x23e4(%rip),%xmm1        # 4b20 <_sk_callback_sse41+0x7c7>
   DB  65,15,94,202                        ; divps         %xmm10,%xmm1
   DB  65,15,88,204                        ; addps         %xmm12,%xmm1
-  DB  15,89,13,233,35,0,0                 ; mulps         0x23e9(%rip),%xmm1        # 4b10 <_sk_callback_sse41+0x7db>
+  DB  15,89,13,229,35,0,0                 ; mulps         0x23e5(%rip),%xmm1        # 4b30 <_sk_callback_sse41+0x7d7>
   DB  102,68,15,91,209                    ; cvtps2dq      %xmm1,%xmm10
   DB  243,15,16,72,20                     ; movss         0x14(%rax),%xmm1
   DB  15,198,201,0                        ; shufps        $0x0,%xmm1,%xmm1
@@ -13325,7 +13349,7 @@ _sk_parametric_g_sse41 LABEL PROC
   DB  102,65,15,56,20,201                 ; blendvps      %xmm0,%xmm9,%xmm1
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,95,200                           ; maxps         %xmm0,%xmm1
-  DB  15,93,13,212,35,0,0                 ; minps         0x23d4(%rip),%xmm1        # 4b20 <_sk_callback_sse41+0x7eb>
+  DB  15,93,13,208,35,0,0                 ; minps         0x23d0(%rip),%xmm1        # 4b40 <_sk_callback_sse41+0x7e7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -13353,31 +13377,31 @@ _sk_parametric_b_sse41 LABEL PROC
   DB  68,15,88,218                        ; addps         %xmm2,%xmm11
   DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
   DB  69,15,91,227                        ; cvtdq2ps      %xmm11,%xmm12
-  DB  68,15,89,37,117,35,0,0              ; mulps         0x2375(%rip),%xmm12        # 4b30 <_sk_callback_sse41+0x7fb>
-  DB  68,15,84,29,125,35,0,0              ; andps         0x237d(%rip),%xmm11        # 4b40 <_sk_callback_sse41+0x80b>
-  DB  68,15,86,29,133,35,0,0              ; orps          0x2385(%rip),%xmm11        # 4b50 <_sk_callback_sse41+0x81b>
-  DB  68,15,88,37,141,35,0,0              ; addps         0x238d(%rip),%xmm12        # 4b60 <_sk_callback_sse41+0x82b>
-  DB  15,40,21,150,35,0,0                 ; movaps        0x2396(%rip),%xmm2        # 4b70 <_sk_callback_sse41+0x83b>
+  DB  68,15,89,37,113,35,0,0              ; mulps         0x2371(%rip),%xmm12        # 4b50 <_sk_callback_sse41+0x7f7>
+  DB  68,15,84,29,121,35,0,0              ; andps         0x2379(%rip),%xmm11        # 4b60 <_sk_callback_sse41+0x807>
+  DB  68,15,86,29,129,35,0,0              ; orps          0x2381(%rip),%xmm11        # 4b70 <_sk_callback_sse41+0x817>
+  DB  68,15,88,37,137,35,0,0              ; addps         0x2389(%rip),%xmm12        # 4b80 <_sk_callback_sse41+0x827>
+  DB  15,40,21,146,35,0,0                 ; movaps        0x2392(%rip),%xmm2        # 4b90 <_sk_callback_sse41+0x837>
   DB  65,15,89,211                        ; mulps         %xmm11,%xmm2
   DB  68,15,92,226                        ; subps         %xmm2,%xmm12
-  DB  68,15,88,29,150,35,0,0              ; addps         0x2396(%rip),%xmm11        # 4b80 <_sk_callback_sse41+0x84b>
-  DB  15,40,21,159,35,0,0                 ; movaps        0x239f(%rip),%xmm2        # 4b90 <_sk_callback_sse41+0x85b>
+  DB  68,15,88,29,146,35,0,0              ; addps         0x2392(%rip),%xmm11        # 4ba0 <_sk_callback_sse41+0x847>
+  DB  15,40,21,155,35,0,0                 ; movaps        0x239b(%rip),%xmm2        # 4bb0 <_sk_callback_sse41+0x857>
   DB  65,15,94,211                        ; divps         %xmm11,%xmm2
   DB  68,15,92,226                        ; subps         %xmm2,%xmm12
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  102,69,15,58,8,212,1                ; roundps       $0x1,%xmm12,%xmm10
   DB  69,15,40,220                        ; movaps        %xmm12,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,37,140,35,0,0              ; addps         0x238c(%rip),%xmm12        # 4ba0 <_sk_callback_sse41+0x86b>
-  DB  15,40,21,149,35,0,0                 ; movaps        0x2395(%rip),%xmm2        # 4bb0 <_sk_callback_sse41+0x87b>
+  DB  68,15,88,37,136,35,0,0              ; addps         0x2388(%rip),%xmm12        # 4bc0 <_sk_callback_sse41+0x867>
+  DB  15,40,21,145,35,0,0                 ; movaps        0x2391(%rip),%xmm2        # 4bd0 <_sk_callback_sse41+0x877>
   DB  65,15,89,211                        ; mulps         %xmm11,%xmm2
   DB  68,15,92,226                        ; subps         %xmm2,%xmm12
-  DB  68,15,40,21,149,35,0,0              ; movaps        0x2395(%rip),%xmm10        # 4bc0 <_sk_callback_sse41+0x88b>
+  DB  68,15,40,21,145,35,0,0              ; movaps        0x2391(%rip),%xmm10        # 4be0 <_sk_callback_sse41+0x887>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  15,40,21,154,35,0,0                 ; movaps        0x239a(%rip),%xmm2        # 4bd0 <_sk_callback_sse41+0x89b>
+  DB  15,40,21,150,35,0,0                 ; movaps        0x2396(%rip),%xmm2        # 4bf0 <_sk_callback_sse41+0x897>
   DB  65,15,94,210                        ; divps         %xmm10,%xmm2
   DB  65,15,88,212                        ; addps         %xmm12,%xmm2
-  DB  15,89,21,155,35,0,0                 ; mulps         0x239b(%rip),%xmm2        # 4be0 <_sk_callback_sse41+0x8ab>
+  DB  15,89,21,151,35,0,0                 ; mulps         0x2397(%rip),%xmm2        # 4c00 <_sk_callback_sse41+0x8a7>
   DB  102,68,15,91,210                    ; cvtps2dq      %xmm2,%xmm10
   DB  243,15,16,80,20                     ; movss         0x14(%rax),%xmm2
   DB  15,198,210,0                        ; shufps        $0x0,%xmm2,%xmm2
@@ -13385,7 +13409,7 @@ _sk_parametric_b_sse41 LABEL PROC
   DB  102,65,15,56,20,209                 ; blendvps      %xmm0,%xmm9,%xmm2
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,95,208                           ; maxps         %xmm0,%xmm2
-  DB  15,93,21,134,35,0,0                 ; minps         0x2386(%rip),%xmm2        # 4bf0 <_sk_callback_sse41+0x8bb>
+  DB  15,93,21,130,35,0,0                 ; minps         0x2382(%rip),%xmm2        # 4c10 <_sk_callback_sse41+0x8b7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -13413,31 +13437,31 @@ _sk_parametric_a_sse41 LABEL PROC
   DB  68,15,88,219                        ; addps         %xmm3,%xmm11
   DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
   DB  69,15,91,227                        ; cvtdq2ps      %xmm11,%xmm12
-  DB  68,15,89,37,39,35,0,0               ; mulps         0x2327(%rip),%xmm12        # 4c00 <_sk_callback_sse41+0x8cb>
-  DB  68,15,84,29,47,35,0,0               ; andps         0x232f(%rip),%xmm11        # 4c10 <_sk_callback_sse41+0x8db>
-  DB  68,15,86,29,55,35,0,0               ; orps          0x2337(%rip),%xmm11        # 4c20 <_sk_callback_sse41+0x8eb>
-  DB  68,15,88,37,63,35,0,0               ; addps         0x233f(%rip),%xmm12        # 4c30 <_sk_callback_sse41+0x8fb>
-  DB  15,40,29,72,35,0,0                  ; movaps        0x2348(%rip),%xmm3        # 4c40 <_sk_callback_sse41+0x90b>
+  DB  68,15,89,37,35,35,0,0               ; mulps         0x2323(%rip),%xmm12        # 4c20 <_sk_callback_sse41+0x8c7>
+  DB  68,15,84,29,43,35,0,0               ; andps         0x232b(%rip),%xmm11        # 4c30 <_sk_callback_sse41+0x8d7>
+  DB  68,15,86,29,51,35,0,0               ; orps          0x2333(%rip),%xmm11        # 4c40 <_sk_callback_sse41+0x8e7>
+  DB  68,15,88,37,59,35,0,0               ; addps         0x233b(%rip),%xmm12        # 4c50 <_sk_callback_sse41+0x8f7>
+  DB  15,40,29,68,35,0,0                  ; movaps        0x2344(%rip),%xmm3        # 4c60 <_sk_callback_sse41+0x907>
   DB  65,15,89,219                        ; mulps         %xmm11,%xmm3
   DB  68,15,92,227                        ; subps         %xmm3,%xmm12
-  DB  68,15,88,29,72,35,0,0               ; addps         0x2348(%rip),%xmm11        # 4c50 <_sk_callback_sse41+0x91b>
-  DB  15,40,29,81,35,0,0                  ; movaps        0x2351(%rip),%xmm3        # 4c60 <_sk_callback_sse41+0x92b>
+  DB  68,15,88,29,68,35,0,0               ; addps         0x2344(%rip),%xmm11        # 4c70 <_sk_callback_sse41+0x917>
+  DB  15,40,29,77,35,0,0                  ; movaps        0x234d(%rip),%xmm3        # 4c80 <_sk_callback_sse41+0x927>
   DB  65,15,94,219                        ; divps         %xmm11,%xmm3
   DB  68,15,92,227                        ; subps         %xmm3,%xmm12
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  102,69,15,58,8,212,1                ; roundps       $0x1,%xmm12,%xmm10
   DB  69,15,40,220                        ; movaps        %xmm12,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,37,62,35,0,0               ; addps         0x233e(%rip),%xmm12        # 4c70 <_sk_callback_sse41+0x93b>
-  DB  15,40,29,71,35,0,0                  ; movaps        0x2347(%rip),%xmm3        # 4c80 <_sk_callback_sse41+0x94b>
+  DB  68,15,88,37,58,35,0,0               ; addps         0x233a(%rip),%xmm12        # 4c90 <_sk_callback_sse41+0x937>
+  DB  15,40,29,67,35,0,0                  ; movaps        0x2343(%rip),%xmm3        # 4ca0 <_sk_callback_sse41+0x947>
   DB  65,15,89,219                        ; mulps         %xmm11,%xmm3
   DB  68,15,92,227                        ; subps         %xmm3,%xmm12
-  DB  68,15,40,21,71,35,0,0               ; movaps        0x2347(%rip),%xmm10        # 4c90 <_sk_callback_sse41+0x95b>
+  DB  68,15,40,21,67,35,0,0               ; movaps        0x2343(%rip),%xmm10        # 4cb0 <_sk_callback_sse41+0x957>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  15,40,29,76,35,0,0                  ; movaps        0x234c(%rip),%xmm3        # 4ca0 <_sk_callback_sse41+0x96b>
+  DB  15,40,29,72,35,0,0                  ; movaps        0x2348(%rip),%xmm3        # 4cc0 <_sk_callback_sse41+0x967>
   DB  65,15,94,218                        ; divps         %xmm10,%xmm3
   DB  65,15,88,220                        ; addps         %xmm12,%xmm3
-  DB  15,89,29,77,35,0,0                  ; mulps         0x234d(%rip),%xmm3        # 4cb0 <_sk_callback_sse41+0x97b>
+  DB  15,89,29,73,35,0,0                  ; mulps         0x2349(%rip),%xmm3        # 4cd0 <_sk_callback_sse41+0x977>
   DB  102,68,15,91,211                    ; cvtps2dq      %xmm3,%xmm10
   DB  243,15,16,88,20                     ; movss         0x14(%rax),%xmm3
   DB  15,198,219,0                        ; shufps        $0x0,%xmm3,%xmm3
@@ -13445,7 +13469,7 @@ _sk_parametric_a_sse41 LABEL PROC
   DB  102,65,15,56,20,217                 ; blendvps      %xmm0,%xmm9,%xmm3
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,95,216                           ; maxps         %xmm0,%xmm3
-  DB  15,93,29,56,35,0,0                  ; minps         0x2338(%rip),%xmm3        # 4cc0 <_sk_callback_sse41+0x98b>
+  DB  15,93,29,52,35,0,0                  ; minps         0x2334(%rip),%xmm3        # 4ce0 <_sk_callback_sse41+0x987>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -13453,29 +13477,29 @@ _sk_parametric_a_sse41 LABEL PROC
 PUBLIC _sk_lab_to_xyz_sse41
 _sk_lab_to_xyz_sse41 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,89,5,52,35,0,0                ; mulps         0x2334(%rip),%xmm8        # 4cd0 <_sk_callback_sse41+0x99b>
-  DB  68,15,40,13,60,35,0,0               ; movaps        0x233c(%rip),%xmm9        # 4ce0 <_sk_callback_sse41+0x9ab>
+  DB  68,15,89,5,48,35,0,0                ; mulps         0x2330(%rip),%xmm8        # 4cf0 <_sk_callback_sse41+0x997>
+  DB  68,15,40,13,56,35,0,0               ; movaps        0x2338(%rip),%xmm9        # 4d00 <_sk_callback_sse41+0x9a7>
   DB  65,15,89,201                        ; mulps         %xmm9,%xmm1
-  DB  15,40,5,65,35,0,0                   ; movaps        0x2341(%rip),%xmm0        # 4cf0 <_sk_callback_sse41+0x9bb>
+  DB  15,40,5,61,35,0,0                   ; movaps        0x233d(%rip),%xmm0        # 4d10 <_sk_callback_sse41+0x9b7>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
   DB  15,88,208                           ; addps         %xmm0,%xmm2
-  DB  68,15,88,5,63,35,0,0                ; addps         0x233f(%rip),%xmm8        # 4d00 <_sk_callback_sse41+0x9cb>
-  DB  68,15,89,5,71,35,0,0                ; mulps         0x2347(%rip),%xmm8        # 4d10 <_sk_callback_sse41+0x9db>
-  DB  15,89,13,80,35,0,0                  ; mulps         0x2350(%rip),%xmm1        # 4d20 <_sk_callback_sse41+0x9eb>
+  DB  68,15,88,5,59,35,0,0                ; addps         0x233b(%rip),%xmm8        # 4d20 <_sk_callback_sse41+0x9c7>
+  DB  68,15,89,5,67,35,0,0                ; mulps         0x2343(%rip),%xmm8        # 4d30 <_sk_callback_sse41+0x9d7>
+  DB  15,89,13,76,35,0,0                  ; mulps         0x234c(%rip),%xmm1        # 4d40 <_sk_callback_sse41+0x9e7>
   DB  65,15,88,200                        ; addps         %xmm8,%xmm1
-  DB  15,89,21,85,35,0,0                  ; mulps         0x2355(%rip),%xmm2        # 4d30 <_sk_callback_sse41+0x9fb>
+  DB  15,89,21,81,35,0,0                  ; mulps         0x2351(%rip),%xmm2        # 4d50 <_sk_callback_sse41+0x9f7>
   DB  69,15,40,208                        ; movaps        %xmm8,%xmm10
   DB  68,15,92,210                        ; subps         %xmm2,%xmm10
   DB  68,15,40,217                        ; movaps        %xmm1,%xmm11
   DB  69,15,89,219                        ; mulps         %xmm11,%xmm11
   DB  68,15,89,217                        ; mulps         %xmm1,%xmm11
-  DB  68,15,40,13,73,35,0,0               ; movaps        0x2349(%rip),%xmm9        # 4d40 <_sk_callback_sse41+0xa0b>
+  DB  68,15,40,13,69,35,0,0               ; movaps        0x2345(%rip),%xmm9        # 4d60 <_sk_callback_sse41+0xa07>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  65,15,194,195,1                     ; cmpltps       %xmm11,%xmm0
-  DB  15,40,21,73,35,0,0                  ; movaps        0x2349(%rip),%xmm2        # 4d50 <_sk_callback_sse41+0xa1b>
+  DB  15,40,21,69,35,0,0                  ; movaps        0x2345(%rip),%xmm2        # 4d70 <_sk_callback_sse41+0xa17>
   DB  15,88,202                           ; addps         %xmm2,%xmm1
-  DB  68,15,40,37,78,35,0,0               ; movaps        0x234e(%rip),%xmm12        # 4d60 <_sk_callback_sse41+0xa2b>
+  DB  68,15,40,37,74,35,0,0               ; movaps        0x234a(%rip),%xmm12        # 4d80 <_sk_callback_sse41+0xa27>
   DB  65,15,89,204                        ; mulps         %xmm12,%xmm1
   DB  102,65,15,56,20,203                 ; blendvps      %xmm0,%xmm11,%xmm1
   DB  69,15,40,216                        ; movaps        %xmm8,%xmm11
@@ -13494,8 +13518,8 @@ _sk_lab_to_xyz_sse41 LABEL PROC
   DB  65,15,89,212                        ; mulps         %xmm12,%xmm2
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  102,65,15,56,20,211                 ; blendvps      %xmm0,%xmm11,%xmm2
-  DB  15,89,13,7,35,0,0                   ; mulps         0x2307(%rip),%xmm1        # 4d70 <_sk_callback_sse41+0xa3b>
-  DB  15,89,21,16,35,0,0                  ; mulps         0x2310(%rip),%xmm2        # 4d80 <_sk_callback_sse41+0xa4b>
+  DB  15,89,13,3,35,0,0                   ; mulps         0x2303(%rip),%xmm1        # 4d90 <_sk_callback_sse41+0xa37>
+  DB  15,89,21,12,35,0,0                  ; mulps         0x230c(%rip),%xmm2        # 4da0 <_sk_callback_sse41+0xa47>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,40,193                           ; movaps        %xmm1,%xmm0
   DB  65,15,40,200                        ; movaps        %xmm8,%xmm1
@@ -13507,7 +13531,7 @@ _sk_load_a8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,49,4,56                   ; pmovzxbd      (%rax,%rdi,1),%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,0,35,0,0                   ; mulps         0x2300(%rip),%xmm3        # 4d90 <_sk_callback_sse41+0xa5b>
+  DB  15,89,29,252,34,0,0                 ; mulps         0x22fc(%rip),%xmm3        # 4db0 <_sk_callback_sse41+0xa57>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,87,201                           ; xorps         %xmm1,%xmm1
@@ -13538,7 +13562,7 @@ _sk_gather_a8_sse41 LABEL PROC
   DB  102,15,58,32,192,3                  ; pinsrb        $0x3,%eax,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,148,34,0,0                 ; mulps         0x2294(%rip),%xmm3        # 4da0 <_sk_callback_sse41+0xa6b>
+  DB  15,89,29,144,34,0,0                 ; mulps         0x2290(%rip),%xmm3        # 4dc0 <_sk_callback_sse41+0xa67>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
@@ -13549,7 +13573,7 @@ PUBLIC _sk_store_a8_sse41
 _sk_store_a8_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,136,34,0,0               ; movaps        0x2288(%rip),%xmm8        # 4db0 <_sk_callback_sse41+0xa7b>
+  DB  68,15,40,5,132,34,0,0               ; movaps        0x2284(%rip),%xmm8        # 4dd0 <_sk_callback_sse41+0xa77>
   DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
   DB  102,69,15,56,43,192                 ; packusdw      %xmm8,%xmm8
@@ -13564,9 +13588,9 @@ _sk_load_g8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,49,4,56                   ; pmovzxbd      (%rax,%rdi,1),%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,101,34,0,0                  ; mulps         0x2265(%rip),%xmm0        # 4dc0 <_sk_callback_sse41+0xa8b>
+  DB  15,89,5,97,34,0,0                   ; mulps         0x2261(%rip),%xmm0        # 4de0 <_sk_callback_sse41+0xa87>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,108,34,0,0                 ; movaps        0x226c(%rip),%xmm3        # 4dd0 <_sk_callback_sse41+0xa9b>
+  DB  15,40,29,104,34,0,0                 ; movaps        0x2268(%rip),%xmm3        # 4df0 <_sk_callback_sse41+0xa97>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -13595,9 +13619,9 @@ _sk_gather_g8_sse41 LABEL PROC
   DB  102,15,58,32,192,3                  ; pinsrb        $0x3,%eax,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,5,34,0,0                    ; mulps         0x2205(%rip),%xmm0        # 4de0 <_sk_callback_sse41+0xaab>
+  DB  15,89,5,1,34,0,0                    ; mulps         0x2201(%rip),%xmm0        # 4e00 <_sk_callback_sse41+0xaa7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,12,34,0,0                  ; movaps        0x220c(%rip),%xmm3        # 4df0 <_sk_callback_sse41+0xabb>
+  DB  15,40,29,8,34,0,0                   ; movaps        0x2208(%rip),%xmm3        # 4e10 <_sk_callback_sse41+0xab7>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -13607,9 +13631,9 @@ _sk_gather_i8_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            2bfb <_sk_gather_i8_sse41+0xf>
+  DB  116,5                               ; je            2c1f <_sk_gather_i8_sse41+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           2bfd <_sk_gather_i8_sse41+0x11>
+  DB  235,2                               ; jmp           2c21 <_sk_gather_i8_sse41+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  243,15,91,201                       ; cvttps2dq     %xmm1,%xmm1
@@ -13640,17 +13664,17 @@ _sk_gather_i8_sse41 LABEL PROC
   DB  102,15,58,34,28,8,1                 ; pinsrd        $0x1,(%rax,%rcx,1),%xmm3
   DB  102,66,15,58,34,28,144,2            ; pinsrd        $0x2,(%rax,%r10,4),%xmm3
   DB  102,66,15,58,34,28,8,3              ; pinsrd        $0x3,(%rax,%r9,1),%xmm3
-  DB  102,15,111,5,99,33,0,0              ; movdqa        0x2163(%rip),%xmm0        # 4e00 <_sk_callback_sse41+0xacb>
+  DB  102,15,111,5,95,33,0,0              ; movdqa        0x215f(%rip),%xmm0        # 4e20 <_sk_callback_sse41+0xac7>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,100,33,0,0               ; movaps        0x2164(%rip),%xmm8        # 4e10 <_sk_callback_sse41+0xadb>
+  DB  68,15,40,5,96,33,0,0                ; movaps        0x2160(%rip),%xmm8        # 4e30 <_sk_callback_sse41+0xad7>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
-  DB  102,15,56,0,13,99,33,0,0            ; pshufb        0x2163(%rip),%xmm1        # 4e20 <_sk_callback_sse41+0xaeb>
+  DB  102,15,56,0,13,95,33,0,0            ; pshufb        0x215f(%rip),%xmm1        # 4e40 <_sk_callback_sse41+0xae7>
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,111,211                      ; movdqa        %xmm3,%xmm2
-  DB  102,15,56,0,21,95,33,0,0            ; pshufb        0x215f(%rip),%xmm2        # 4e30 <_sk_callback_sse41+0xafb>
+  DB  102,15,56,0,21,91,33,0,0            ; pshufb        0x215b(%rip),%xmm2        # 4e50 <_sk_callback_sse41+0xaf7>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  102,15,114,211,24                   ; psrld         $0x18,%xmm3
@@ -13664,19 +13688,19 @@ _sk_load_565_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,51,20,120                 ; pmovzxwd      (%rax,%rdi,2),%xmm2
-  DB  102,15,111,5,69,33,0,0              ; movdqa        0x2145(%rip),%xmm0        # 4e40 <_sk_callback_sse41+0xb0b>
+  DB  102,15,111,5,65,33,0,0              ; movdqa        0x2141(%rip),%xmm0        # 4e60 <_sk_callback_sse41+0xb07>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,71,33,0,0                   ; mulps         0x2147(%rip),%xmm0        # 4e50 <_sk_callback_sse41+0xb1b>
-  DB  102,15,111,13,79,33,0,0             ; movdqa        0x214f(%rip),%xmm1        # 4e60 <_sk_callback_sse41+0xb2b>
+  DB  15,89,5,67,33,0,0                   ; mulps         0x2143(%rip),%xmm0        # 4e70 <_sk_callback_sse41+0xb17>
+  DB  102,15,111,13,75,33,0,0             ; movdqa        0x214b(%rip),%xmm1        # 4e80 <_sk_callback_sse41+0xb27>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,81,33,0,0                  ; mulps         0x2151(%rip),%xmm1        # 4e70 <_sk_callback_sse41+0xb3b>
-  DB  102,15,219,21,89,33,0,0             ; pand          0x2159(%rip),%xmm2        # 4e80 <_sk_callback_sse41+0xb4b>
+  DB  15,89,13,77,33,0,0                  ; mulps         0x214d(%rip),%xmm1        # 4e90 <_sk_callback_sse41+0xb37>
+  DB  102,15,219,21,85,33,0,0             ; pand          0x2155(%rip),%xmm2        # 4ea0 <_sk_callback_sse41+0xb47>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,95,33,0,0                  ; mulps         0x215f(%rip),%xmm2        # 4e90 <_sk_callback_sse41+0xb5b>
+  DB  15,89,21,91,33,0,0                  ; mulps         0x215b(%rip),%xmm2        # 4eb0 <_sk_callback_sse41+0xb57>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,102,33,0,0                 ; movaps        0x2166(%rip),%xmm3        # 4ea0 <_sk_callback_sse41+0xb6b>
+  DB  15,40,29,98,33,0,0                  ; movaps        0x2162(%rip),%xmm3        # 4ec0 <_sk_callback_sse41+0xb67>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_gather_565_sse41
@@ -13702,31 +13726,31 @@ _sk_gather_565_sse41 LABEL PROC
   DB  65,15,183,4,65                      ; movzwl        (%r9,%rax,2),%eax
   DB  102,15,196,192,3                    ; pinsrw        $0x3,%eax,%xmm0
   DB  102,15,56,51,208                    ; pmovzxwd      %xmm0,%xmm2
-  DB  102,15,111,5,11,33,0,0              ; movdqa        0x210b(%rip),%xmm0        # 4eb0 <_sk_callback_sse41+0xb7b>
+  DB  102,15,111,5,7,33,0,0               ; movdqa        0x2107(%rip),%xmm0        # 4ed0 <_sk_callback_sse41+0xb77>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,13,33,0,0                   ; mulps         0x210d(%rip),%xmm0        # 4ec0 <_sk_callback_sse41+0xb8b>
-  DB  102,15,111,13,21,33,0,0             ; movdqa        0x2115(%rip),%xmm1        # 4ed0 <_sk_callback_sse41+0xb9b>
+  DB  15,89,5,9,33,0,0                    ; mulps         0x2109(%rip),%xmm0        # 4ee0 <_sk_callback_sse41+0xb87>
+  DB  102,15,111,13,17,33,0,0             ; movdqa        0x2111(%rip),%xmm1        # 4ef0 <_sk_callback_sse41+0xb97>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,23,33,0,0                  ; mulps         0x2117(%rip),%xmm1        # 4ee0 <_sk_callback_sse41+0xbab>
-  DB  102,15,219,21,31,33,0,0             ; pand          0x211f(%rip),%xmm2        # 4ef0 <_sk_callback_sse41+0xbbb>
+  DB  15,89,13,19,33,0,0                  ; mulps         0x2113(%rip),%xmm1        # 4f00 <_sk_callback_sse41+0xba7>
+  DB  102,15,219,21,27,33,0,0             ; pand          0x211b(%rip),%xmm2        # 4f10 <_sk_callback_sse41+0xbb7>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,37,33,0,0                  ; mulps         0x2125(%rip),%xmm2        # 4f00 <_sk_callback_sse41+0xbcb>
+  DB  15,89,21,33,33,0,0                  ; mulps         0x2121(%rip),%xmm2        # 4f20 <_sk_callback_sse41+0xbc7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,44,33,0,0                  ; movaps        0x212c(%rip),%xmm3        # 4f10 <_sk_callback_sse41+0xbdb>
+  DB  15,40,29,40,33,0,0                  ; movaps        0x2128(%rip),%xmm3        # 4f30 <_sk_callback_sse41+0xbd7>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_565_sse41
 _sk_store_565_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,45,33,0,0                ; movaps        0x212d(%rip),%xmm8        # 4f20 <_sk_callback_sse41+0xbeb>
+  DB  68,15,40,5,41,33,0,0                ; movaps        0x2129(%rip),%xmm8        # 4f40 <_sk_callback_sse41+0xbe7>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
   DB  102,65,15,114,241,11                ; pslld         $0xb,%xmm9
-  DB  68,15,40,21,34,33,0,0               ; movaps        0x2122(%rip),%xmm10        # 4f30 <_sk_callback_sse41+0xbfb>
+  DB  68,15,40,21,30,33,0,0               ; movaps        0x211e(%rip),%xmm10        # 4f50 <_sk_callback_sse41+0xbf7>
   DB  68,15,89,209                        ; mulps         %xmm1,%xmm10
   DB  102,69,15,91,210                    ; cvtps2dq      %xmm10,%xmm10
   DB  102,65,15,114,242,5                 ; pslld         $0x5,%xmm10
@@ -13744,21 +13768,21 @@ _sk_load_4444_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,51,28,120                 ; pmovzxwd      (%rax,%rdi,2),%xmm3
-  DB  102,15,111,5,237,32,0,0             ; movdqa        0x20ed(%rip),%xmm0        # 4f40 <_sk_callback_sse41+0xc0b>
+  DB  102,15,111,5,233,32,0,0             ; movdqa        0x20e9(%rip),%xmm0        # 4f60 <_sk_callback_sse41+0xc07>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,239,32,0,0                  ; mulps         0x20ef(%rip),%xmm0        # 4f50 <_sk_callback_sse41+0xc1b>
-  DB  102,15,111,13,247,32,0,0            ; movdqa        0x20f7(%rip),%xmm1        # 4f60 <_sk_callback_sse41+0xc2b>
+  DB  15,89,5,235,32,0,0                  ; mulps         0x20eb(%rip),%xmm0        # 4f70 <_sk_callback_sse41+0xc17>
+  DB  102,15,111,13,243,32,0,0            ; movdqa        0x20f3(%rip),%xmm1        # 4f80 <_sk_callback_sse41+0xc27>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,249,32,0,0                 ; mulps         0x20f9(%rip),%xmm1        # 4f70 <_sk_callback_sse41+0xc3b>
-  DB  102,15,111,21,1,33,0,0              ; movdqa        0x2101(%rip),%xmm2        # 4f80 <_sk_callback_sse41+0xc4b>
+  DB  15,89,13,245,32,0,0                 ; mulps         0x20f5(%rip),%xmm1        # 4f90 <_sk_callback_sse41+0xc37>
+  DB  102,15,111,21,253,32,0,0            ; movdqa        0x20fd(%rip),%xmm2        # 4fa0 <_sk_callback_sse41+0xc47>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,3,33,0,0                   ; mulps         0x2103(%rip),%xmm2        # 4f90 <_sk_callback_sse41+0xc5b>
-  DB  102,15,219,29,11,33,0,0             ; pand          0x210b(%rip),%xmm3        # 4fa0 <_sk_callback_sse41+0xc6b>
+  DB  15,89,21,255,32,0,0                 ; mulps         0x20ff(%rip),%xmm2        # 4fb0 <_sk_callback_sse41+0xc57>
+  DB  102,15,219,29,7,33,0,0              ; pand          0x2107(%rip),%xmm3        # 4fc0 <_sk_callback_sse41+0xc67>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,17,33,0,0                  ; mulps         0x2111(%rip),%xmm3        # 4fb0 <_sk_callback_sse41+0xc7b>
+  DB  15,89,29,13,33,0,0                  ; mulps         0x210d(%rip),%xmm3        # 4fd0 <_sk_callback_sse41+0xc77>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -13785,21 +13809,21 @@ _sk_gather_4444_sse41 LABEL PROC
   DB  65,15,183,4,65                      ; movzwl        (%r9,%rax,2),%eax
   DB  102,15,196,192,3                    ; pinsrw        $0x3,%eax,%xmm0
   DB  102,15,56,51,216                    ; pmovzxwd      %xmm0,%xmm3
-  DB  102,15,111,5,180,32,0,0             ; movdqa        0x20b4(%rip),%xmm0        # 4fc0 <_sk_callback_sse41+0xc8b>
+  DB  102,15,111,5,176,32,0,0             ; movdqa        0x20b0(%rip),%xmm0        # 4fe0 <_sk_callback_sse41+0xc87>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,182,32,0,0                  ; mulps         0x20b6(%rip),%xmm0        # 4fd0 <_sk_callback_sse41+0xc9b>
-  DB  102,15,111,13,190,32,0,0            ; movdqa        0x20be(%rip),%xmm1        # 4fe0 <_sk_callback_sse41+0xcab>
+  DB  15,89,5,178,32,0,0                  ; mulps         0x20b2(%rip),%xmm0        # 4ff0 <_sk_callback_sse41+0xc97>
+  DB  102,15,111,13,186,32,0,0            ; movdqa        0x20ba(%rip),%xmm1        # 5000 <_sk_callback_sse41+0xca7>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,192,32,0,0                 ; mulps         0x20c0(%rip),%xmm1        # 4ff0 <_sk_callback_sse41+0xcbb>
-  DB  102,15,111,21,200,32,0,0            ; movdqa        0x20c8(%rip),%xmm2        # 5000 <_sk_callback_sse41+0xccb>
+  DB  15,89,13,188,32,0,0                 ; mulps         0x20bc(%rip),%xmm1        # 5010 <_sk_callback_sse41+0xcb7>
+  DB  102,15,111,21,196,32,0,0            ; movdqa        0x20c4(%rip),%xmm2        # 5020 <_sk_callback_sse41+0xcc7>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,202,32,0,0                 ; mulps         0x20ca(%rip),%xmm2        # 5010 <_sk_callback_sse41+0xcdb>
-  DB  102,15,219,29,210,32,0,0            ; pand          0x20d2(%rip),%xmm3        # 5020 <_sk_callback_sse41+0xceb>
+  DB  15,89,21,198,32,0,0                 ; mulps         0x20c6(%rip),%xmm2        # 5030 <_sk_callback_sse41+0xcd7>
+  DB  102,15,219,29,206,32,0,0            ; pand          0x20ce(%rip),%xmm3        # 5040 <_sk_callback_sse41+0xce7>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,216,32,0,0                 ; mulps         0x20d8(%rip),%xmm3        # 5030 <_sk_callback_sse41+0xcfb>
+  DB  15,89,29,212,32,0,0                 ; mulps         0x20d4(%rip),%xmm3        # 5050 <_sk_callback_sse41+0xcf7>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -13807,7 +13831,7 @@ PUBLIC _sk_store_4444_sse41
 _sk_store_4444_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,215,32,0,0               ; movaps        0x20d7(%rip),%xmm8        # 5040 <_sk_callback_sse41+0xd0b>
+  DB  68,15,40,5,211,32,0,0               ; movaps        0x20d3(%rip),%xmm8        # 5060 <_sk_callback_sse41+0xd07>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -13835,17 +13859,17 @@ _sk_load_8888_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  15,16,28,184                        ; movups        (%rax,%rdi,4),%xmm3
-  DB  15,40,5,118,32,0,0                  ; movaps        0x2076(%rip),%xmm0        # 5050 <_sk_callback_sse41+0xd1b>
+  DB  15,40,5,114,32,0,0                  ; movaps        0x2072(%rip),%xmm0        # 5070 <_sk_callback_sse41+0xd17>
   DB  15,84,195                           ; andps         %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,120,32,0,0               ; movaps        0x2078(%rip),%xmm8        # 5060 <_sk_callback_sse41+0xd2b>
+  DB  68,15,40,5,116,32,0,0               ; movaps        0x2074(%rip),%xmm8        # 5080 <_sk_callback_sse41+0xd27>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,40,203                           ; movaps        %xmm3,%xmm1
-  DB  102,15,56,0,13,120,32,0,0           ; pshufb        0x2078(%rip),%xmm1        # 5070 <_sk_callback_sse41+0xd3b>
+  DB  102,15,56,0,13,116,32,0,0           ; pshufb        0x2074(%rip),%xmm1        # 5090 <_sk_callback_sse41+0xd37>
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  15,40,211                           ; movaps        %xmm3,%xmm2
-  DB  102,15,56,0,21,117,32,0,0           ; pshufb        0x2075(%rip),%xmm2        # 5080 <_sk_callback_sse41+0xd4b>
+  DB  102,15,56,0,21,113,32,0,0           ; pshufb        0x2071(%rip),%xmm2        # 50a0 <_sk_callback_sse41+0xd47>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  102,15,114,211,24                   ; psrld         $0x18,%xmm3
@@ -13874,17 +13898,17 @@ _sk_gather_8888_sse41 LABEL PROC
   DB  102,65,15,58,34,28,129,1            ; pinsrd        $0x1,(%r9,%rax,4),%xmm3
   DB  102,67,15,58,34,28,145,2            ; pinsrd        $0x2,(%r9,%r10,4),%xmm3
   DB  102,65,15,58,34,28,137,3            ; pinsrd        $0x3,(%r9,%rcx,4),%xmm3
-  DB  102,15,111,5,14,32,0,0              ; movdqa        0x200e(%rip),%xmm0        # 5090 <_sk_callback_sse41+0xd5b>
+  DB  102,15,111,5,10,32,0,0              ; movdqa        0x200a(%rip),%xmm0        # 50b0 <_sk_callback_sse41+0xd57>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,15,32,0,0                ; movaps        0x200f(%rip),%xmm8        # 50a0 <_sk_callback_sse41+0xd6b>
+  DB  68,15,40,5,11,32,0,0                ; movaps        0x200b(%rip),%xmm8        # 50c0 <_sk_callback_sse41+0xd67>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
-  DB  102,15,56,0,13,14,32,0,0            ; pshufb        0x200e(%rip),%xmm1        # 50b0 <_sk_callback_sse41+0xd7b>
+  DB  102,15,56,0,13,10,32,0,0            ; pshufb        0x200a(%rip),%xmm1        # 50d0 <_sk_callback_sse41+0xd77>
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,111,211                      ; movdqa        %xmm3,%xmm2
-  DB  102,15,56,0,21,10,32,0,0            ; pshufb        0x200a(%rip),%xmm2        # 50c0 <_sk_callback_sse41+0xd8b>
+  DB  102,15,56,0,21,6,32,0,0             ; pshufb        0x2006(%rip),%xmm2        # 50e0 <_sk_callback_sse41+0xd87>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  102,15,114,211,24                   ; psrld         $0x18,%xmm3
@@ -13897,7 +13921,7 @@ PUBLIC _sk_store_8888_sse41
 _sk_store_8888_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,246,31,0,0               ; movaps        0x1ff6(%rip),%xmm8        # 50d0 <_sk_callback_sse41+0xd9b>
+  DB  68,15,40,5,242,31,0,0               ; movaps        0x1ff2(%rip),%xmm8        # 50f0 <_sk_callback_sse41+0xd97>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -13932,18 +13956,18 @@ _sk_load_f16_sse41 LABEL PROC
   DB  102,68,15,97,216                    ; punpcklwd     %xmm0,%xmm11
   DB  102,68,15,105,200                   ; punpckhwd     %xmm0,%xmm9
   DB  102,65,15,56,51,203                 ; pmovzxwd      %xmm11,%xmm1
-  DB  102,68,15,111,5,111,31,0,0          ; movdqa        0x1f6f(%rip),%xmm8        # 50e0 <_sk_callback_sse41+0xdab>
+  DB  102,68,15,111,5,107,31,0,0          ; movdqa        0x1f6b(%rip),%xmm8        # 5100 <_sk_callback_sse41+0xda7>
   DB  102,15,111,209                      ; movdqa        %xmm1,%xmm2
   DB  102,65,15,219,208                   ; pand          %xmm8,%xmm2
   DB  102,15,239,202                      ; pxor          %xmm2,%xmm1
-  DB  102,15,111,29,106,31,0,0            ; movdqa        0x1f6a(%rip),%xmm3        # 50f0 <_sk_callback_sse41+0xdbb>
+  DB  102,15,111,29,102,31,0,0            ; movdqa        0x1f66(%rip),%xmm3        # 5110 <_sk_callback_sse41+0xdb7>
   DB  102,15,114,242,16                   ; pslld         $0x10,%xmm2
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,15,56,63,195                    ; pmaxud        %xmm3,%xmm0
   DB  102,15,118,193                      ; pcmpeqd       %xmm1,%xmm0
   DB  102,15,114,241,13                   ; pslld         $0xd,%xmm1
   DB  102,15,235,202                      ; por           %xmm2,%xmm1
-  DB  102,68,15,111,21,86,31,0,0          ; movdqa        0x1f56(%rip),%xmm10        # 5100 <_sk_callback_sse41+0xdcb>
+  DB  102,68,15,111,21,82,31,0,0          ; movdqa        0x1f52(%rip),%xmm10        # 5120 <_sk_callback_sse41+0xdc7>
   DB  102,65,15,254,202                   ; paddd         %xmm10,%xmm1
   DB  102,15,219,193                      ; pand          %xmm1,%xmm0
   DB  102,65,15,115,219,8                 ; psrldq        $0x8,%xmm11
@@ -14014,18 +14038,18 @@ _sk_gather_f16_sse41 LABEL PROC
   DB  102,68,15,97,218                    ; punpcklwd     %xmm2,%xmm11
   DB  102,68,15,105,202                   ; punpckhwd     %xmm2,%xmm9
   DB  102,65,15,56,51,203                 ; pmovzxwd      %xmm11,%xmm1
-  DB  102,68,15,111,5,20,30,0,0           ; movdqa        0x1e14(%rip),%xmm8        # 5110 <_sk_callback_sse41+0xddb>
+  DB  102,68,15,111,5,16,30,0,0           ; movdqa        0x1e10(%rip),%xmm8        # 5130 <_sk_callback_sse41+0xdd7>
   DB  102,15,111,209                      ; movdqa        %xmm1,%xmm2
   DB  102,65,15,219,208                   ; pand          %xmm8,%xmm2
   DB  102,15,239,202                      ; pxor          %xmm2,%xmm1
-  DB  102,15,111,29,15,30,0,0             ; movdqa        0x1e0f(%rip),%xmm3        # 5120 <_sk_callback_sse41+0xdeb>
+  DB  102,15,111,29,11,30,0,0             ; movdqa        0x1e0b(%rip),%xmm3        # 5140 <_sk_callback_sse41+0xde7>
   DB  102,15,114,242,16                   ; pslld         $0x10,%xmm2
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,15,56,63,195                    ; pmaxud        %xmm3,%xmm0
   DB  102,15,118,193                      ; pcmpeqd       %xmm1,%xmm0
   DB  102,15,114,241,13                   ; pslld         $0xd,%xmm1
   DB  102,15,235,202                      ; por           %xmm2,%xmm1
-  DB  102,68,15,111,21,251,29,0,0         ; movdqa        0x1dfb(%rip),%xmm10        # 5130 <_sk_callback_sse41+0xdfb>
+  DB  102,68,15,111,21,247,29,0,0         ; movdqa        0x1df7(%rip),%xmm10        # 5150 <_sk_callback_sse41+0xdf7>
   DB  102,65,15,254,202                   ; paddd         %xmm10,%xmm1
   DB  102,15,219,193                      ; pand          %xmm1,%xmm0
   DB  102,65,15,115,219,8                 ; psrldq        $0x8,%xmm11
@@ -14071,17 +14095,17 @@ PUBLIC _sk_store_f16_sse41
 _sk_store_f16_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  102,68,15,111,21,49,29,0,0          ; movdqa        0x1d31(%rip),%xmm10        # 5140 <_sk_callback_sse41+0xe0b>
+  DB  102,68,15,111,21,45,29,0,0          ; movdqa        0x1d2d(%rip),%xmm10        # 5160 <_sk_callback_sse41+0xe07>
   DB  102,68,15,111,224                   ; movdqa        %xmm0,%xmm12
   DB  102,68,15,111,232                   ; movdqa        %xmm0,%xmm13
   DB  102,69,15,219,234                   ; pand          %xmm10,%xmm13
   DB  102,69,15,239,229                   ; pxor          %xmm13,%xmm12
-  DB  102,68,15,111,13,36,29,0,0          ; movdqa        0x1d24(%rip),%xmm9        # 5150 <_sk_callback_sse41+0xe1b>
+  DB  102,68,15,111,13,32,29,0,0          ; movdqa        0x1d20(%rip),%xmm9        # 5170 <_sk_callback_sse41+0xe17>
   DB  102,65,15,114,213,16                ; psrld         $0x10,%xmm13
   DB  102,69,15,111,193                   ; movdqa        %xmm9,%xmm8
   DB  102,69,15,102,196                   ; pcmpgtd       %xmm12,%xmm8
   DB  102,65,15,114,212,13                ; psrld         $0xd,%xmm12
-  DB  102,68,15,111,29,21,29,0,0          ; movdqa        0x1d15(%rip),%xmm11        # 5160 <_sk_callback_sse41+0xe2b>
+  DB  102,68,15,111,29,17,29,0,0          ; movdqa        0x1d11(%rip),%xmm11        # 5180 <_sk_callback_sse41+0xe27>
   DB  102,69,15,235,235                   ; por           %xmm11,%xmm13
   DB  102,69,15,254,236                   ; paddd         %xmm12,%xmm13
   DB  102,69,15,223,197                   ; pandn         %xmm13,%xmm8
@@ -14149,7 +14173,7 @@ _sk_load_u16_be_sse41 LABEL PROC
   DB  102,15,235,200                      ; por           %xmm0,%xmm1
   DB  102,15,56,51,193                    ; pmovzxwd      %xmm1,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,228,27,0,0               ; movaps        0x1be4(%rip),%xmm8        # 5170 <_sk_callback_sse41+0xe3b>
+  DB  68,15,40,5,224,27,0,0               ; movaps        0x1be0(%rip),%xmm8        # 5190 <_sk_callback_sse41+0xe37>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -14199,7 +14223,7 @@ _sk_load_rgb_u16_be_sse41 LABEL PROC
   DB  102,15,235,193                      ; por           %xmm1,%xmm0
   DB  102,15,56,51,192                    ; pmovzxwd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,37,27,0,0                ; movaps        0x1b25(%rip),%xmm8        # 5180 <_sk_callback_sse41+0xe4b>
+  DB  68,15,40,5,33,27,0,0                ; movaps        0x1b21(%rip),%xmm8        # 51a0 <_sk_callback_sse41+0xe47>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -14216,14 +14240,14 @@ _sk_load_rgb_u16_be_sse41 LABEL PROC
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,236,26,0,0                 ; movaps        0x1aec(%rip),%xmm3        # 5190 <_sk_callback_sse41+0xe5b>
+  DB  15,40,29,232,26,0,0                 ; movaps        0x1ae8(%rip),%xmm3        # 51b0 <_sk_callback_sse41+0xe57>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_u16_be_sse41
 _sk_store_u16_be_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,13,237,26,0,0              ; movaps        0x1aed(%rip),%xmm9        # 51a0 <_sk_callback_sse41+0xe6b>
+  DB  68,15,40,13,233,26,0,0              ; movaps        0x1ae9(%rip),%xmm9        # 51c0 <_sk_callback_sse41+0xe67>
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
@@ -14428,10 +14452,10 @@ _sk_mirror_y_sse41 LABEL PROC
 PUBLIC _sk_luminance_to_alpha_sse41
 _sk_luminance_to_alpha_sse41 LABEL PROC
   DB  15,40,218                           ; movaps        %xmm2,%xmm3
-  DB  15,89,5,11,24,0,0                   ; mulps         0x180b(%rip),%xmm0        # 51b0 <_sk_callback_sse41+0xe7b>
-  DB  15,89,13,20,24,0,0                  ; mulps         0x1814(%rip),%xmm1        # 51c0 <_sk_callback_sse41+0xe8b>
+  DB  15,89,5,7,24,0,0                    ; mulps         0x1807(%rip),%xmm0        # 51d0 <_sk_callback_sse41+0xe77>
+  DB  15,89,13,16,24,0,0                  ; mulps         0x1810(%rip),%xmm1        # 51e0 <_sk_callback_sse41+0xe87>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  15,89,29,26,24,0,0                  ; mulps         0x181a(%rip),%xmm3        # 51d0 <_sk_callback_sse41+0xe9b>
+  DB  15,89,29,22,24,0,0                  ; mulps         0x1816(%rip),%xmm3        # 51f0 <_sk_callback_sse41+0xe97>
   DB  15,88,217                           ; addps         %xmm1,%xmm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
@@ -14654,7 +14678,7 @@ _sk_linear_gradient_sse41 LABEL PROC
   DB  69,15,198,237,0                     ; shufps        $0x0,%xmm13,%xmm13
   DB  72,139,8                            ; mov           (%rax),%rcx
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,132,4,1,0,0                      ; je            3e5e <_sk_linear_gradient_sse41+0x13e>
+  DB  15,132,4,1,0,0                      ; je            3e82 <_sk_linear_gradient_sse41+0x13e>
   DB  72,131,236,88                       ; sub           $0x58,%rsp
   DB  15,41,36,36                         ; movaps        %xmm4,(%rsp)
   DB  15,41,108,36,16                     ; movaps        %xmm5,0x10(%rsp)
@@ -14705,13 +14729,13 @@ _sk_linear_gradient_sse41 LABEL PROC
   DB  15,40,196                           ; movaps        %xmm4,%xmm0
   DB  72,131,192,36                       ; add           $0x24,%rax
   DB  72,255,201                          ; dec           %rcx
-  DB  15,133,65,255,255,255               ; jne           3d86 <_sk_linear_gradient_sse41+0x66>
+  DB  15,133,65,255,255,255               ; jne           3daa <_sk_linear_gradient_sse41+0x66>
   DB  15,40,124,36,48                     ; movaps        0x30(%rsp),%xmm7
   DB  15,40,116,36,32                     ; movaps        0x20(%rsp),%xmm6
   DB  15,40,108,36,16                     ; movaps        0x10(%rsp),%xmm5
   DB  15,40,36,36                         ; movaps        (%rsp),%xmm4
   DB  72,131,196,88                       ; add           $0x58,%rsp
-  DB  235,13                              ; jmp           3e6b <_sk_linear_gradient_sse41+0x14b>
+  DB  235,13                              ; jmp           3e8f <_sk_linear_gradient_sse41+0x14b>
   DB  15,87,201                           ; xorps         %xmm1,%xmm1
   DB  15,87,210                           ; xorps         %xmm2,%xmm2
   DB  15,87,219                           ; xorps         %xmm3,%xmm3
@@ -14776,26 +14800,26 @@ _sk_xy_to_polar_unit_sse41 LABEL PROC
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,40,236                        ; movaps        %xmm12,%xmm13
   DB  69,15,89,237                        ; mulps         %xmm13,%xmm13
-  DB  68,15,40,21,157,18,0,0              ; movaps        0x129d(%rip),%xmm10        # 51e0 <_sk_callback_sse41+0xeab>
+  DB  68,15,40,21,153,18,0,0              ; movaps        0x1299(%rip),%xmm10        # 5200 <_sk_callback_sse41+0xea7>
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,88,21,161,18,0,0              ; addps         0x12a1(%rip),%xmm10        # 51f0 <_sk_callback_sse41+0xebb>
+  DB  68,15,88,21,157,18,0,0              ; addps         0x129d(%rip),%xmm10        # 5210 <_sk_callback_sse41+0xeb7>
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,88,21,165,18,0,0              ; addps         0x12a5(%rip),%xmm10        # 5200 <_sk_callback_sse41+0xecb>
+  DB  68,15,88,21,161,18,0,0              ; addps         0x12a1(%rip),%xmm10        # 5220 <_sk_callback_sse41+0xec7>
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,88,21,169,18,0,0              ; addps         0x12a9(%rip),%xmm10        # 5210 <_sk_callback_sse41+0xedb>
+  DB  68,15,88,21,165,18,0,0              ; addps         0x12a5(%rip),%xmm10        # 5230 <_sk_callback_sse41+0xed7>
   DB  69,15,89,212                        ; mulps         %xmm12,%xmm10
   DB  65,15,194,195,1                     ; cmpltps       %xmm11,%xmm0
-  DB  68,15,40,29,168,18,0,0              ; movaps        0x12a8(%rip),%xmm11        # 5220 <_sk_callback_sse41+0xeeb>
+  DB  68,15,40,29,164,18,0,0              ; movaps        0x12a4(%rip),%xmm11        # 5240 <_sk_callback_sse41+0xee7>
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  102,69,15,56,20,211                 ; blendvps      %xmm0,%xmm11,%xmm10
   DB  69,15,194,200,1                     ; cmpltps       %xmm8,%xmm9
-  DB  68,15,40,29,161,18,0,0              ; movaps        0x12a1(%rip),%xmm11        # 5230 <_sk_callback_sse41+0xefb>
+  DB  68,15,40,29,157,18,0,0              ; movaps        0x129d(%rip),%xmm11        # 5250 <_sk_callback_sse41+0xef7>
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  102,69,15,56,20,211                 ; blendvps      %xmm0,%xmm11,%xmm10
   DB  15,40,193                           ; movaps        %xmm1,%xmm0
   DB  65,15,194,192,1                     ; cmpltps       %xmm8,%xmm0
-  DB  68,15,40,13,147,18,0,0              ; movaps        0x1293(%rip),%xmm9        # 5240 <_sk_callback_sse41+0xf0b>
+  DB  68,15,40,13,143,18,0,0              ; movaps        0x128f(%rip),%xmm9        # 5260 <_sk_callback_sse41+0xf07>
   DB  69,15,92,202                        ; subps         %xmm10,%xmm9
   DB  102,69,15,56,20,209                 ; blendvps      %xmm0,%xmm9,%xmm10
   DB  69,15,194,194,7                     ; cmpordps      %xmm10,%xmm8
@@ -14818,7 +14842,7 @@ _sk_xy_to_radius_sse41 LABEL PROC
 PUBLIC _sk_save_xy_sse41
 _sk_save_xy_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,100,18,0,0               ; movaps        0x1264(%rip),%xmm8        # 5250 <_sk_callback_sse41+0xf1b>
+  DB  68,15,40,5,96,18,0,0                ; movaps        0x1260(%rip),%xmm8        # 5270 <_sk_callback_sse41+0xf17>
   DB  15,17,0                             ; movups        %xmm0,(%rax)
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,88,200                        ; addps         %xmm8,%xmm9
@@ -14858,8 +14882,8 @@ _sk_bilinear_nx_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,230,17,0,0                  ; addps         0x11e6(%rip),%xmm0        # 5260 <_sk_callback_sse41+0xf2b>
-  DB  68,15,40,13,238,17,0,0              ; movaps        0x11ee(%rip),%xmm9        # 5270 <_sk_callback_sse41+0xf3b>
+  DB  15,88,5,226,17,0,0                  ; addps         0x11e2(%rip),%xmm0        # 5280 <_sk_callback_sse41+0xf27>
+  DB  68,15,40,13,234,17,0,0              ; movaps        0x11ea(%rip),%xmm9        # 5290 <_sk_callback_sse41+0xf37>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -14870,7 +14894,7 @@ _sk_bilinear_px_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,221,17,0,0                  ; addps         0x11dd(%rip),%xmm0        # 5280 <_sk_callback_sse41+0xf4b>
+  DB  15,88,5,217,17,0,0                  ; addps         0x11d9(%rip),%xmm0        # 52a0 <_sk_callback_sse41+0xf47>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -14880,8 +14904,8 @@ _sk_bilinear_ny_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,207,17,0,0                 ; addps         0x11cf(%rip),%xmm1        # 5290 <_sk_callback_sse41+0xf5b>
-  DB  68,15,40,13,215,17,0,0              ; movaps        0x11d7(%rip),%xmm9        # 52a0 <_sk_callback_sse41+0xf6b>
+  DB  15,88,13,203,17,0,0                 ; addps         0x11cb(%rip),%xmm1        # 52b0 <_sk_callback_sse41+0xf57>
+  DB  68,15,40,13,211,17,0,0              ; movaps        0x11d3(%rip),%xmm9        # 52c0 <_sk_callback_sse41+0xf67>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -14892,7 +14916,7 @@ _sk_bilinear_py_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,197,17,0,0                 ; addps         0x11c5(%rip),%xmm1        # 52b0 <_sk_callback_sse41+0xf7b>
+  DB  15,88,13,193,17,0,0                 ; addps         0x11c1(%rip),%xmm1        # 52d0 <_sk_callback_sse41+0xf77>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -14902,13 +14926,13 @@ _sk_bicubic_n3x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,184,17,0,0                  ; addps         0x11b8(%rip),%xmm0        # 52c0 <_sk_callback_sse41+0xf8b>
-  DB  68,15,40,13,192,17,0,0              ; movaps        0x11c0(%rip),%xmm9        # 52d0 <_sk_callback_sse41+0xf9b>
+  DB  15,88,5,180,17,0,0                  ; addps         0x11b4(%rip),%xmm0        # 52e0 <_sk_callback_sse41+0xf87>
+  DB  68,15,40,13,188,17,0,0              ; movaps        0x11bc(%rip),%xmm9        # 52f0 <_sk_callback_sse41+0xf97>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,188,17,0,0              ; mulps         0x11bc(%rip),%xmm9        # 52e0 <_sk_callback_sse41+0xfab>
-  DB  68,15,88,13,196,17,0,0              ; addps         0x11c4(%rip),%xmm9        # 52f0 <_sk_callback_sse41+0xfbb>
+  DB  68,15,89,13,184,17,0,0              ; mulps         0x11b8(%rip),%xmm9        # 5300 <_sk_callback_sse41+0xfa7>
+  DB  68,15,88,13,192,17,0,0              ; addps         0x11c0(%rip),%xmm9        # 5310 <_sk_callback_sse41+0xfb7>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -14919,16 +14943,16 @@ _sk_bicubic_n1x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,179,17,0,0                  ; addps         0x11b3(%rip),%xmm0        # 5300 <_sk_callback_sse41+0xfcb>
-  DB  68,15,40,13,187,17,0,0              ; movaps        0x11bb(%rip),%xmm9        # 5310 <_sk_callback_sse41+0xfdb>
+  DB  15,88,5,175,17,0,0                  ; addps         0x11af(%rip),%xmm0        # 5320 <_sk_callback_sse41+0xfc7>
+  DB  68,15,40,13,183,17,0,0              ; movaps        0x11b7(%rip),%xmm9        # 5330 <_sk_callback_sse41+0xfd7>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,191,17,0,0               ; movaps        0x11bf(%rip),%xmm8        # 5320 <_sk_callback_sse41+0xfeb>
+  DB  68,15,40,5,187,17,0,0               ; movaps        0x11bb(%rip),%xmm8        # 5340 <_sk_callback_sse41+0xfe7>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,195,17,0,0               ; addps         0x11c3(%rip),%xmm8        # 5330 <_sk_callback_sse41+0xffb>
+  DB  68,15,88,5,191,17,0,0               ; addps         0x11bf(%rip),%xmm8        # 5350 <_sk_callback_sse41+0xff7>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,199,17,0,0               ; addps         0x11c7(%rip),%xmm8        # 5340 <_sk_callback_sse41+0x100b>
+  DB  68,15,88,5,195,17,0,0               ; addps         0x11c3(%rip),%xmm8        # 5360 <_sk_callback_sse41+0x1007>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,203,17,0,0               ; addps         0x11cb(%rip),%xmm8        # 5350 <_sk_callback_sse41+0x101b>
+  DB  68,15,88,5,199,17,0,0               ; addps         0x11c7(%rip),%xmm8        # 5370 <_sk_callback_sse41+0x1017>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -14936,17 +14960,17 @@ _sk_bicubic_n1x_sse41 LABEL PROC
 PUBLIC _sk_bicubic_p1x_sse41
 _sk_bicubic_p1x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,197,17,0,0               ; movaps        0x11c5(%rip),%xmm8        # 5360 <_sk_callback_sse41+0x102b>
+  DB  68,15,40,5,193,17,0,0               ; movaps        0x11c1(%rip),%xmm8        # 5380 <_sk_callback_sse41+0x1027>
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,72,64                      ; movups        0x40(%rax),%xmm9
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
-  DB  68,15,40,21,193,17,0,0              ; movaps        0x11c1(%rip),%xmm10        # 5370 <_sk_callback_sse41+0x103b>
+  DB  68,15,40,21,189,17,0,0              ; movaps        0x11bd(%rip),%xmm10        # 5390 <_sk_callback_sse41+0x1037>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,197,17,0,0              ; addps         0x11c5(%rip),%xmm10        # 5380 <_sk_callback_sse41+0x104b>
+  DB  68,15,88,21,193,17,0,0              ; addps         0x11c1(%rip),%xmm10        # 53a0 <_sk_callback_sse41+0x1047>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,193,17,0,0              ; addps         0x11c1(%rip),%xmm10        # 5390 <_sk_callback_sse41+0x105b>
+  DB  68,15,88,21,189,17,0,0              ; addps         0x11bd(%rip),%xmm10        # 53b0 <_sk_callback_sse41+0x1057>
   DB  68,15,17,144,128,0,0,0              ; movups        %xmm10,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -14956,11 +14980,11 @@ _sk_bicubic_p3x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,180,17,0,0                  ; addps         0x11b4(%rip),%xmm0        # 53a0 <_sk_callback_sse41+0x106b>
+  DB  15,88,5,176,17,0,0                  ; addps         0x11b0(%rip),%xmm0        # 53c0 <_sk_callback_sse41+0x1067>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,180,17,0,0               ; mulps         0x11b4(%rip),%xmm8        # 53b0 <_sk_callback_sse41+0x107b>
-  DB  68,15,88,5,188,17,0,0               ; addps         0x11bc(%rip),%xmm8        # 53c0 <_sk_callback_sse41+0x108b>
+  DB  68,15,89,5,176,17,0,0               ; mulps         0x11b0(%rip),%xmm8        # 53d0 <_sk_callback_sse41+0x1077>
+  DB  68,15,88,5,184,17,0,0               ; addps         0x11b8(%rip),%xmm8        # 53e0 <_sk_callback_sse41+0x1087>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -14971,13 +14995,13 @@ _sk_bicubic_n3y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,170,17,0,0                 ; addps         0x11aa(%rip),%xmm1        # 53d0 <_sk_callback_sse41+0x109b>
-  DB  68,15,40,13,178,17,0,0              ; movaps        0x11b2(%rip),%xmm9        # 53e0 <_sk_callback_sse41+0x10ab>
+  DB  15,88,13,166,17,0,0                 ; addps         0x11a6(%rip),%xmm1        # 53f0 <_sk_callback_sse41+0x1097>
+  DB  68,15,40,13,174,17,0,0              ; movaps        0x11ae(%rip),%xmm9        # 5400 <_sk_callback_sse41+0x10a7>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,174,17,0,0              ; mulps         0x11ae(%rip),%xmm9        # 53f0 <_sk_callback_sse41+0x10bb>
-  DB  68,15,88,13,182,17,0,0              ; addps         0x11b6(%rip),%xmm9        # 5400 <_sk_callback_sse41+0x10cb>
+  DB  68,15,89,13,170,17,0,0              ; mulps         0x11aa(%rip),%xmm9        # 5410 <_sk_callback_sse41+0x10b7>
+  DB  68,15,88,13,178,17,0,0              ; addps         0x11b2(%rip),%xmm9        # 5420 <_sk_callback_sse41+0x10c7>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -14988,16 +15012,16 @@ _sk_bicubic_n1y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,164,17,0,0                 ; addps         0x11a4(%rip),%xmm1        # 5410 <_sk_callback_sse41+0x10db>
-  DB  68,15,40,13,172,17,0,0              ; movaps        0x11ac(%rip),%xmm9        # 5420 <_sk_callback_sse41+0x10eb>
+  DB  15,88,13,160,17,0,0                 ; addps         0x11a0(%rip),%xmm1        # 5430 <_sk_callback_sse41+0x10d7>
+  DB  68,15,40,13,168,17,0,0              ; movaps        0x11a8(%rip),%xmm9        # 5440 <_sk_callback_sse41+0x10e7>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,176,17,0,0               ; movaps        0x11b0(%rip),%xmm8        # 5430 <_sk_callback_sse41+0x10fb>
+  DB  68,15,40,5,172,17,0,0               ; movaps        0x11ac(%rip),%xmm8        # 5450 <_sk_callback_sse41+0x10f7>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,180,17,0,0               ; addps         0x11b4(%rip),%xmm8        # 5440 <_sk_callback_sse41+0x110b>
+  DB  68,15,88,5,176,17,0,0               ; addps         0x11b0(%rip),%xmm8        # 5460 <_sk_callback_sse41+0x1107>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,184,17,0,0               ; addps         0x11b8(%rip),%xmm8        # 5450 <_sk_callback_sse41+0x111b>
+  DB  68,15,88,5,180,17,0,0               ; addps         0x11b4(%rip),%xmm8        # 5470 <_sk_callback_sse41+0x1117>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,188,17,0,0               ; addps         0x11bc(%rip),%xmm8        # 5460 <_sk_callback_sse41+0x112b>
+  DB  68,15,88,5,184,17,0,0               ; addps         0x11b8(%rip),%xmm8        # 5480 <_sk_callback_sse41+0x1127>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15005,17 +15029,17 @@ _sk_bicubic_n1y_sse41 LABEL PROC
 PUBLIC _sk_bicubic_p1y_sse41
 _sk_bicubic_p1y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,182,17,0,0               ; movaps        0x11b6(%rip),%xmm8        # 5470 <_sk_callback_sse41+0x113b>
+  DB  68,15,40,5,178,17,0,0               ; movaps        0x11b2(%rip),%xmm8        # 5490 <_sk_callback_sse41+0x1137>
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,72,96                      ; movups        0x60(%rax),%xmm9
   DB  65,15,88,200                        ; addps         %xmm8,%xmm1
-  DB  68,15,40,21,177,17,0,0              ; movaps        0x11b1(%rip),%xmm10        # 5480 <_sk_callback_sse41+0x114b>
+  DB  68,15,40,21,173,17,0,0              ; movaps        0x11ad(%rip),%xmm10        # 54a0 <_sk_callback_sse41+0x1147>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,181,17,0,0              ; addps         0x11b5(%rip),%xmm10        # 5490 <_sk_callback_sse41+0x115b>
+  DB  68,15,88,21,177,17,0,0              ; addps         0x11b1(%rip),%xmm10        # 54b0 <_sk_callback_sse41+0x1157>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,177,17,0,0              ; addps         0x11b1(%rip),%xmm10        # 54a0 <_sk_callback_sse41+0x116b>
+  DB  68,15,88,21,173,17,0,0              ; addps         0x11ad(%rip),%xmm10        # 54c0 <_sk_callback_sse41+0x1167>
   DB  68,15,17,144,160,0,0,0              ; movups        %xmm10,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15025,11 +15049,11 @@ _sk_bicubic_p3y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,163,17,0,0                 ; addps         0x11a3(%rip),%xmm1        # 54b0 <_sk_callback_sse41+0x117b>
+  DB  15,88,13,159,17,0,0                 ; addps         0x119f(%rip),%xmm1        # 54d0 <_sk_callback_sse41+0x1177>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,163,17,0,0               ; mulps         0x11a3(%rip),%xmm8        # 54c0 <_sk_callback_sse41+0x118b>
-  DB  68,15,88,5,171,17,0,0               ; addps         0x11ab(%rip),%xmm8        # 54d0 <_sk_callback_sse41+0x119b>
+  DB  68,15,89,5,159,17,0,0               ; mulps         0x119f(%rip),%xmm8        # 54e0 <_sk_callback_sse41+0x1187>
+  DB  68,15,88,5,167,17,0,0               ; addps         0x11a7(%rip),%xmm8        # 54f0 <_sk_callback_sse41+0x1197>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15234,11 +15258,11 @@ ALIGN 16
   DB  128,191,0,0,128,191,0               ; cmpb          $0x0,-0x40800000(%rdi)
   DB  0,224                               ; add           %ah,%al
   DB  64,0,0                              ; add           %al,(%rax)
-  DB  224,64                              ; loopne        45b8 <.literal16+0x1d8>
+  DB  224,64                              ; loopne        45e8 <.literal16+0x1d8>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        45bc <.literal16+0x1dc>
+  DB  224,64                              ; loopne        45ec <.literal16+0x1dc>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        45c0 <.literal16+0x1e0>
+  DB  224,64                              ; loopne        45f0 <.literal16+0x1e0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -15263,13 +15287,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 45f1 <.literal16+0x211>
+  DB  71,225,61                           ; rex.RXB       loope 4621 <.literal16+0x211>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 45f5 <.literal16+0x215>
+  DB  71,225,61                           ; rex.RXB       loope 4625 <.literal16+0x215>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 45f9 <.literal16+0x219>
+  DB  71,225,61                           ; rex.RXB       loope 4629 <.literal16+0x219>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 45fd <.literal16+0x21d>
+  DB  71,225,61                           ; rex.RXB       loope 462d <.literal16+0x21d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -15294,13 +15318,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4631 <.literal16+0x251>
+  DB  71,225,61                           ; rex.RXB       loope 4661 <.literal16+0x251>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4635 <.literal16+0x255>
+  DB  71,225,61                           ; rex.RXB       loope 4665 <.literal16+0x255>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4639 <.literal16+0x259>
+  DB  71,225,61                           ; rex.RXB       loope 4669 <.literal16+0x259>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 463d <.literal16+0x25d>
+  DB  71,225,61                           ; rex.RXB       loope 466d <.literal16+0x25d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -15325,13 +15349,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4671 <.literal16+0x291>
+  DB  71,225,61                           ; rex.RXB       loope 46a1 <.literal16+0x291>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4675 <.literal16+0x295>
+  DB  71,225,61                           ; rex.RXB       loope 46a5 <.literal16+0x295>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4679 <.literal16+0x299>
+  DB  71,225,61                           ; rex.RXB       loope 46a9 <.literal16+0x299>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 467d <.literal16+0x29d>
+  DB  71,225,61                           ; rex.RXB       loope 46ad <.literal16+0x29d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -15356,13 +15380,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 46b1 <.literal16+0x2d1>
+  DB  71,225,61                           ; rex.RXB       loope 46e1 <.literal16+0x2d1>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 46b5 <.literal16+0x2d5>
+  DB  71,225,61                           ; rex.RXB       loope 46e5 <.literal16+0x2d5>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 46b9 <.literal16+0x2d9>
+  DB  71,225,61                           ; rex.RXB       loope 46e9 <.literal16+0x2d9>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 46bd <.literal16+0x2dd>
+  DB  71,225,61                           ; rex.RXB       loope 46ed <.literal16+0x2dd>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -15581,13 +15605,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4879 <.literal16+0x499>
+  DB  224,7                               ; loopne        48a9 <.literal16+0x499>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        487d <.literal16+0x49d>
+  DB  224,7                               ; loopne        48ad <.literal16+0x49d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4881 <.literal16+0x4a1>
+  DB  224,7                               ; loopne        48b1 <.literal16+0x4a1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4885 <.literal16+0x4a5>
+  DB  224,7                               ; loopne        48b5 <.literal16+0x4a5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -15611,26 +15635,20 @@ ALIGN 16
   DB  4,61                                ; add           $0x3d,%al
   DB  8,33                                ; or            %ah,(%rcx)
   DB  4,61                                ; add           $0x3d,%al
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  128,63,0                            ; cmpb          $0x0,(%rdi)
-  DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
-  DB  63                                  ; (bad)
-  DB  0,0                                 ; add           %al,(%rax)
-  DB  128,63,255                          ; cmpb          $0xff,(%rdi)
-  DB  0,0                                 ; add           %al,(%rax)
-  DB  0,255                               ; add           %bh,%bh
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  0,255                               ; add           %bh,%bh
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  0,255                               ; add           %bh,%bh
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  0,1                                 ; add           %al,(%rcx)
-  DB  255                                 ; (bad)
+  DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a0048d8 <_sk_callback_sse41+0xa0005a3>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a0048f8 <_sk_callback_sse41+0xa00059f>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 30048e0 <_sk_callback_sse41+0x30005ab>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3004900 <_sk_callback_sse41+0x30005a7>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -15685,11 +15703,11 @@ ALIGN 16
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            49ab <.literal16+0x5cb>
+  DB  127,67                              ; jg            49cb <.literal16+0x5bb>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            49af <.literal16+0x5cf>
+  DB  127,67                              ; jg            49cf <.literal16+0x5bf>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            49b3 <.literal16+0x5d3>
+  DB  127,67                              ; jg            49d3 <.literal16+0x5c3>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,129,128,128,59           ; addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -15704,16 +15722,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            49a4 <.literal16+0x5c4>
+  DB  127,0                               ; jg            49c4 <.literal16+0x5b4>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            49a8 <.literal16+0x5c8>
+  DB  127,0                               ; jg            49c8 <.literal16+0x5b8>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            49ac <.literal16+0x5cc>
+  DB  127,0                               ; jg            49cc <.literal16+0x5bc>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            49b0 <.literal16+0x5d0>
+  DB  127,0                               ; jg            49d0 <.literal16+0x5c0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -15722,7 +15740,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4a35 <.literal16+0x655>
+  DB  119,115                             ; ja            4a55 <.literal16+0x645>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -15733,7 +15751,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4999 <.literal16+0x5b9>
+  DB  117,191                             ; jne           49b9 <.literal16+0x5a9>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -15745,7 +15763,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a389da <_sk_callback_sse41+0xffffffffe9a346a5>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a389fa <_sk_callback_sse41+0xffffffffe9a346a1>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -15800,16 +15818,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4a74 <.literal16+0x694>
+  DB  127,0                               ; jg            4a94 <.literal16+0x684>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4a78 <.literal16+0x698>
+  DB  127,0                               ; jg            4a98 <.literal16+0x688>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4a7c <.literal16+0x69c>
+  DB  127,0                               ; jg            4a9c <.literal16+0x68c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4a80 <.literal16+0x6a0>
+  DB  127,0                               ; jg            4aa0 <.literal16+0x690>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -15818,7 +15836,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4b05 <.literal16+0x725>
+  DB  119,115                             ; ja            4b25 <.literal16+0x715>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -15829,7 +15847,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4a69 <.literal16+0x689>
+  DB  117,191                             ; jne           4a89 <.literal16+0x679>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -15841,7 +15859,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38aaa <_sk_callback_sse41+0xffffffffe9a34775>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38aca <_sk_callback_sse41+0xffffffffe9a34771>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -15896,16 +15914,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4b44 <.literal16+0x764>
+  DB  127,0                               ; jg            4b64 <.literal16+0x754>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4b48 <.literal16+0x768>
+  DB  127,0                               ; jg            4b68 <.literal16+0x758>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4b4c <.literal16+0x76c>
+  DB  127,0                               ; jg            4b6c <.literal16+0x75c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4b50 <.literal16+0x770>
+  DB  127,0                               ; jg            4b70 <.literal16+0x760>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -15914,7 +15932,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4bd5 <.literal16+0x7f5>
+  DB  119,115                             ; ja            4bf5 <.literal16+0x7e5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -15925,7 +15943,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4b39 <.literal16+0x759>
+  DB  117,191                             ; jne           4b59 <.literal16+0x749>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -15937,7 +15955,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38b7a <_sk_callback_sse41+0xffffffffe9a34845>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38b9a <_sk_callback_sse41+0xffffffffe9a34841>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -15992,16 +16010,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4c14 <.literal16+0x834>
+  DB  127,0                               ; jg            4c34 <.literal16+0x824>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4c18 <.literal16+0x838>
+  DB  127,0                               ; jg            4c38 <.literal16+0x828>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4c1c <.literal16+0x83c>
+  DB  127,0                               ; jg            4c3c <.literal16+0x82c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4c20 <.literal16+0x840>
+  DB  127,0                               ; jg            4c40 <.literal16+0x830>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16010,7 +16028,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4ca5 <.literal16+0x8c5>
+  DB  119,115                             ; ja            4cc5 <.literal16+0x8b5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -16021,7 +16039,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4c09 <.literal16+0x829>
+  DB  117,191                             ; jne           4c29 <.literal16+0x819>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -16033,7 +16051,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38c4a <_sk_callback_sse41+0xffffffffe9a34915>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38c6a <_sk_callback_sse41+0xffffffffe9a34911>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -16084,13 +16102,13 @@ ALIGN 16
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
-  DB  127,67                              ; jg            4d27 <.literal16+0x947>
+  DB  127,67                              ; jg            4d47 <.literal16+0x937>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4d2b <.literal16+0x94b>
+  DB  127,67                              ; jg            4d4b <.literal16+0x93b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4d2f <.literal16+0x94f>
+  DB  127,67                              ; jg            4d4f <.literal16+0x93f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4d33 <.literal16+0x953>
+  DB  127,67                              ; jg            4d53 <.literal16+0x943>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -16137,16 +16155,16 @@ ALIGN 16
   DB  128,3,62                            ; addb          $0x3e,(%rbx)
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           4db3 <.literal16+0x9d3>
+  DB  118,63                              ; jbe           4dd3 <.literal16+0x9c3>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           4db7 <.literal16+0x9d7>
+  DB  118,63                              ; jbe           4dd7 <.literal16+0x9c7>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           4dbb <.literal16+0x9db>
+  DB  118,63                              ; jbe           4ddb <.literal16+0x9cb>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           4dbf <.literal16+0x9df>
+  DB  118,63                              ; jbe           4ddf <.literal16+0x9cf>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
@@ -16158,11 +16176,11 @@ ALIGN 16
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4dfb <.literal16+0xa1b>
+  DB  127,67                              ; jg            4e1b <.literal16+0xa0b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4dff <.literal16+0xa1f>
+  DB  127,67                              ; jg            4e1f <.literal16+0xa0f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4e03 <.literal16+0xa23>
+  DB  127,67                              ; jg            4e23 <.literal16+0xa13>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,0,0,128,63               ; addb          $0x3f,-0x7fffffc5(%rax)
@@ -16191,7 +16209,7 @@ ALIGN 16
   DB  5,255,255,255,9                     ; add           $0x9ffffff,%eax
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3004e30 <_sk_callback_sse41+0x3000afb>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3004e50 <_sk_callback_sse41+0x3000af7>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -16220,13 +16238,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4e69 <.literal16+0xa89>
+  DB  224,7                               ; loopne        4e89 <.literal16+0xa79>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4e6d <.literal16+0xa8d>
+  DB  224,7                               ; loopne        4e8d <.literal16+0xa7d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4e71 <.literal16+0xa91>
+  DB  224,7                               ; loopne        4e91 <.literal16+0xa81>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4e75 <.literal16+0xa95>
+  DB  224,7                               ; loopne        4e95 <.literal16+0xa85>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -16272,13 +16290,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4ed9 <.literal16+0xaf9>
+  DB  224,7                               ; loopne        4ef9 <.literal16+0xae9>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4edd <.literal16+0xafd>
+  DB  224,7                               ; loopne        4efd <.literal16+0xaed>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4ee1 <.literal16+0xb01>
+  DB  224,7                               ; loopne        4f01 <.literal16+0xaf1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4ee5 <.literal16+0xb05>
+  DB  224,7                               ; loopne        4f05 <.literal16+0xaf5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -16316,13 +16334,13 @@ ALIGN 16
   DB  65,0,0                              ; add           %al,(%r8)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            4f76 <.literal16+0xb96>
+  DB  124,66                              ; jl            4f96 <.literal16+0xb86>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            4f7a <.literal16+0xb9a>
+  DB  124,66                              ; jl            4f9a <.literal16+0xb8a>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            4f7e <.literal16+0xb9e>
+  DB  124,66                              ; jl            4f9e <.literal16+0xb8e>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            4f82 <.literal16+0xba2>
+  DB  124,66                              ; jl            4fa2 <.literal16+0xb92>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,240                               ; add           %dh,%al
@@ -16412,13 +16430,13 @@ ALIGN 16
   DB  136,136,61,137,136,136              ; mov           %cl,-0x777776c3(%rax)
   DB  61,137,136,136,61                   ; cmp           $0x3d888889,%eax
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5085 <.literal16+0xca5>
+  DB  112,65                              ; jo            50a5 <.literal16+0xc95>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5089 <.literal16+0xca9>
+  DB  112,65                              ; jo            50a9 <.literal16+0xc99>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            508d <.literal16+0xcad>
+  DB  112,65                              ; jo            50ad <.literal16+0xc9d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5091 <.literal16+0xcb1>
+  DB  112,65                              ; jo            50b1 <.literal16+0xca1>
   DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  255,0                               ; incl          (%rax)
@@ -16433,7 +16451,7 @@ ALIGN 16
   DB  5,255,255,255,9                     ; add           $0x9ffffff,%eax
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3005080 <_sk_callback_sse41+0x3000d4b>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 30050a0 <_sk_callback_sse41+0x3000d47>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -16460,7 +16478,7 @@ ALIGN 16
   DB  5,255,255,255,9                     ; add           $0x9ffffff,%eax
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 30050c0 <_sk_callback_sse41+0x3000d8b>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 30050e0 <_sk_callback_sse41+0x3000d87>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -16475,11 +16493,11 @@ ALIGN 16
   DB  255,0                               ; incl          (%rax)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            511b <.literal16+0xd3b>
+  DB  127,67                              ; jg            513b <.literal16+0xd2b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            511f <.literal16+0xd3f>
+  DB  127,67                              ; jg            513f <.literal16+0xd2f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            5123 <.literal16+0xd43>
+  DB  127,67                              ; jg            5143 <.literal16+0xd33>
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
@@ -16555,13 +16573,13 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  255                                 ; (bad)
-  DB  127,71                              ; jg            51eb <.literal16+0xe0b>
+  DB  127,71                              ; jg            520b <.literal16+0xdfb>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            51ef <.literal16+0xe0f>
+  DB  127,71                              ; jg            520f <.literal16+0xdff>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            51f3 <.literal16+0xe13>
+  DB  127,71                              ; jg            5213 <.literal16+0xe03>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            51f7 <.literal16+0xe17>
+  DB  127,71                              ; jg            5217 <.literal16+0xe07>
   DB  208                                 ; (bad)
   DB  179,89                              ; mov           $0x59,%bl
   DB  62,208                              ; ds            (bad)
@@ -16687,11 +16705,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         5302 <.literal16+0xf22>
+  DB  62,114,28                           ; jb,pt         5322 <.literal16+0xf12>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5306 <.literal16+0xf26>
+  DB  62,114,28                           ; jb,pt         5326 <.literal16+0xf16>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         530a <.literal16+0xf2a>
+  DB  62,114,28                           ; jb,pt         532a <.literal16+0xf1a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -16735,7 +16753,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e195 <_sk_callback_sse41+0x3d639e60>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e1b5 <_sk_callback_sse41+0x3d639e5c>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16761,7 +16779,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e1d5 <_sk_callback_sse41+0x3d639ea0>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e1f5 <_sk_callback_sse41+0x3d639e9c>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -16770,13 +16788,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            53ce <.literal16+0xfee>
+  DB  114,28                              ; jb            53ee <.literal16+0xfde>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         53d2 <.literal16+0xff2>
+  DB  62,114,28                           ; jb,pt         53f2 <.literal16+0xfe2>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         53d6 <.literal16+0xff6>
+  DB  62,114,28                           ; jb,pt         53f6 <.literal16+0xfe6>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         53da <.literal16+0xffa>
+  DB  62,114,28                           ; jb,pt         53fa <.literal16+0xfea>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -16797,11 +16815,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         5412 <.literal16+0x1032>
+  DB  62,114,28                           ; jb,pt         5432 <.literal16+0x1022>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5416 <.literal16+0x1036>
+  DB  62,114,28                           ; jb,pt         5436 <.literal16+0x1026>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         541a <.literal16+0x103a>
+  DB  62,114,28                           ; jb,pt         543a <.literal16+0x102a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -16845,7 +16863,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e2a5 <_sk_callback_sse41+0x3d639f70>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e2c5 <_sk_callback_sse41+0x3d639f6c>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16871,7 +16889,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e2e5 <_sk_callback_sse41+0x3d639fb0>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e305 <_sk_callback_sse41+0x3d639fac>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -16880,13 +16898,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            54de <.literal16+0x10fe>
+  DB  114,28                              ; jb            54fe <.literal16+0x10ee>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         54e2 <_sk_callback_sse41+0x11ad>
+  DB  62,114,28                           ; jb,pt         5502 <_sk_callback_sse41+0x11a9>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         54e6 <_sk_callback_sse41+0x11b1>
+  DB  62,114,28                           ; jb,pt         5506 <_sk_callback_sse41+0x11ad>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         54ea <_sk_callback_sse41+0x11b5>
+  DB  62,114,28                           ; jb,pt         550a <_sk_callback_sse41+0x11b1>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -16977,7 +16995,7 @@ _sk_seed_shader_sse2 LABEL PROC
   DB  102,15,110,199                      ; movd          %edi,%xmm0
   DB  102,15,112,192,0                    ; pshufd        $0x0,%xmm0,%xmm0
   DB  15,91,200                           ; cvtdq2ps      %xmm0,%xmm1
-  DB  15,40,21,161,71,0,0                 ; movaps        0x47a1(%rip),%xmm2        # 48b0 <_sk_callback_sse2+0xad>
+  DB  15,40,21,209,71,0,0                 ; movaps        0x47d1(%rip),%xmm2        # 48e0 <_sk_callback_sse2+0xb8>
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  15,16,2                             ; movups        (%rdx),%xmm0
   DB  15,88,193                           ; addps         %xmm1,%xmm0
@@ -16986,7 +17004,7 @@ _sk_seed_shader_sse2 LABEL PROC
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,21,144,71,0,0                 ; movaps        0x4790(%rip),%xmm2        # 48c0 <_sk_callback_sse2+0xbd>
+  DB  15,40,21,192,71,0,0                 ; movaps        0x47c0(%rip),%xmm2        # 48f0 <_sk_callback_sse2+0xc8>
   DB  15,87,219                           ; xorps         %xmm3,%xmm3
   DB  15,87,228                           ; xorps         %xmm4,%xmm4
   DB  15,87,237                           ; xorps         %xmm5,%xmm5
@@ -17007,14 +17025,14 @@ _sk_dither_sse2 LABEL PROC
   DB  102,68,15,110,1                     ; movd          (%rcx),%xmm8
   DB  102,69,15,112,192,0                 ; pshufd        $0x0,%xmm8,%xmm8
   DB  102,69,15,239,193                   ; pxor          %xmm9,%xmm8
-  DB  102,68,15,111,21,85,71,0,0          ; movdqa        0x4755(%rip),%xmm10        # 48d0 <_sk_callback_sse2+0xcd>
+  DB  102,68,15,111,21,133,71,0,0         ; movdqa        0x4785(%rip),%xmm10        # 4900 <_sk_callback_sse2+0xd8>
   DB  102,69,15,111,216                   ; movdqa        %xmm8,%xmm11
   DB  102,69,15,219,218                   ; pand          %xmm10,%xmm11
   DB  102,65,15,114,243,5                 ; pslld         $0x5,%xmm11
   DB  102,69,15,219,209                   ; pand          %xmm9,%xmm10
   DB  102,65,15,114,242,4                 ; pslld         $0x4,%xmm10
-  DB  102,68,15,111,37,65,71,0,0          ; movdqa        0x4741(%rip),%xmm12        # 48e0 <_sk_callback_sse2+0xdd>
-  DB  102,68,15,111,45,72,71,0,0          ; movdqa        0x4748(%rip),%xmm13        # 48f0 <_sk_callback_sse2+0xed>
+  DB  102,68,15,111,37,113,71,0,0         ; movdqa        0x4771(%rip),%xmm12        # 4910 <_sk_callback_sse2+0xe8>
+  DB  102,68,15,111,45,120,71,0,0         ; movdqa        0x4778(%rip),%xmm13        # 4920 <_sk_callback_sse2+0xf8>
   DB  102,69,15,111,240                   ; movdqa        %xmm8,%xmm14
   DB  102,69,15,219,245                   ; pand          %xmm13,%xmm14
   DB  102,65,15,114,246,2                 ; pslld         $0x2,%xmm14
@@ -17030,8 +17048,8 @@ _sk_dither_sse2 LABEL PROC
   DB  102,69,15,235,245                   ; por           %xmm13,%xmm14
   DB  102,69,15,235,240                   ; por           %xmm8,%xmm14
   DB  69,15,91,198                        ; cvtdq2ps      %xmm14,%xmm8
-  DB  68,15,89,5,3,71,0,0                 ; mulps         0x4703(%rip),%xmm8        # 4900 <_sk_callback_sse2+0xfd>
-  DB  68,15,88,5,11,71,0,0                ; addps         0x470b(%rip),%xmm8        # 4910 <_sk_callback_sse2+0x10d>
+  DB  68,15,89,5,51,71,0,0                ; mulps         0x4733(%rip),%xmm8        # 4930 <_sk_callback_sse2+0x108>
+  DB  68,15,88,5,59,71,0,0                ; addps         0x473b(%rip),%xmm8        # 4940 <_sk_callback_sse2+0x118>
   DB  243,68,15,16,72,8                   ; movss         0x8(%rax),%xmm9
   DB  69,15,198,201,0                     ; shufps        $0x0,%xmm9,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
@@ -17067,7 +17085,7 @@ _sk_clear_sse2 LABEL PROC
 PUBLIC _sk_srcatop_sse2
 _sk_srcatop_sse2 LABEL PROC
   DB  15,89,199                           ; mulps         %xmm7,%xmm0
-  DB  68,15,40,5,184,70,0,0               ; movaps        0x46b8(%rip),%xmm8        # 4920 <_sk_callback_sse2+0x11d>
+  DB  68,15,40,5,232,70,0,0               ; movaps        0x46e8(%rip),%xmm8        # 4950 <_sk_callback_sse2+0x128>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -17090,7 +17108,7 @@ PUBLIC _sk_dstatop_sse2
 _sk_dstatop_sse2 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
   DB  68,15,89,196                        ; mulps         %xmm4,%xmm8
-  DB  68,15,40,13,123,70,0,0              ; movaps        0x467b(%rip),%xmm9        # 4930 <_sk_callback_sse2+0x12d>
+  DB  68,15,40,13,171,70,0,0              ; movaps        0x46ab(%rip),%xmm9        # 4960 <_sk_callback_sse2+0x138>
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
@@ -17131,7 +17149,7 @@ _sk_dstin_sse2 LABEL PROC
 
 PUBLIC _sk_srcout_sse2
 _sk_srcout_sse2 LABEL PROC
-  DB  68,15,40,5,31,70,0,0                ; movaps        0x461f(%rip),%xmm8        # 4940 <_sk_callback_sse2+0x13d>
+  DB  68,15,40,5,79,70,0,0                ; movaps        0x464f(%rip),%xmm8        # 4970 <_sk_callback_sse2+0x148>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
@@ -17142,7 +17160,7 @@ _sk_srcout_sse2 LABEL PROC
 
 PUBLIC _sk_dstout_sse2
 _sk_dstout_sse2 LABEL PROC
-  DB  68,15,40,5,15,70,0,0                ; movaps        0x460f(%rip),%xmm8        # 4950 <_sk_callback_sse2+0x14d>
+  DB  68,15,40,5,63,70,0,0                ; movaps        0x463f(%rip),%xmm8        # 4980 <_sk_callback_sse2+0x158>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  15,89,196                           ; mulps         %xmm4,%xmm0
@@ -17157,7 +17175,7 @@ _sk_dstout_sse2 LABEL PROC
 
 PUBLIC _sk_srcover_sse2
 _sk_srcover_sse2 LABEL PROC
-  DB  68,15,40,5,242,69,0,0               ; movaps        0x45f2(%rip),%xmm8        # 4960 <_sk_callback_sse2+0x15d>
+  DB  68,15,40,5,34,70,0,0                ; movaps        0x4622(%rip),%xmm8        # 4990 <_sk_callback_sse2+0x168>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -17175,7 +17193,7 @@ _sk_srcover_sse2 LABEL PROC
 
 PUBLIC _sk_dstover_sse2
 _sk_dstover_sse2 LABEL PROC
-  DB  68,15,40,5,198,69,0,0               ; movaps        0x45c6(%rip),%xmm8        # 4970 <_sk_callback_sse2+0x16d>
+  DB  68,15,40,5,246,69,0,0               ; movaps        0x45f6(%rip),%xmm8        # 49a0 <_sk_callback_sse2+0x178>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -17199,7 +17217,7 @@ _sk_modulate_sse2 LABEL PROC
 
 PUBLIC _sk_multiply_sse2
 _sk_multiply_sse2 LABEL PROC
-  DB  68,15,40,5,154,69,0,0               ; movaps        0x459a(%rip),%xmm8        # 4980 <_sk_callback_sse2+0x17d>
+  DB  68,15,40,5,202,69,0,0               ; movaps        0x45ca(%rip),%xmm8        # 49b0 <_sk_callback_sse2+0x188>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
@@ -17269,7 +17287,7 @@ _sk_screen_sse2 LABEL PROC
 PUBLIC _sk_xor__sse2
 _sk_xor__sse2 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
-  DB  15,40,29,203,68,0,0                 ; movaps        0x44cb(%rip),%xmm3        # 4990 <_sk_callback_sse2+0x18d>
+  DB  15,40,29,251,68,0,0                 ; movaps        0x44fb(%rip),%xmm3        # 49c0 <_sk_callback_sse2+0x198>
   DB  68,15,40,203                        ; movaps        %xmm3,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
@@ -17315,7 +17333,7 @@ _sk_darken_sse2 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,95,209                        ; maxps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,54,68,0,0                  ; movaps        0x4436(%rip),%xmm2        # 49a0 <_sk_callback_sse2+0x19d>
+  DB  15,40,21,102,68,0,0                 ; movaps        0x4466(%rip),%xmm2        # 49d0 <_sk_callback_sse2+0x1a8>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -17347,7 +17365,7 @@ _sk_lighten_sse2 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,219,67,0,0                 ; movaps        0x43db(%rip),%xmm2        # 49b0 <_sk_callback_sse2+0x1ad>
+  DB  15,40,21,11,68,0,0                  ; movaps        0x440b(%rip),%xmm2        # 49e0 <_sk_callback_sse2+0x1b8>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -17382,7 +17400,7 @@ _sk_difference_sse2 LABEL PROC
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,117,67,0,0                 ; movaps        0x4375(%rip),%xmm2        # 49c0 <_sk_callback_sse2+0x1bd>
+  DB  15,40,21,165,67,0,0                 ; movaps        0x43a5(%rip),%xmm2        # 49f0 <_sk_callback_sse2+0x1c8>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -17407,7 +17425,7 @@ _sk_exclusion_sse2 LABEL PROC
   DB  15,89,214                           ; mulps         %xmm6,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,202                        ; subps         %xmm2,%xmm9
-  DB  15,40,13,54,67,0,0                  ; movaps        0x4336(%rip),%xmm1        # 49d0 <_sk_callback_sse2+0x1cd>
+  DB  15,40,13,102,67,0,0                 ; movaps        0x4366(%rip),%xmm1        # 4a00 <_sk_callback_sse2+0x1d8>
   DB  15,92,203                           ; subps         %xmm3,%xmm1
   DB  15,89,207                           ; mulps         %xmm7,%xmm1
   DB  15,88,217                           ; addps         %xmm1,%xmm3
@@ -17419,7 +17437,7 @@ _sk_exclusion_sse2 LABEL PROC
 PUBLIC _sk_colorburn_sse2
 _sk_colorburn_sse2 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,40,21,37,67,0,0               ; movaps        0x4325(%rip),%xmm10        # 49e0 <_sk_callback_sse2+0x1dd>
+  DB  68,15,40,21,85,67,0,0               ; movaps        0x4355(%rip),%xmm10        # 4a10 <_sk_callback_sse2+0x1e8>
   DB  69,15,40,202                        ; movaps        %xmm10,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  69,15,40,217                        ; movaps        %xmm9,%xmm11
@@ -17511,7 +17529,7 @@ _sk_colorburn_sse2 LABEL PROC
 PUBLIC _sk_colordodge_sse2
 _sk_colordodge_sse2 LABEL PROC
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
-  DB  68,15,40,21,219,65,0,0              ; movaps        0x41db(%rip),%xmm10        # 49f0 <_sk_callback_sse2+0x1ed>
+  DB  68,15,40,21,11,66,0,0               ; movaps        0x420b(%rip),%xmm10        # 4a20 <_sk_callback_sse2+0x1f8>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  69,15,40,227                        ; movaps        %xmm11,%xmm12
@@ -17604,7 +17622,7 @@ _sk_hardlight_sse2 LABEL PROC
   DB  15,41,52,36                         ; movaps        %xmm6,(%rsp)
   DB  15,40,245                           ; movaps        %xmm5,%xmm6
   DB  15,40,236                           ; movaps        %xmm4,%xmm5
-  DB  68,15,40,29,141,64,0,0              ; movaps        0x408d(%rip),%xmm11        # 4a00 <_sk_callback_sse2+0x1fd>
+  DB  68,15,40,29,189,64,0,0              ; movaps        0x40bd(%rip),%xmm11        # 4a30 <_sk_callback_sse2+0x208>
   DB  69,15,40,211                        ; movaps        %xmm11,%xmm10
   DB  68,15,92,215                        ; subps         %xmm7,%xmm10
   DB  69,15,40,194                        ; movaps        %xmm10,%xmm8
@@ -17691,7 +17709,7 @@ PUBLIC _sk_overlay_sse2
 _sk_overlay_sse2 LABEL PROC
   DB  68,15,40,193                        ; movaps        %xmm1,%xmm8
   DB  68,15,40,232                        ; movaps        %xmm0,%xmm13
-  DB  68,15,40,13,88,63,0,0               ; movaps        0x3f58(%rip),%xmm9        # 4a10 <_sk_callback_sse2+0x20d>
+  DB  68,15,40,13,136,63,0,0              ; movaps        0x3f88(%rip),%xmm9        # 4a40 <_sk_callback_sse2+0x218>
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
   DB  68,15,92,215                        ; subps         %xmm7,%xmm10
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
@@ -17781,7 +17799,7 @@ _sk_softlight_sse2 LABEL PROC
   DB  68,15,40,213                        ; movaps        %xmm5,%xmm10
   DB  68,15,94,215                        ; divps         %xmm7,%xmm10
   DB  69,15,84,212                        ; andps         %xmm12,%xmm10
-  DB  68,15,40,13,18,62,0,0               ; movaps        0x3e12(%rip),%xmm9        # 4a20 <_sk_callback_sse2+0x21d>
+  DB  68,15,40,13,66,62,0,0               ; movaps        0x3e42(%rip),%xmm9        # 4a50 <_sk_callback_sse2+0x228>
   DB  69,15,40,249                        ; movaps        %xmm9,%xmm15
   DB  69,15,92,250                        ; subps         %xmm10,%xmm15
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
@@ -17794,10 +17812,10 @@ _sk_softlight_sse2 LABEL PROC
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  15,89,192                           ; mulps         %xmm0,%xmm0
   DB  65,15,88,194                        ; addps         %xmm10,%xmm0
-  DB  68,15,40,53,236,61,0,0              ; movaps        0x3dec(%rip),%xmm14        # 4a30 <_sk_callback_sse2+0x22d>
+  DB  68,15,40,53,28,62,0,0               ; movaps        0x3e1c(%rip),%xmm14        # 4a60 <_sk_callback_sse2+0x238>
   DB  69,15,88,222                        ; addps         %xmm14,%xmm11
   DB  68,15,89,216                        ; mulps         %xmm0,%xmm11
-  DB  68,15,40,21,236,61,0,0              ; movaps        0x3dec(%rip),%xmm10        # 4a40 <_sk_callback_sse2+0x23d>
+  DB  68,15,40,21,28,62,0,0               ; movaps        0x3e1c(%rip),%xmm10        # 4a70 <_sk_callback_sse2+0x248>
   DB  69,15,89,234                        ; mulps         %xmm10,%xmm13
   DB  69,15,88,235                        ; addps         %xmm11,%xmm13
   DB  15,88,228                           ; addps         %xmm4,%xmm4
@@ -17943,7 +17961,7 @@ _sk_hue_sse2 LABEL PROC
   DB  15,40,236                           ; movaps        %xmm4,%xmm5
   DB  15,40,227                           ; movaps        %xmm3,%xmm4
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
-  DB  68,15,40,13,248,59,0,0              ; movaps        0x3bf8(%rip),%xmm9        # 4a50 <_sk_callback_sse2+0x24d>
+  DB  68,15,40,13,40,60,0,0               ; movaps        0x3c28(%rip),%xmm9        # 4a80 <_sk_callback_sse2+0x258>
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
   DB  68,15,94,212                        ; divps         %xmm4,%xmm10
   DB  68,15,40,228                        ; movaps        %xmm4,%xmm12
@@ -17985,12 +18003,12 @@ _sk_hue_sse2 LABEL PROC
   DB  65,15,84,199                        ; andps         %xmm15,%xmm0
   DB  65,15,84,207                        ; andps         %xmm15,%xmm1
   DB  69,15,84,231                        ; andps         %xmm15,%xmm12
-  DB  68,15,40,61,93,59,0,0               ; movaps        0x3b5d(%rip),%xmm15        # 4a60 <_sk_callback_sse2+0x25d>
+  DB  68,15,40,61,141,59,0,0              ; movaps        0x3b8d(%rip),%xmm15        # 4a90 <_sk_callback_sse2+0x268>
   DB  69,15,89,247                        ; mulps         %xmm15,%xmm14
-  DB  15,40,29,98,59,0,0                  ; movaps        0x3b62(%rip),%xmm3        # 4a70 <_sk_callback_sse2+0x26d>
+  DB  15,40,29,146,59,0,0                 ; movaps        0x3b92(%rip),%xmm3        # 4aa0 <_sk_callback_sse2+0x278>
   DB  68,15,89,235                        ; mulps         %xmm3,%xmm13
   DB  69,15,88,238                        ; addps         %xmm14,%xmm13
-  DB  68,15,40,21,98,59,0,0               ; movaps        0x3b62(%rip),%xmm10        # 4a80 <_sk_callback_sse2+0x27d>
+  DB  68,15,40,21,146,59,0,0              ; movaps        0x3b92(%rip),%xmm10        # 4ab0 <_sk_callback_sse2+0x288>
   DB  68,15,40,223                        ; movaps        %xmm7,%xmm11
   DB  69,15,89,218                        ; mulps         %xmm10,%xmm11
   DB  69,15,88,221                        ; addps         %xmm13,%xmm11
@@ -18106,7 +18124,7 @@ _sk_saturation_sse2 LABEL PROC
   DB  68,15,40,193                        ; movaps        %xmm1,%xmm8
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  15,87,201                           ; xorps         %xmm1,%xmm1
-  DB  68,15,40,29,193,57,0,0              ; movaps        0x39c1(%rip),%xmm11        # 4a90 <_sk_callback_sse2+0x28d>
+  DB  68,15,40,29,241,57,0,0              ; movaps        0x39f1(%rip),%xmm11        # 4ac0 <_sk_callback_sse2+0x298>
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  15,94,199                           ; divps         %xmm7,%xmm0
   DB  68,15,40,231                        ; movaps        %xmm7,%xmm12
@@ -18146,14 +18164,14 @@ _sk_saturation_sse2 LABEL PROC
   DB  15,84,194                           ; andps         %xmm2,%xmm0
   DB  68,15,84,250                        ; andps         %xmm2,%xmm15
   DB  68,15,84,226                        ; andps         %xmm2,%xmm12
-  DB  68,15,40,45,49,57,0,0               ; movaps        0x3931(%rip),%xmm13        # 4aa0 <_sk_callback_sse2+0x29d>
+  DB  68,15,40,45,97,57,0,0               ; movaps        0x3961(%rip),%xmm13        # 4ad0 <_sk_callback_sse2+0x2a8>
   DB  68,15,40,197                        ; movaps        %xmm5,%xmm8
   DB  69,15,89,197                        ; mulps         %xmm13,%xmm8
-  DB  68,15,40,53,49,57,0,0               ; movaps        0x3931(%rip),%xmm14        # 4ab0 <_sk_callback_sse2+0x2ad>
+  DB  68,15,40,53,97,57,0,0               ; movaps        0x3961(%rip),%xmm14        # 4ae0 <_sk_callback_sse2+0x2b8>
   DB  15,40,214                           ; movaps        %xmm6,%xmm2
   DB  65,15,89,214                        ; mulps         %xmm14,%xmm2
   DB  65,15,88,208                        ; addps         %xmm8,%xmm2
-  DB  68,15,40,5,46,57,0,0                ; movaps        0x392e(%rip),%xmm8        # 4ac0 <_sk_callback_sse2+0x2bd>
+  DB  68,15,40,5,94,57,0,0                ; movaps        0x395e(%rip),%xmm8        # 4af0 <_sk_callback_sse2+0x2c8>
   DB  69,15,40,202                        ; movaps        %xmm10,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,88,202                        ; addps         %xmm2,%xmm9
@@ -18268,7 +18286,7 @@ _sk_color_sse2 LABEL PROC
   DB  15,40,227                           ; movaps        %xmm3,%xmm4
   DB  68,15,40,249                        ; movaps        %xmm1,%xmm15
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
-  DB  68,15,40,13,144,55,0,0              ; movaps        0x3790(%rip),%xmm9        # 4ad0 <_sk_callback_sse2+0x2cd>
+  DB  68,15,40,13,192,55,0,0              ; movaps        0x37c0(%rip),%xmm9        # 4b00 <_sk_callback_sse2+0x2d8>
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
   DB  68,15,94,212                        ; divps         %xmm4,%xmm10
   DB  68,15,40,228                        ; movaps        %xmm4,%xmm12
@@ -18277,14 +18295,14 @@ _sk_color_sse2 LABEL PROC
   DB  65,15,89,196                        ; mulps         %xmm12,%xmm0
   DB  69,15,89,252                        ; mulps         %xmm12,%xmm15
   DB  68,15,89,226                        ; mulps         %xmm2,%xmm12
-  DB  68,15,40,45,119,55,0,0              ; movaps        0x3777(%rip),%xmm13        # 4ae0 <_sk_callback_sse2+0x2dd>
+  DB  68,15,40,45,167,55,0,0              ; movaps        0x37a7(%rip),%xmm13        # 4b10 <_sk_callback_sse2+0x2e8>
   DB  68,15,40,213                        ; movaps        %xmm5,%xmm10
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,40,53,119,55,0,0              ; movaps        0x3777(%rip),%xmm14        # 4af0 <_sk_callback_sse2+0x2ed>
+  DB  68,15,40,53,167,55,0,0              ; movaps        0x37a7(%rip),%xmm14        # 4b20 <_sk_callback_sse2+0x2f8>
   DB  65,15,40,211                        ; movaps        %xmm11,%xmm2
   DB  65,15,89,214                        ; mulps         %xmm14,%xmm2
   DB  65,15,88,210                        ; addps         %xmm10,%xmm2
-  DB  68,15,40,21,115,55,0,0              ; movaps        0x3773(%rip),%xmm10        # 4b00 <_sk_callback_sse2+0x2fd>
+  DB  68,15,40,21,163,55,0,0              ; movaps        0x37a3(%rip),%xmm10        # 4b30 <_sk_callback_sse2+0x308>
   DB  68,15,40,222                        ; movaps        %xmm6,%xmm11
   DB  69,15,89,218                        ; mulps         %xmm10,%xmm11
   DB  68,15,88,218                        ; addps         %xmm2,%xmm11
@@ -18401,7 +18419,7 @@ _sk_luminosity_sse2 LABEL PROC
   DB  68,15,40,193                        ; movaps        %xmm1,%xmm8
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,87,210                        ; xorps         %xmm10,%xmm10
-  DB  68,15,40,29,203,53,0,0              ; movaps        0x35cb(%rip),%xmm11        # 4b10 <_sk_callback_sse2+0x30d>
+  DB  68,15,40,29,251,53,0,0              ; movaps        0x35fb(%rip),%xmm11        # 4b40 <_sk_callback_sse2+0x318>
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  15,94,199                           ; divps         %xmm7,%xmm0
   DB  68,15,40,231                        ; movaps        %xmm7,%xmm12
@@ -18412,12 +18430,12 @@ _sk_luminosity_sse2 LABEL PROC
   DB  65,15,40,204                        ; movaps        %xmm12,%xmm1
   DB  15,89,206                           ; mulps         %xmm6,%xmm1
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
-  DB  68,15,40,53,173,53,0,0              ; movaps        0x35ad(%rip),%xmm14        # 4b20 <_sk_callback_sse2+0x31d>
+  DB  68,15,40,53,221,53,0,0              ; movaps        0x35dd(%rip),%xmm14        # 4b50 <_sk_callback_sse2+0x328>
   DB  69,15,89,206                        ; mulps         %xmm14,%xmm9
-  DB  68,15,40,45,177,53,0,0              ; movaps        0x35b1(%rip),%xmm13        # 4b30 <_sk_callback_sse2+0x32d>
+  DB  68,15,40,45,225,53,0,0              ; movaps        0x35e1(%rip),%xmm13        # 4b60 <_sk_callback_sse2+0x338>
   DB  69,15,89,197                        ; mulps         %xmm13,%xmm8
   DB  69,15,88,193                        ; addps         %xmm9,%xmm8
-  DB  68,15,40,13,177,53,0,0              ; movaps        0x35b1(%rip),%xmm9        # 4b40 <_sk_callback_sse2+0x33d>
+  DB  68,15,40,13,225,53,0,0              ; movaps        0x35e1(%rip),%xmm9        # 4b70 <_sk_callback_sse2+0x348>
   DB  65,15,89,217                        ; mulps         %xmm9,%xmm3
   DB  65,15,88,216                        ; addps         %xmm8,%xmm3
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
@@ -18534,7 +18552,7 @@ _sk_clamp_0_sse2 LABEL PROC
 
 PUBLIC _sk_clamp_1_sse2
 _sk_clamp_1_sse2 LABEL PROC
-  DB  68,15,40,5,16,52,0,0                ; movaps        0x3410(%rip),%xmm8        # 4b50 <_sk_callback_sse2+0x34d>
+  DB  68,15,40,5,64,52,0,0                ; movaps        0x3440(%rip),%xmm8        # 4b80 <_sk_callback_sse2+0x358>
   DB  65,15,93,192                        ; minps         %xmm8,%xmm0
   DB  65,15,93,200                        ; minps         %xmm8,%xmm1
   DB  65,15,93,208                        ; minps         %xmm8,%xmm2
@@ -18544,7 +18562,7 @@ _sk_clamp_1_sse2 LABEL PROC
 
 PUBLIC _sk_clamp_a_sse2
 _sk_clamp_a_sse2 LABEL PROC
-  DB  15,93,29,5,52,0,0                   ; minps         0x3405(%rip),%xmm3        # 4b60 <_sk_callback_sse2+0x35d>
+  DB  15,93,29,53,52,0,0                  ; minps         0x3435(%rip),%xmm3        # 4b90 <_sk_callback_sse2+0x368>
   DB  15,93,195                           ; minps         %xmm3,%xmm0
   DB  15,93,203                           ; minps         %xmm3,%xmm1
   DB  15,93,211                           ; minps         %xmm3,%xmm2
@@ -18617,7 +18635,7 @@ _sk_premul_sse2 LABEL PROC
 PUBLIC _sk_unpremul_sse2
 _sk_unpremul_sse2 LABEL PROC
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
-  DB  68,15,40,13,112,51,0,0              ; movaps        0x3370(%rip),%xmm9        # 4b70 <_sk_callback_sse2+0x36d>
+  DB  68,15,40,13,160,51,0,0              ; movaps        0x33a0(%rip),%xmm9        # 4ba0 <_sk_callback_sse2+0x378>
   DB  68,15,94,203                        ; divps         %xmm3,%xmm9
   DB  68,15,194,195,4                     ; cmpneqps      %xmm3,%xmm8
   DB  69,15,84,193                        ; andps         %xmm9,%xmm8
@@ -18629,20 +18647,20 @@ _sk_unpremul_sse2 LABEL PROC
 
 PUBLIC _sk_from_srgb_sse2
 _sk_from_srgb_sse2 LABEL PROC
-  DB  68,15,40,5,91,51,0,0                ; movaps        0x335b(%rip),%xmm8        # 4b80 <_sk_callback_sse2+0x37d>
+  DB  68,15,40,5,139,51,0,0               ; movaps        0x338b(%rip),%xmm8        # 4bb0 <_sk_callback_sse2+0x388>
   DB  68,15,40,232                        ; movaps        %xmm0,%xmm13
   DB  69,15,89,232                        ; mulps         %xmm8,%xmm13
   DB  68,15,40,216                        ; movaps        %xmm0,%xmm11
   DB  69,15,89,219                        ; mulps         %xmm11,%xmm11
-  DB  68,15,40,13,83,51,0,0               ; movaps        0x3353(%rip),%xmm9        # 4b90 <_sk_callback_sse2+0x38d>
+  DB  68,15,40,13,131,51,0,0              ; movaps        0x3383(%rip),%xmm9        # 4bc0 <_sk_callback_sse2+0x398>
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
   DB  69,15,89,241                        ; mulps         %xmm9,%xmm14
-  DB  68,15,40,21,83,51,0,0               ; movaps        0x3353(%rip),%xmm10        # 4ba0 <_sk_callback_sse2+0x39d>
+  DB  68,15,40,21,131,51,0,0              ; movaps        0x3383(%rip),%xmm10        # 4bd0 <_sk_callback_sse2+0x3a8>
   DB  69,15,88,242                        ; addps         %xmm10,%xmm14
   DB  69,15,89,243                        ; mulps         %xmm11,%xmm14
-  DB  68,15,40,29,83,51,0,0               ; movaps        0x3353(%rip),%xmm11        # 4bb0 <_sk_callback_sse2+0x3ad>
+  DB  68,15,40,29,131,51,0,0              ; movaps        0x3383(%rip),%xmm11        # 4be0 <_sk_callback_sse2+0x3b8>
   DB  69,15,88,243                        ; addps         %xmm11,%xmm14
-  DB  68,15,40,37,87,51,0,0               ; movaps        0x3357(%rip),%xmm12        # 4bc0 <_sk_callback_sse2+0x3bd>
+  DB  68,15,40,37,135,51,0,0              ; movaps        0x3387(%rip),%xmm12        # 4bf0 <_sk_callback_sse2+0x3c8>
   DB  65,15,194,196,1                     ; cmpltps       %xmm12,%xmm0
   DB  68,15,84,232                        ; andps         %xmm0,%xmm13
   DB  65,15,85,198                        ; andnps        %xmm14,%xmm0
@@ -18679,20 +18697,20 @@ _sk_to_srgb_sse2 LABEL PROC
   DB  68,15,82,192                        ; rsqrtps       %xmm0,%xmm8
   DB  69,15,83,200                        ; rcpps         %xmm8,%xmm9
   DB  69,15,82,232                        ; rsqrtps       %xmm8,%xmm13
-  DB  68,15,40,5,220,50,0,0               ; movaps        0x32dc(%rip),%xmm8        # 4bd0 <_sk_callback_sse2+0x3cd>
+  DB  68,15,40,5,12,51,0,0                ; movaps        0x330c(%rip),%xmm8        # 4c00 <_sk_callback_sse2+0x3d8>
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
   DB  69,15,89,240                        ; mulps         %xmm8,%xmm14
-  DB  68,15,40,21,220,50,0,0              ; movaps        0x32dc(%rip),%xmm10        # 4be0 <_sk_callback_sse2+0x3dd>
+  DB  68,15,40,21,12,51,0,0               ; movaps        0x330c(%rip),%xmm10        # 4c10 <_sk_callback_sse2+0x3e8>
   DB  69,15,89,202                        ; mulps         %xmm10,%xmm9
-  DB  68,15,40,29,224,50,0,0              ; movaps        0x32e0(%rip),%xmm11        # 4bf0 <_sk_callback_sse2+0x3ed>
+  DB  68,15,40,29,16,51,0,0               ; movaps        0x3310(%rip),%xmm11        # 4c20 <_sk_callback_sse2+0x3f8>
   DB  69,15,88,203                        ; addps         %xmm11,%xmm9
-  DB  68,15,40,37,228,50,0,0              ; movaps        0x32e4(%rip),%xmm12        # 4c00 <_sk_callback_sse2+0x3fd>
+  DB  68,15,40,37,20,51,0,0               ; movaps        0x3314(%rip),%xmm12        # 4c30 <_sk_callback_sse2+0x408>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,40,13,228,50,0,0              ; movaps        0x32e4(%rip),%xmm9        # 4c10 <_sk_callback_sse2+0x40d>
+  DB  68,15,40,13,20,51,0,0               ; movaps        0x3314(%rip),%xmm9        # 4c40 <_sk_callback_sse2+0x418>
   DB  69,15,40,249                        ; movaps        %xmm9,%xmm15
   DB  69,15,93,253                        ; minps         %xmm13,%xmm15
-  DB  68,15,40,45,228,50,0,0              ; movaps        0x32e4(%rip),%xmm13        # 4c20 <_sk_callback_sse2+0x41d>
+  DB  68,15,40,45,20,51,0,0               ; movaps        0x3314(%rip),%xmm13        # 4c50 <_sk_callback_sse2+0x428>
   DB  65,15,194,197,1                     ; cmpltps       %xmm13,%xmm0
   DB  68,15,84,240                        ; andps         %xmm0,%xmm14
   DB  65,15,85,199                        ; andnps        %xmm15,%xmm0
@@ -18740,7 +18758,7 @@ _sk_rgb_to_hsl_sse2 LABEL PROC
   DB  68,15,93,218                        ; minps         %xmm2,%xmm11
   DB  65,15,40,202                        ; movaps        %xmm10,%xmm1
   DB  65,15,92,203                        ; subps         %xmm11,%xmm1
-  DB  68,15,40,45,61,50,0,0               ; movaps        0x323d(%rip),%xmm13        # 4c30 <_sk_callback_sse2+0x42d>
+  DB  68,15,40,45,109,50,0,0              ; movaps        0x326d(%rip),%xmm13        # 4c60 <_sk_callback_sse2+0x438>
   DB  68,15,94,233                        ; divps         %xmm1,%xmm13
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  65,15,194,192,0                     ; cmpeqps       %xmm8,%xmm0
@@ -18749,30 +18767,30 @@ _sk_rgb_to_hsl_sse2 LABEL PROC
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,40,241                        ; movaps        %xmm9,%xmm14
   DB  68,15,194,242,1                     ; cmpltps       %xmm2,%xmm14
-  DB  68,15,84,53,35,50,0,0               ; andps         0x3223(%rip),%xmm14        # 4c40 <_sk_callback_sse2+0x43d>
+  DB  68,15,84,53,83,50,0,0               ; andps         0x3253(%rip),%xmm14        # 4c70 <_sk_callback_sse2+0x448>
   DB  69,15,88,244                        ; addps         %xmm12,%xmm14
   DB  69,15,40,250                        ; movaps        %xmm10,%xmm15
   DB  69,15,194,249,0                     ; cmpeqps       %xmm9,%xmm15
   DB  65,15,92,208                        ; subps         %xmm8,%xmm2
   DB  65,15,89,213                        ; mulps         %xmm13,%xmm2
-  DB  68,15,40,37,22,50,0,0               ; movaps        0x3216(%rip),%xmm12        # 4c50 <_sk_callback_sse2+0x44d>
+  DB  68,15,40,37,70,50,0,0               ; movaps        0x3246(%rip),%xmm12        # 4c80 <_sk_callback_sse2+0x458>
   DB  65,15,88,212                        ; addps         %xmm12,%xmm2
   DB  69,15,92,193                        ; subps         %xmm9,%xmm8
   DB  69,15,89,197                        ; mulps         %xmm13,%xmm8
-  DB  68,15,88,5,18,50,0,0                ; addps         0x3212(%rip),%xmm8        # 4c60 <_sk_callback_sse2+0x45d>
+  DB  68,15,88,5,66,50,0,0                ; addps         0x3242(%rip),%xmm8        # 4c90 <_sk_callback_sse2+0x468>
   DB  65,15,84,215                        ; andps         %xmm15,%xmm2
   DB  69,15,85,248                        ; andnps        %xmm8,%xmm15
   DB  68,15,86,250                        ; orps          %xmm2,%xmm15
   DB  68,15,84,240                        ; andps         %xmm0,%xmm14
   DB  65,15,85,199                        ; andnps        %xmm15,%xmm0
   DB  65,15,86,198                        ; orps          %xmm14,%xmm0
-  DB  15,89,5,3,50,0,0                    ; mulps         0x3203(%rip),%xmm0        # 4c70 <_sk_callback_sse2+0x46d>
+  DB  15,89,5,51,50,0,0                   ; mulps         0x3233(%rip),%xmm0        # 4ca0 <_sk_callback_sse2+0x478>
   DB  69,15,40,194                        ; movaps        %xmm10,%xmm8
   DB  69,15,194,195,4                     ; cmpneqps      %xmm11,%xmm8
   DB  65,15,84,192                        ; andps         %xmm8,%xmm0
   DB  69,15,92,226                        ; subps         %xmm10,%xmm12
   DB  69,15,88,211                        ; addps         %xmm11,%xmm10
-  DB  68,15,40,13,246,49,0,0              ; movaps        0x31f6(%rip),%xmm9        # 4c80 <_sk_callback_sse2+0x47d>
+  DB  68,15,40,13,38,50,0,0               ; movaps        0x3226(%rip),%xmm9        # 4cb0 <_sk_callback_sse2+0x488>
   DB  65,15,40,210                        ; movaps        %xmm10,%xmm2
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
   DB  68,15,194,202,1                     ; cmpltps       %xmm2,%xmm9
@@ -18795,7 +18813,7 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  15,41,92,36,32                      ; movaps        %xmm3,0x20(%rsp)
   DB  68,15,40,218                        ; movaps        %xmm2,%xmm11
   DB  15,40,240                           ; movaps        %xmm0,%xmm6
-  DB  68,15,40,13,177,49,0,0              ; movaps        0x31b1(%rip),%xmm9        # 4c90 <_sk_callback_sse2+0x48d>
+  DB  68,15,40,13,225,49,0,0              ; movaps        0x31e1(%rip),%xmm9        # 4cc0 <_sk_callback_sse2+0x498>
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
   DB  69,15,194,211,2                     ; cmpleps       %xmm11,%xmm10
   DB  15,40,193                           ; movaps        %xmm1,%xmm0
@@ -18812,28 +18830,28 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  69,15,88,211                        ; addps         %xmm11,%xmm10
   DB  69,15,88,219                        ; addps         %xmm11,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  15,40,5,123,49,0,0                  ; movaps        0x317b(%rip),%xmm0        # 4ca0 <_sk_callback_sse2+0x49d>
+  DB  15,40,5,171,49,0,0                  ; movaps        0x31ab(%rip),%xmm0        # 4cd0 <_sk_callback_sse2+0x4a8>
   DB  15,88,198                           ; addps         %xmm6,%xmm0
   DB  243,15,91,200                       ; cvttps2dq     %xmm0,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  15,40,216                           ; movaps        %xmm0,%xmm3
   DB  15,194,217,1                        ; cmpltps       %xmm1,%xmm3
-  DB  15,84,29,115,49,0,0                 ; andps         0x3173(%rip),%xmm3        # 4cb0 <_sk_callback_sse2+0x4ad>
+  DB  15,84,29,163,49,0,0                 ; andps         0x31a3(%rip),%xmm3        # 4ce0 <_sk_callback_sse2+0x4b8>
   DB  15,92,203                           ; subps         %xmm3,%xmm1
   DB  15,92,193                           ; subps         %xmm1,%xmm0
-  DB  68,15,40,45,117,49,0,0              ; movaps        0x3175(%rip),%xmm13        # 4cc0 <_sk_callback_sse2+0x4bd>
+  DB  68,15,40,45,165,49,0,0              ; movaps        0x31a5(%rip),%xmm13        # 4cf0 <_sk_callback_sse2+0x4c8>
   DB  69,15,40,197                        ; movaps        %xmm13,%xmm8
   DB  68,15,194,192,2                     ; cmpleps       %xmm0,%xmm8
   DB  69,15,40,242                        ; movaps        %xmm10,%xmm14
   DB  69,15,92,243                        ; subps         %xmm11,%xmm14
   DB  65,15,40,217                        ; movaps        %xmm9,%xmm3
   DB  15,194,216,2                        ; cmpleps       %xmm0,%xmm3
-  DB  15,40,21,133,49,0,0                 ; movaps        0x3185(%rip),%xmm2        # 4cf0 <_sk_callback_sse2+0x4ed>
+  DB  15,40,21,181,49,0,0                 ; movaps        0x31b5(%rip),%xmm2        # 4d20 <_sk_callback_sse2+0x4f8>
   DB  68,15,40,250                        ; movaps        %xmm2,%xmm15
   DB  68,15,194,248,2                     ; cmpleps       %xmm0,%xmm15
-  DB  15,40,13,85,49,0,0                  ; movaps        0x3155(%rip),%xmm1        # 4cd0 <_sk_callback_sse2+0x4cd>
+  DB  15,40,13,133,49,0,0                 ; movaps        0x3185(%rip),%xmm1        # 4d00 <_sk_callback_sse2+0x4d8>
   DB  15,89,193                           ; mulps         %xmm1,%xmm0
-  DB  15,40,45,91,49,0,0                  ; movaps        0x315b(%rip),%xmm5        # 4ce0 <_sk_callback_sse2+0x4dd>
+  DB  15,40,45,139,49,0,0                 ; movaps        0x318b(%rip),%xmm5        # 4d10 <_sk_callback_sse2+0x4e8>
   DB  15,40,229                           ; movaps        %xmm5,%xmm4
   DB  15,92,224                           ; subps         %xmm0,%xmm4
   DB  65,15,89,230                        ; mulps         %xmm14,%xmm4
@@ -18856,7 +18874,7 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
   DB  15,40,222                           ; movaps        %xmm6,%xmm3
   DB  15,194,216,1                        ; cmpltps       %xmm0,%xmm3
-  DB  15,84,29,208,48,0,0                 ; andps         0x30d0(%rip),%xmm3        # 4cb0 <_sk_callback_sse2+0x4ad>
+  DB  15,84,29,0,49,0,0                   ; andps         0x3100(%rip),%xmm3        # 4ce0 <_sk_callback_sse2+0x4b8>
   DB  15,92,195                           ; subps         %xmm3,%xmm0
   DB  68,15,40,230                        ; movaps        %xmm6,%xmm12
   DB  68,15,92,224                        ; subps         %xmm0,%xmm12
@@ -18886,12 +18904,12 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  15,40,60,36                         ; movaps        (%rsp),%xmm7
   DB  15,40,231                           ; movaps        %xmm7,%xmm4
   DB  15,85,227                           ; andnps        %xmm3,%xmm4
-  DB  15,88,53,169,48,0,0                 ; addps         0x30a9(%rip),%xmm6        # 4d00 <_sk_callback_sse2+0x4fd>
+  DB  15,88,53,217,48,0,0                 ; addps         0x30d9(%rip),%xmm6        # 4d30 <_sk_callback_sse2+0x508>
   DB  243,15,91,198                       ; cvttps2dq     %xmm6,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
   DB  15,40,222                           ; movaps        %xmm6,%xmm3
   DB  15,194,216,1                        ; cmpltps       %xmm0,%xmm3
-  DB  15,84,29,68,48,0,0                  ; andps         0x3044(%rip),%xmm3        # 4cb0 <_sk_callback_sse2+0x4ad>
+  DB  15,84,29,116,48,0,0                 ; andps         0x3074(%rip),%xmm3        # 4ce0 <_sk_callback_sse2+0x4b8>
   DB  15,92,195                           ; subps         %xmm3,%xmm0
   DB  15,92,240                           ; subps         %xmm0,%xmm6
   DB  15,89,206                           ; mulps         %xmm6,%xmm1
@@ -18952,7 +18970,7 @@ _sk_scale_u8_sse2 LABEL PROC
   DB  102,69,15,96,193                    ; punpcklbw     %xmm9,%xmm8
   DB  102,69,15,97,193                    ; punpcklwd     %xmm9,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,206,47,0,0               ; mulps         0x2fce(%rip),%xmm8        # 4d10 <_sk_callback_sse2+0x50d>
+  DB  68,15,89,5,254,47,0,0               ; mulps         0x2ffe(%rip),%xmm8        # 4d40 <_sk_callback_sse2+0x518>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
@@ -18989,7 +19007,7 @@ _sk_lerp_u8_sse2 LABEL PROC
   DB  102,69,15,96,193                    ; punpcklbw     %xmm9,%xmm8
   DB  102,69,15,97,193                    ; punpcklwd     %xmm9,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,108,47,0,0               ; mulps         0x2f6c(%rip),%xmm8        # 4d20 <_sk_callback_sse2+0x51d>
+  DB  68,15,89,5,156,47,0,0               ; mulps         0x2f9c(%rip),%xmm8        # 4d50 <_sk_callback_sse2+0x528>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -19009,31 +19027,40 @@ PUBLIC _sk_lerp_565_sse2
 _sk_lerp_565_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  243,68,15,126,4,120                 ; movq          (%rax,%rdi,2),%xmm8
-  DB  102,15,239,219                      ; pxor          %xmm3,%xmm3
-  DB  102,68,15,97,195                    ; punpcklwd     %xmm3,%xmm8
-  DB  102,15,111,29,52,47,0,0             ; movdqa        0x2f34(%rip),%xmm3        # 4d30 <_sk_callback_sse2+0x52d>
-  DB  102,65,15,219,216                   ; pand          %xmm8,%xmm3
-  DB  68,15,91,203                        ; cvtdq2ps      %xmm3,%xmm9
-  DB  68,15,89,13,51,47,0,0               ; mulps         0x2f33(%rip),%xmm9        # 4d40 <_sk_callback_sse2+0x53d>
-  DB  102,15,111,29,59,47,0,0             ; movdqa        0x2f3b(%rip),%xmm3        # 4d50 <_sk_callback_sse2+0x54d>
-  DB  102,65,15,219,216                   ; pand          %xmm8,%xmm3
-  DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,60,47,0,0                  ; mulps         0x2f3c(%rip),%xmm3        # 4d60 <_sk_callback_sse2+0x55d>
-  DB  102,68,15,219,5,67,47,0,0           ; pand          0x2f43(%rip),%xmm8        # 4d70 <_sk_callback_sse2+0x56d>
+  DB  243,68,15,126,20,120                ; movq          (%rax,%rdi,2),%xmm10
+  DB  102,69,15,239,192                   ; pxor          %xmm8,%xmm8
+  DB  102,69,15,97,208                    ; punpcklwd     %xmm8,%xmm10
+  DB  102,68,15,111,5,98,47,0,0           ; movdqa        0x2f62(%rip),%xmm8        # 4d60 <_sk_callback_sse2+0x538>
+  DB  102,69,15,219,194                   ; pand          %xmm10,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,71,47,0,0                ; mulps         0x2f47(%rip),%xmm8        # 4d80 <_sk_callback_sse2+0x57d>
+  DB  68,15,89,5,97,47,0,0                ; mulps         0x2f61(%rip),%xmm8        # 4d70 <_sk_callback_sse2+0x548>
+  DB  102,68,15,111,13,104,47,0,0         ; movdqa        0x2f68(%rip),%xmm9        # 4d80 <_sk_callback_sse2+0x558>
+  DB  102,69,15,219,202                   ; pand          %xmm10,%xmm9
+  DB  69,15,91,201                        ; cvtdq2ps      %xmm9,%xmm9
+  DB  68,15,89,13,103,47,0,0              ; mulps         0x2f67(%rip),%xmm9        # 4d90 <_sk_callback_sse2+0x568>
+  DB  102,68,15,219,21,110,47,0,0         ; pand          0x2f6e(%rip),%xmm10        # 4da0 <_sk_callback_sse2+0x578>
+  DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
+  DB  68,15,89,21,114,47,0,0              ; mulps         0x2f72(%rip),%xmm10        # 4db0 <_sk_callback_sse2+0x588>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
-  DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
+  DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
   DB  15,92,205                           ; subps         %xmm5,%xmm1
-  DB  15,89,203                           ; mulps         %xmm3,%xmm1
+  DB  65,15,89,201                        ; mulps         %xmm9,%xmm1
   DB  15,88,205                           ; addps         %xmm5,%xmm1
   DB  15,92,214                           ; subps         %xmm6,%xmm2
-  DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
+  DB  65,15,89,210                        ; mulps         %xmm10,%xmm2
   DB  15,88,214                           ; addps         %xmm6,%xmm2
+  DB  15,92,223                           ; subps         %xmm7,%xmm3
+  DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
+  DB  68,15,88,199                        ; addps         %xmm7,%xmm8
+  DB  68,15,89,203                        ; mulps         %xmm3,%xmm9
+  DB  68,15,88,207                        ; addps         %xmm7,%xmm9
+  DB  65,15,89,218                        ; mulps         %xmm10,%xmm3
+  DB  15,88,223                           ; addps         %xmm7,%xmm3
+  DB  68,15,95,203                        ; maxps         %xmm3,%xmm9
+  DB  69,15,95,193                        ; maxps         %xmm9,%xmm8
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,49,47,0,0                  ; movaps        0x2f31(%rip),%xmm3        # 4d90 <_sk_callback_sse2+0x58d>
+  DB  65,15,40,216                        ; movaps        %xmm8,%xmm3
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_load_tables_sse2
@@ -19042,7 +19069,7 @@ _sk_load_tables_sse2 LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,139,72,8                         ; mov           0x8(%rax),%r9
   DB  243,69,15,111,12,184                ; movdqu        (%r8,%rdi,4),%xmm9
-  DB  102,68,15,111,5,39,47,0,0           ; movdqa        0x2f27(%rip),%xmm8        # 4da0 <_sk_callback_sse2+0x59d>
+  DB  102,68,15,111,5,34,47,0,0           ; movdqa        0x2f22(%rip),%xmm8        # 4dc0 <_sk_callback_sse2+0x598>
   DB  102,65,15,111,193                   ; movdqa        %xmm9,%xmm0
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,15,112,200,78                   ; pshufd        $0x4e,%xmm0,%xmm1
@@ -19097,7 +19124,7 @@ _sk_load_tables_sse2 LABEL PROC
   DB  65,15,20,208                        ; unpcklps      %xmm8,%xmm2
   DB  102,65,15,114,209,24                ; psrld         $0x18,%xmm9
   DB  65,15,91,217                        ; cvtdq2ps      %xmm9,%xmm3
-  DB  15,89,29,52,46,0,0                  ; mulps         0x2e34(%rip),%xmm3        # 4db0 <_sk_callback_sse2+0x5ad>
+  DB  15,89,29,47,46,0,0                  ; mulps         0x2e2f(%rip),%xmm3        # 4dd0 <_sk_callback_sse2+0x5a8>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -19114,7 +19141,7 @@ _sk_load_tables_u16_be_sse2 LABEL PROC
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,97,200                       ; punpcklwd     %xmm0,%xmm1
   DB  102,68,15,105,200                   ; punpckhwd     %xmm0,%xmm9
-  DB  102,68,15,111,21,7,46,0,0           ; movdqa        0x2e07(%rip),%xmm10        # 4dc0 <_sk_callback_sse2+0x5bd>
+  DB  102,68,15,111,21,2,46,0,0           ; movdqa        0x2e02(%rip),%xmm10        # 4de0 <_sk_callback_sse2+0x5b8>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,194                   ; pand          %xmm10,%xmm0
   DB  102,69,15,239,192                   ; pxor          %xmm8,%xmm8
@@ -19175,7 +19202,7 @@ _sk_load_tables_u16_be_sse2 LABEL PROC
   DB  102,65,15,235,217                   ; por           %xmm9,%xmm3
   DB  102,65,15,97,216                    ; punpcklwd     %xmm8,%xmm3
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,246,44,0,0                 ; mulps         0x2cf6(%rip),%xmm3        # 4dd0 <_sk_callback_sse2+0x5cd>
+  DB  15,89,29,241,44,0,0                 ; mulps         0x2cf1(%rip),%xmm3        # 4df0 <_sk_callback_sse2+0x5c8>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -19195,7 +19222,7 @@ _sk_load_tables_rgb_u16_be_sse2 LABEL PROC
   DB  102,68,15,97,208                    ; punpcklwd     %xmm0,%xmm10
   DB  102,65,15,111,195                   ; movdqa        %xmm11,%xmm0
   DB  102,65,15,97,194                    ; punpcklwd     %xmm10,%xmm0
-  DB  102,68,15,111,5,182,44,0,0          ; movdqa        0x2cb6(%rip),%xmm8        # 4de0 <_sk_callback_sse2+0x5dd>
+  DB  102,68,15,111,5,177,44,0,0          ; movdqa        0x2cb1(%rip),%xmm8        # 4e00 <_sk_callback_sse2+0x5d8>
   DB  102,15,112,200,78                   ; pshufd        $0x4e,%xmm0,%xmm1
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,69,15,239,201                   ; pxor          %xmm9,%xmm9
@@ -19250,7 +19277,7 @@ _sk_load_tables_rgb_u16_be_sse2 LABEL PROC
   DB  15,20,211                           ; unpcklps      %xmm3,%xmm2
   DB  65,15,20,208                        ; unpcklps      %xmm8,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,197,43,0,0                 ; movaps        0x2bc5(%rip),%xmm3        # 4df0 <_sk_callback_sse2+0x5ed>
+  DB  15,40,29,192,43,0,0                 ; movaps        0x2bc0(%rip),%xmm3        # 4e10 <_sk_callback_sse2+0x5e8>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_byte_tables_sse2
@@ -19258,7 +19285,7 @@ _sk_byte_tables_sse2 LABEL PROC
   DB  65,86                               ; push          %r14
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,198,43,0,0               ; movaps        0x2bc6(%rip),%xmm8        # 4e00 <_sk_callback_sse2+0x5fd>
+  DB  68,15,40,5,193,43,0,0               ; movaps        0x2bc1(%rip),%xmm8        # 4e20 <_sk_callback_sse2+0x5f8>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,91,192                       ; cvtps2dq      %xmm0,%xmm0
   DB  102,72,15,126,193                   ; movq          %xmm0,%rcx
@@ -19285,7 +19312,7 @@ _sk_byte_tables_sse2 LABEL PROC
   DB  102,65,15,96,193                    ; punpcklbw     %xmm9,%xmm0
   DB  102,65,15,97,193                    ; punpcklwd     %xmm9,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,21,99,43,0,0               ; movaps        0x2b63(%rip),%xmm10        # 4e10 <_sk_callback_sse2+0x60d>
+  DB  68,15,40,21,94,43,0,0               ; movaps        0x2b5e(%rip),%xmm10        # 4e30 <_sk_callback_sse2+0x608>
   DB  65,15,89,194                        ; mulps         %xmm10,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -19399,7 +19426,7 @@ _sk_byte_tables_rgb_sse2 LABEL PROC
   DB  102,65,15,96,193                    ; punpcklbw     %xmm9,%xmm0
   DB  102,65,15,97,193                    ; punpcklwd     %xmm9,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,21,182,41,0,0              ; movaps        0x29b6(%rip),%xmm10        # 4e20 <_sk_callback_sse2+0x61d>
+  DB  68,15,40,21,177,41,0,0              ; movaps        0x29b1(%rip),%xmm10        # 4e40 <_sk_callback_sse2+0x618>
   DB  65,15,89,194                        ; mulps         %xmm10,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -19586,15 +19613,15 @@ _sk_parametric_r_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,245,38,0,0              ; mulps         0x26f5(%rip),%xmm9        # 4e30 <_sk_callback_sse2+0x62d>
-  DB  68,15,84,21,253,38,0,0              ; andps         0x26fd(%rip),%xmm10        # 4e40 <_sk_callback_sse2+0x63d>
-  DB  68,15,86,21,5,39,0,0                ; orps          0x2705(%rip),%xmm10        # 4e50 <_sk_callback_sse2+0x64d>
-  DB  68,15,88,13,13,39,0,0               ; addps         0x270d(%rip),%xmm9        # 4e60 <_sk_callback_sse2+0x65d>
-  DB  68,15,40,37,21,39,0,0               ; movaps        0x2715(%rip),%xmm12        # 4e70 <_sk_callback_sse2+0x66d>
+  DB  68,15,89,13,240,38,0,0              ; mulps         0x26f0(%rip),%xmm9        # 4e50 <_sk_callback_sse2+0x628>
+  DB  68,15,84,21,248,38,0,0              ; andps         0x26f8(%rip),%xmm10        # 4e60 <_sk_callback_sse2+0x638>
+  DB  68,15,86,21,0,39,0,0                ; orps          0x2700(%rip),%xmm10        # 4e70 <_sk_callback_sse2+0x648>
+  DB  68,15,88,13,8,39,0,0                ; addps         0x2708(%rip),%xmm9        # 4e80 <_sk_callback_sse2+0x658>
+  DB  68,15,40,37,16,39,0,0               ; movaps        0x2710(%rip),%xmm12        # 4e90 <_sk_callback_sse2+0x668>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,21,39,0,0               ; addps         0x2715(%rip),%xmm10        # 4e80 <_sk_callback_sse2+0x67d>
-  DB  68,15,40,37,29,39,0,0               ; movaps        0x271d(%rip),%xmm12        # 4e90 <_sk_callback_sse2+0x68d>
+  DB  68,15,88,21,16,39,0,0               ; addps         0x2710(%rip),%xmm10        # 4ea0 <_sk_callback_sse2+0x678>
+  DB  68,15,40,37,24,39,0,0               ; movaps        0x2718(%rip),%xmm12        # 4eb0 <_sk_callback_sse2+0x688>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -19602,22 +19629,22 @@ _sk_parametric_r_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,7,39,0,0                ; movaps        0x2707(%rip),%xmm10        # 4ea0 <_sk_callback_sse2+0x69d>
+  DB  68,15,40,21,2,39,0,0                ; movaps        0x2702(%rip),%xmm10        # 4ec0 <_sk_callback_sse2+0x698>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,251,38,0,0              ; addps         0x26fb(%rip),%xmm9        # 4eb0 <_sk_callback_sse2+0x6ad>
-  DB  68,15,40,37,3,39,0,0                ; movaps        0x2703(%rip),%xmm12        # 4ec0 <_sk_callback_sse2+0x6bd>
+  DB  68,15,88,13,246,38,0,0              ; addps         0x26f6(%rip),%xmm9        # 4ed0 <_sk_callback_sse2+0x6a8>
+  DB  68,15,40,37,254,38,0,0              ; movaps        0x26fe(%rip),%xmm12        # 4ee0 <_sk_callback_sse2+0x6b8>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,3,39,0,0                ; movaps        0x2703(%rip),%xmm12        # 4ed0 <_sk_callback_sse2+0x6cd>
+  DB  68,15,40,37,254,38,0,0              ; movaps        0x26fe(%rip),%xmm12        # 4ef0 <_sk_callback_sse2+0x6c8>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,7,39,0,0                ; movaps        0x2707(%rip),%xmm13        # 4ee0 <_sk_callback_sse2+0x6dd>
+  DB  68,15,40,45,2,39,0,0                ; movaps        0x2702(%rip),%xmm13        # 4f00 <_sk_callback_sse2+0x6d8>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,7,39,0,0                ; mulps         0x2707(%rip),%xmm13        # 4ef0 <_sk_callback_sse2+0x6ed>
+  DB  68,15,89,45,2,39,0,0                ; mulps         0x2702(%rip),%xmm13        # 4f10 <_sk_callback_sse2+0x6e8>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -19651,15 +19678,15 @@ _sk_parametric_g_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,135,38,0,0              ; mulps         0x2687(%rip),%xmm9        # 4f00 <_sk_callback_sse2+0x6fd>
-  DB  68,15,84,21,143,38,0,0              ; andps         0x268f(%rip),%xmm10        # 4f10 <_sk_callback_sse2+0x70d>
-  DB  68,15,86,21,151,38,0,0              ; orps          0x2697(%rip),%xmm10        # 4f20 <_sk_callback_sse2+0x71d>
-  DB  68,15,88,13,159,38,0,0              ; addps         0x269f(%rip),%xmm9        # 4f30 <_sk_callback_sse2+0x72d>
-  DB  68,15,40,37,167,38,0,0              ; movaps        0x26a7(%rip),%xmm12        # 4f40 <_sk_callback_sse2+0x73d>
+  DB  68,15,89,13,130,38,0,0              ; mulps         0x2682(%rip),%xmm9        # 4f20 <_sk_callback_sse2+0x6f8>
+  DB  68,15,84,21,138,38,0,0              ; andps         0x268a(%rip),%xmm10        # 4f30 <_sk_callback_sse2+0x708>
+  DB  68,15,86,21,146,38,0,0              ; orps          0x2692(%rip),%xmm10        # 4f40 <_sk_callback_sse2+0x718>
+  DB  68,15,88,13,154,38,0,0              ; addps         0x269a(%rip),%xmm9        # 4f50 <_sk_callback_sse2+0x728>
+  DB  68,15,40,37,162,38,0,0              ; movaps        0x26a2(%rip),%xmm12        # 4f60 <_sk_callback_sse2+0x738>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,167,38,0,0              ; addps         0x26a7(%rip),%xmm10        # 4f50 <_sk_callback_sse2+0x74d>
-  DB  68,15,40,37,175,38,0,0              ; movaps        0x26af(%rip),%xmm12        # 4f60 <_sk_callback_sse2+0x75d>
+  DB  68,15,88,21,162,38,0,0              ; addps         0x26a2(%rip),%xmm10        # 4f70 <_sk_callback_sse2+0x748>
+  DB  68,15,40,37,170,38,0,0              ; movaps        0x26aa(%rip),%xmm12        # 4f80 <_sk_callback_sse2+0x758>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -19667,22 +19694,22 @@ _sk_parametric_g_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,153,38,0,0              ; movaps        0x2699(%rip),%xmm10        # 4f70 <_sk_callback_sse2+0x76d>
+  DB  68,15,40,21,148,38,0,0              ; movaps        0x2694(%rip),%xmm10        # 4f90 <_sk_callback_sse2+0x768>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,141,38,0,0              ; addps         0x268d(%rip),%xmm9        # 4f80 <_sk_callback_sse2+0x77d>
-  DB  68,15,40,37,149,38,0,0              ; movaps        0x2695(%rip),%xmm12        # 4f90 <_sk_callback_sse2+0x78d>
+  DB  68,15,88,13,136,38,0,0              ; addps         0x2688(%rip),%xmm9        # 4fa0 <_sk_callback_sse2+0x778>
+  DB  68,15,40,37,144,38,0,0              ; movaps        0x2690(%rip),%xmm12        # 4fb0 <_sk_callback_sse2+0x788>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,149,38,0,0              ; movaps        0x2695(%rip),%xmm12        # 4fa0 <_sk_callback_sse2+0x79d>
+  DB  68,15,40,37,144,38,0,0              ; movaps        0x2690(%rip),%xmm12        # 4fc0 <_sk_callback_sse2+0x798>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,153,38,0,0              ; movaps        0x2699(%rip),%xmm13        # 4fb0 <_sk_callback_sse2+0x7ad>
+  DB  68,15,40,45,148,38,0,0              ; movaps        0x2694(%rip),%xmm13        # 4fd0 <_sk_callback_sse2+0x7a8>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,153,38,0,0              ; mulps         0x2699(%rip),%xmm13        # 4fc0 <_sk_callback_sse2+0x7bd>
+  DB  68,15,89,45,148,38,0,0              ; mulps         0x2694(%rip),%xmm13        # 4fe0 <_sk_callback_sse2+0x7b8>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -19716,15 +19743,15 @@ _sk_parametric_b_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,25,38,0,0               ; mulps         0x2619(%rip),%xmm9        # 4fd0 <_sk_callback_sse2+0x7cd>
-  DB  68,15,84,21,33,38,0,0               ; andps         0x2621(%rip),%xmm10        # 4fe0 <_sk_callback_sse2+0x7dd>
-  DB  68,15,86,21,41,38,0,0               ; orps          0x2629(%rip),%xmm10        # 4ff0 <_sk_callback_sse2+0x7ed>
-  DB  68,15,88,13,49,38,0,0               ; addps         0x2631(%rip),%xmm9        # 5000 <_sk_callback_sse2+0x7fd>
-  DB  68,15,40,37,57,38,0,0               ; movaps        0x2639(%rip),%xmm12        # 5010 <_sk_callback_sse2+0x80d>
+  DB  68,15,89,13,20,38,0,0               ; mulps         0x2614(%rip),%xmm9        # 4ff0 <_sk_callback_sse2+0x7c8>
+  DB  68,15,84,21,28,38,0,0               ; andps         0x261c(%rip),%xmm10        # 5000 <_sk_callback_sse2+0x7d8>
+  DB  68,15,86,21,36,38,0,0               ; orps          0x2624(%rip),%xmm10        # 5010 <_sk_callback_sse2+0x7e8>
+  DB  68,15,88,13,44,38,0,0               ; addps         0x262c(%rip),%xmm9        # 5020 <_sk_callback_sse2+0x7f8>
+  DB  68,15,40,37,52,38,0,0               ; movaps        0x2634(%rip),%xmm12        # 5030 <_sk_callback_sse2+0x808>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,57,38,0,0               ; addps         0x2639(%rip),%xmm10        # 5020 <_sk_callback_sse2+0x81d>
-  DB  68,15,40,37,65,38,0,0               ; movaps        0x2641(%rip),%xmm12        # 5030 <_sk_callback_sse2+0x82d>
+  DB  68,15,88,21,52,38,0,0               ; addps         0x2634(%rip),%xmm10        # 5040 <_sk_callback_sse2+0x818>
+  DB  68,15,40,37,60,38,0,0               ; movaps        0x263c(%rip),%xmm12        # 5050 <_sk_callback_sse2+0x828>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -19732,22 +19759,22 @@ _sk_parametric_b_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,43,38,0,0               ; movaps        0x262b(%rip),%xmm10        # 5040 <_sk_callback_sse2+0x83d>
+  DB  68,15,40,21,38,38,0,0               ; movaps        0x2626(%rip),%xmm10        # 5060 <_sk_callback_sse2+0x838>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,31,38,0,0               ; addps         0x261f(%rip),%xmm9        # 5050 <_sk_callback_sse2+0x84d>
-  DB  68,15,40,37,39,38,0,0               ; movaps        0x2627(%rip),%xmm12        # 5060 <_sk_callback_sse2+0x85d>
+  DB  68,15,88,13,26,38,0,0               ; addps         0x261a(%rip),%xmm9        # 5070 <_sk_callback_sse2+0x848>
+  DB  68,15,40,37,34,38,0,0               ; movaps        0x2622(%rip),%xmm12        # 5080 <_sk_callback_sse2+0x858>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,39,38,0,0               ; movaps        0x2627(%rip),%xmm12        # 5070 <_sk_callback_sse2+0x86d>
+  DB  68,15,40,37,34,38,0,0               ; movaps        0x2622(%rip),%xmm12        # 5090 <_sk_callback_sse2+0x868>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,43,38,0,0               ; movaps        0x262b(%rip),%xmm13        # 5080 <_sk_callback_sse2+0x87d>
+  DB  68,15,40,45,38,38,0,0               ; movaps        0x2626(%rip),%xmm13        # 50a0 <_sk_callback_sse2+0x878>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,43,38,0,0               ; mulps         0x262b(%rip),%xmm13        # 5090 <_sk_callback_sse2+0x88d>
+  DB  68,15,89,45,38,38,0,0               ; mulps         0x2626(%rip),%xmm13        # 50b0 <_sk_callback_sse2+0x888>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -19781,15 +19808,15 @@ _sk_parametric_a_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,171,37,0,0              ; mulps         0x25ab(%rip),%xmm9        # 50a0 <_sk_callback_sse2+0x89d>
-  DB  68,15,84,21,179,37,0,0              ; andps         0x25b3(%rip),%xmm10        # 50b0 <_sk_callback_sse2+0x8ad>
-  DB  68,15,86,21,187,37,0,0              ; orps          0x25bb(%rip),%xmm10        # 50c0 <_sk_callback_sse2+0x8bd>
-  DB  68,15,88,13,195,37,0,0              ; addps         0x25c3(%rip),%xmm9        # 50d0 <_sk_callback_sse2+0x8cd>
-  DB  68,15,40,37,203,37,0,0              ; movaps        0x25cb(%rip),%xmm12        # 50e0 <_sk_callback_sse2+0x8dd>
+  DB  68,15,89,13,166,37,0,0              ; mulps         0x25a6(%rip),%xmm9        # 50c0 <_sk_callback_sse2+0x898>
+  DB  68,15,84,21,174,37,0,0              ; andps         0x25ae(%rip),%xmm10        # 50d0 <_sk_callback_sse2+0x8a8>
+  DB  68,15,86,21,182,37,0,0              ; orps          0x25b6(%rip),%xmm10        # 50e0 <_sk_callback_sse2+0x8b8>
+  DB  68,15,88,13,190,37,0,0              ; addps         0x25be(%rip),%xmm9        # 50f0 <_sk_callback_sse2+0x8c8>
+  DB  68,15,40,37,198,37,0,0              ; movaps        0x25c6(%rip),%xmm12        # 5100 <_sk_callback_sse2+0x8d8>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,203,37,0,0              ; addps         0x25cb(%rip),%xmm10        # 50f0 <_sk_callback_sse2+0x8ed>
-  DB  68,15,40,37,211,37,0,0              ; movaps        0x25d3(%rip),%xmm12        # 5100 <_sk_callback_sse2+0x8fd>
+  DB  68,15,88,21,198,37,0,0              ; addps         0x25c6(%rip),%xmm10        # 5110 <_sk_callback_sse2+0x8e8>
+  DB  68,15,40,37,206,37,0,0              ; movaps        0x25ce(%rip),%xmm12        # 5120 <_sk_callback_sse2+0x8f8>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -19797,22 +19824,22 @@ _sk_parametric_a_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,189,37,0,0              ; movaps        0x25bd(%rip),%xmm10        # 5110 <_sk_callback_sse2+0x90d>
+  DB  68,15,40,21,184,37,0,0              ; movaps        0x25b8(%rip),%xmm10        # 5130 <_sk_callback_sse2+0x908>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,177,37,0,0              ; addps         0x25b1(%rip),%xmm9        # 5120 <_sk_callback_sse2+0x91d>
-  DB  68,15,40,37,185,37,0,0              ; movaps        0x25b9(%rip),%xmm12        # 5130 <_sk_callback_sse2+0x92d>
+  DB  68,15,88,13,172,37,0,0              ; addps         0x25ac(%rip),%xmm9        # 5140 <_sk_callback_sse2+0x918>
+  DB  68,15,40,37,180,37,0,0              ; movaps        0x25b4(%rip),%xmm12        # 5150 <_sk_callback_sse2+0x928>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,185,37,0,0              ; movaps        0x25b9(%rip),%xmm12        # 5140 <_sk_callback_sse2+0x93d>
+  DB  68,15,40,37,180,37,0,0              ; movaps        0x25b4(%rip),%xmm12        # 5160 <_sk_callback_sse2+0x938>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,189,37,0,0              ; movaps        0x25bd(%rip),%xmm13        # 5150 <_sk_callback_sse2+0x94d>
+  DB  68,15,40,45,184,37,0,0              ; movaps        0x25b8(%rip),%xmm13        # 5170 <_sk_callback_sse2+0x948>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,189,37,0,0              ; mulps         0x25bd(%rip),%xmm13        # 5160 <_sk_callback_sse2+0x95d>
+  DB  68,15,89,45,184,37,0,0              ; mulps         0x25b8(%rip),%xmm13        # 5180 <_sk_callback_sse2+0x958>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -19827,29 +19854,29 @@ _sk_parametric_a_sse2 LABEL PROC
 
 PUBLIC _sk_lab_to_xyz_sse2
 _sk_lab_to_xyz_sse2 LABEL PROC
-  DB  15,89,5,154,37,0,0                  ; mulps         0x259a(%rip),%xmm0        # 5170 <_sk_callback_sse2+0x96d>
-  DB  68,15,40,5,162,37,0,0               ; movaps        0x25a2(%rip),%xmm8        # 5180 <_sk_callback_sse2+0x97d>
+  DB  15,89,5,149,37,0,0                  ; mulps         0x2595(%rip),%xmm0        # 5190 <_sk_callback_sse2+0x968>
+  DB  68,15,40,5,157,37,0,0               ; movaps        0x259d(%rip),%xmm8        # 51a0 <_sk_callback_sse2+0x978>
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
-  DB  68,15,40,13,166,37,0,0              ; movaps        0x25a6(%rip),%xmm9        # 5190 <_sk_callback_sse2+0x98d>
+  DB  68,15,40,13,161,37,0,0              ; movaps        0x25a1(%rip),%xmm9        # 51b0 <_sk_callback_sse2+0x988>
   DB  65,15,88,201                        ; addps         %xmm9,%xmm1
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  65,15,88,209                        ; addps         %xmm9,%xmm2
-  DB  15,88,5,163,37,0,0                  ; addps         0x25a3(%rip),%xmm0        # 51a0 <_sk_callback_sse2+0x99d>
-  DB  15,89,5,172,37,0,0                  ; mulps         0x25ac(%rip),%xmm0        # 51b0 <_sk_callback_sse2+0x9ad>
-  DB  15,89,13,181,37,0,0                 ; mulps         0x25b5(%rip),%xmm1        # 51c0 <_sk_callback_sse2+0x9bd>
+  DB  15,88,5,158,37,0,0                  ; addps         0x259e(%rip),%xmm0        # 51c0 <_sk_callback_sse2+0x998>
+  DB  15,89,5,167,37,0,0                  ; mulps         0x25a7(%rip),%xmm0        # 51d0 <_sk_callback_sse2+0x9a8>
+  DB  15,89,13,176,37,0,0                 ; mulps         0x25b0(%rip),%xmm1        # 51e0 <_sk_callback_sse2+0x9b8>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  15,89,21,187,37,0,0                 ; mulps         0x25bb(%rip),%xmm2        # 51d0 <_sk_callback_sse2+0x9cd>
+  DB  15,89,21,182,37,0,0                 ; mulps         0x25b6(%rip),%xmm2        # 51f0 <_sk_callback_sse2+0x9c8>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  68,15,92,202                        ; subps         %xmm2,%xmm9
   DB  68,15,40,225                        ; movaps        %xmm1,%xmm12
   DB  69,15,89,228                        ; mulps         %xmm12,%xmm12
   DB  68,15,89,225                        ; mulps         %xmm1,%xmm12
-  DB  15,40,21,176,37,0,0                 ; movaps        0x25b0(%rip),%xmm2        # 51e0 <_sk_callback_sse2+0x9dd>
+  DB  15,40,21,171,37,0,0                 ; movaps        0x25ab(%rip),%xmm2        # 5200 <_sk_callback_sse2+0x9d8>
   DB  68,15,40,194                        ; movaps        %xmm2,%xmm8
   DB  69,15,194,196,1                     ; cmpltps       %xmm12,%xmm8
-  DB  68,15,40,21,175,37,0,0              ; movaps        0x25af(%rip),%xmm10        # 51f0 <_sk_callback_sse2+0x9ed>
+  DB  68,15,40,21,170,37,0,0              ; movaps        0x25aa(%rip),%xmm10        # 5210 <_sk_callback_sse2+0x9e8>
   DB  65,15,88,202                        ; addps         %xmm10,%xmm1
-  DB  68,15,40,29,179,37,0,0              ; movaps        0x25b3(%rip),%xmm11        # 5200 <_sk_callback_sse2+0x9fd>
+  DB  68,15,40,29,174,37,0,0              ; movaps        0x25ae(%rip),%xmm11        # 5220 <_sk_callback_sse2+0x9f8>
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
   DB  69,15,84,224                        ; andps         %xmm8,%xmm12
   DB  68,15,85,193                        ; andnps        %xmm1,%xmm8
@@ -19873,8 +19900,8 @@ _sk_lab_to_xyz_sse2 LABEL PROC
   DB  15,84,194                           ; andps         %xmm2,%xmm0
   DB  65,15,85,209                        ; andnps        %xmm9,%xmm2
   DB  15,86,208                           ; orps          %xmm0,%xmm2
-  DB  68,15,89,5,99,37,0,0                ; mulps         0x2563(%rip),%xmm8        # 5210 <_sk_callback_sse2+0xa0d>
-  DB  15,89,21,108,37,0,0                 ; mulps         0x256c(%rip),%xmm2        # 5220 <_sk_callback_sse2+0xa1d>
+  DB  68,15,89,5,94,37,0,0                ; mulps         0x255e(%rip),%xmm8        # 5230 <_sk_callback_sse2+0xa08>
+  DB  15,89,21,103,37,0,0                 ; mulps         0x2567(%rip),%xmm2        # 5240 <_sk_callback_sse2+0xa18>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -19888,7 +19915,7 @@ _sk_load_a8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,84,37,0,0                  ; mulps         0x2554(%rip),%xmm3        # 5230 <_sk_callback_sse2+0xa2d>
+  DB  15,89,29,79,37,0,0                  ; mulps         0x254f(%rip),%xmm3        # 5250 <_sk_callback_sse2+0xa28>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
@@ -19931,7 +19958,7 @@ _sk_gather_a8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,195,36,0,0                 ; mulps         0x24c3(%rip),%xmm3        # 5240 <_sk_callback_sse2+0xa3d>
+  DB  15,89,29,190,36,0,0                 ; mulps         0x24be(%rip),%xmm3        # 5260 <_sk_callback_sse2+0xa38>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
@@ -19942,7 +19969,7 @@ PUBLIC _sk_store_a8_sse2
 _sk_store_a8_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,183,36,0,0               ; movaps        0x24b7(%rip),%xmm8        # 5250 <_sk_callback_sse2+0xa4d>
+  DB  68,15,40,5,178,36,0,0               ; movaps        0x24b2(%rip),%xmm8        # 5270 <_sk_callback_sse2+0xa48>
   DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
   DB  102,65,15,114,240,16                ; pslld         $0x10,%xmm8
@@ -19962,9 +19989,9 @@ _sk_load_g8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,126,36,0,0                  ; mulps         0x247e(%rip),%xmm0        # 5260 <_sk_callback_sse2+0xa5d>
+  DB  15,89,5,121,36,0,0                  ; mulps         0x2479(%rip),%xmm0        # 5280 <_sk_callback_sse2+0xa58>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,133,36,0,0                 ; movaps        0x2485(%rip),%xmm3        # 5270 <_sk_callback_sse2+0xa6d>
+  DB  15,40,29,128,36,0,0                 ; movaps        0x2480(%rip),%xmm3        # 5290 <_sk_callback_sse2+0xa68>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -20005,9 +20032,9 @@ _sk_gather_g8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,250,35,0,0                  ; mulps         0x23fa(%rip),%xmm0        # 5280 <_sk_callback_sse2+0xa7d>
+  DB  15,89,5,245,35,0,0                  ; mulps         0x23f5(%rip),%xmm0        # 52a0 <_sk_callback_sse2+0xa78>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,1,36,0,0                   ; movaps        0x2401(%rip),%xmm3        # 5290 <_sk_callback_sse2+0xa8d>
+  DB  15,40,29,252,35,0,0                 ; movaps        0x23fc(%rip),%xmm3        # 52b0 <_sk_callback_sse2+0xa88>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -20017,9 +20044,9 @@ _sk_gather_i8_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            2ea6 <_sk_gather_i8_sse2+0xf>
+  DB  116,5                               ; je            2ecb <_sk_gather_i8_sse2+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           2ea8 <_sk_gather_i8_sse2+0x11>
+  DB  235,2                               ; jmp           2ecd <_sk_gather_i8_sse2+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  243,15,91,201                       ; cvttps2dq     %xmm1,%xmm1
@@ -20068,11 +20095,11 @@ _sk_gather_i8_sse2 LABEL PROC
   DB  102,67,15,110,12,136                ; movd          (%r8,%r9,4),%xmm1
   DB  102,68,15,98,201                    ; punpckldq     %xmm1,%xmm9
   DB  102,68,15,98,200                    ; punpckldq     %xmm0,%xmm9
-  DB  102,15,111,21,32,35,0,0             ; movdqa        0x2320(%rip),%xmm2        # 52a0 <_sk_callback_sse2+0xa9d>
+  DB  102,15,111,21,27,35,0,0             ; movdqa        0x231b(%rip),%xmm2        # 52c0 <_sk_callback_sse2+0xa98>
   DB  102,65,15,111,193                   ; movdqa        %xmm9,%xmm0
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,28,35,0,0                ; movaps        0x231c(%rip),%xmm8        # 52b0 <_sk_callback_sse2+0xaad>
+  DB  68,15,40,5,23,35,0,0                ; movaps        0x2317(%rip),%xmm8        # 52d0 <_sk_callback_sse2+0xaa8>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,114,209,8                    ; psrld         $0x8,%xmm1
@@ -20097,19 +20124,19 @@ _sk_load_565_sse2 LABEL PROC
   DB  243,15,126,20,120                   ; movq          (%rax,%rdi,2),%xmm2
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,208                       ; punpcklwd     %xmm0,%xmm2
-  DB  102,15,111,5,210,34,0,0             ; movdqa        0x22d2(%rip),%xmm0        # 52c0 <_sk_callback_sse2+0xabd>
+  DB  102,15,111,5,205,34,0,0             ; movdqa        0x22cd(%rip),%xmm0        # 52e0 <_sk_callback_sse2+0xab8>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,212,34,0,0                  ; mulps         0x22d4(%rip),%xmm0        # 52d0 <_sk_callback_sse2+0xacd>
-  DB  102,15,111,13,220,34,0,0            ; movdqa        0x22dc(%rip),%xmm1        # 52e0 <_sk_callback_sse2+0xadd>
+  DB  15,89,5,207,34,0,0                  ; mulps         0x22cf(%rip),%xmm0        # 52f0 <_sk_callback_sse2+0xac8>
+  DB  102,15,111,13,215,34,0,0            ; movdqa        0x22d7(%rip),%xmm1        # 5300 <_sk_callback_sse2+0xad8>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,222,34,0,0                 ; mulps         0x22de(%rip),%xmm1        # 52f0 <_sk_callback_sse2+0xaed>
-  DB  102,15,219,21,230,34,0,0            ; pand          0x22e6(%rip),%xmm2        # 5300 <_sk_callback_sse2+0xafd>
+  DB  15,89,13,217,34,0,0                 ; mulps         0x22d9(%rip),%xmm1        # 5310 <_sk_callback_sse2+0xae8>
+  DB  102,15,219,21,225,34,0,0            ; pand          0x22e1(%rip),%xmm2        # 5320 <_sk_callback_sse2+0xaf8>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,236,34,0,0                 ; mulps         0x22ec(%rip),%xmm2        # 5310 <_sk_callback_sse2+0xb0d>
+  DB  15,89,21,231,34,0,0                 ; mulps         0x22e7(%rip),%xmm2        # 5330 <_sk_callback_sse2+0xb08>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,243,34,0,0                 ; movaps        0x22f3(%rip),%xmm3        # 5320 <_sk_callback_sse2+0xb1d>
+  DB  15,40,29,238,34,0,0                 ; movaps        0x22ee(%rip),%xmm3        # 5340 <_sk_callback_sse2+0xb18>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_gather_565_sse2
@@ -20142,31 +20169,31 @@ _sk_gather_565_sse2 LABEL PROC
   DB  102,15,196,208,3                    ; pinsrw        $0x3,%eax,%xmm2
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,208                       ; punpcklwd     %xmm0,%xmm2
-  DB  102,15,111,5,124,34,0,0             ; movdqa        0x227c(%rip),%xmm0        # 5330 <_sk_callback_sse2+0xb2d>
+  DB  102,15,111,5,119,34,0,0             ; movdqa        0x2277(%rip),%xmm0        # 5350 <_sk_callback_sse2+0xb28>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,126,34,0,0                  ; mulps         0x227e(%rip),%xmm0        # 5340 <_sk_callback_sse2+0xb3d>
-  DB  102,15,111,13,134,34,0,0            ; movdqa        0x2286(%rip),%xmm1        # 5350 <_sk_callback_sse2+0xb4d>
+  DB  15,89,5,121,34,0,0                  ; mulps         0x2279(%rip),%xmm0        # 5360 <_sk_callback_sse2+0xb38>
+  DB  102,15,111,13,129,34,0,0            ; movdqa        0x2281(%rip),%xmm1        # 5370 <_sk_callback_sse2+0xb48>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,136,34,0,0                 ; mulps         0x2288(%rip),%xmm1        # 5360 <_sk_callback_sse2+0xb5d>
-  DB  102,15,219,21,144,34,0,0            ; pand          0x2290(%rip),%xmm2        # 5370 <_sk_callback_sse2+0xb6d>
+  DB  15,89,13,131,34,0,0                 ; mulps         0x2283(%rip),%xmm1        # 5380 <_sk_callback_sse2+0xb58>
+  DB  102,15,219,21,139,34,0,0            ; pand          0x228b(%rip),%xmm2        # 5390 <_sk_callback_sse2+0xb68>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,150,34,0,0                 ; mulps         0x2296(%rip),%xmm2        # 5380 <_sk_callback_sse2+0xb7d>
+  DB  15,89,21,145,34,0,0                 ; mulps         0x2291(%rip),%xmm2        # 53a0 <_sk_callback_sse2+0xb78>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,157,34,0,0                 ; movaps        0x229d(%rip),%xmm3        # 5390 <_sk_callback_sse2+0xb8d>
+  DB  15,40,29,152,34,0,0                 ; movaps        0x2298(%rip),%xmm3        # 53b0 <_sk_callback_sse2+0xb88>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_565_sse2
 _sk_store_565_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,158,34,0,0               ; movaps        0x229e(%rip),%xmm8        # 53a0 <_sk_callback_sse2+0xb9d>
+  DB  68,15,40,5,153,34,0,0               ; movaps        0x2299(%rip),%xmm8        # 53c0 <_sk_callback_sse2+0xb98>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
   DB  102,65,15,114,241,11                ; pslld         $0xb,%xmm9
-  DB  68,15,40,21,147,34,0,0              ; movaps        0x2293(%rip),%xmm10        # 53b0 <_sk_callback_sse2+0xbad>
+  DB  68,15,40,21,142,34,0,0              ; movaps        0x228e(%rip),%xmm10        # 53d0 <_sk_callback_sse2+0xba8>
   DB  68,15,89,209                        ; mulps         %xmm1,%xmm10
   DB  102,69,15,91,210                    ; cvtps2dq      %xmm10,%xmm10
   DB  102,65,15,114,242,5                 ; pslld         $0x5,%xmm10
@@ -20188,21 +20215,21 @@ _sk_load_4444_sse2 LABEL PROC
   DB  243,15,126,28,120                   ; movq          (%rax,%rdi,2),%xmm3
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,216                       ; punpcklwd     %xmm0,%xmm3
-  DB  102,15,111,5,76,34,0,0              ; movdqa        0x224c(%rip),%xmm0        # 53c0 <_sk_callback_sse2+0xbbd>
+  DB  102,15,111,5,71,34,0,0              ; movdqa        0x2247(%rip),%xmm0        # 53e0 <_sk_callback_sse2+0xbb8>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,78,34,0,0                   ; mulps         0x224e(%rip),%xmm0        # 53d0 <_sk_callback_sse2+0xbcd>
-  DB  102,15,111,13,86,34,0,0             ; movdqa        0x2256(%rip),%xmm1        # 53e0 <_sk_callback_sse2+0xbdd>
+  DB  15,89,5,73,34,0,0                   ; mulps         0x2249(%rip),%xmm0        # 53f0 <_sk_callback_sse2+0xbc8>
+  DB  102,15,111,13,81,34,0,0             ; movdqa        0x2251(%rip),%xmm1        # 5400 <_sk_callback_sse2+0xbd8>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,88,34,0,0                  ; mulps         0x2258(%rip),%xmm1        # 53f0 <_sk_callback_sse2+0xbed>
-  DB  102,15,111,21,96,34,0,0             ; movdqa        0x2260(%rip),%xmm2        # 5400 <_sk_callback_sse2+0xbfd>
+  DB  15,89,13,83,34,0,0                  ; mulps         0x2253(%rip),%xmm1        # 5410 <_sk_callback_sse2+0xbe8>
+  DB  102,15,111,21,91,34,0,0             ; movdqa        0x225b(%rip),%xmm2        # 5420 <_sk_callback_sse2+0xbf8>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,98,34,0,0                  ; mulps         0x2262(%rip),%xmm2        # 5410 <_sk_callback_sse2+0xc0d>
-  DB  102,15,219,29,106,34,0,0            ; pand          0x226a(%rip),%xmm3        # 5420 <_sk_callback_sse2+0xc1d>
+  DB  15,89,21,93,34,0,0                  ; mulps         0x225d(%rip),%xmm2        # 5430 <_sk_callback_sse2+0xc08>
+  DB  102,15,219,29,101,34,0,0            ; pand          0x2265(%rip),%xmm3        # 5440 <_sk_callback_sse2+0xc18>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,112,34,0,0                 ; mulps         0x2270(%rip),%xmm3        # 5430 <_sk_callback_sse2+0xc2d>
+  DB  15,89,29,107,34,0,0                 ; mulps         0x226b(%rip),%xmm3        # 5450 <_sk_callback_sse2+0xc28>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -20236,21 +20263,21 @@ _sk_gather_4444_sse2 LABEL PROC
   DB  102,15,196,216,3                    ; pinsrw        $0x3,%eax,%xmm3
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,216                       ; punpcklwd     %xmm0,%xmm3
-  DB  102,15,111,5,247,33,0,0             ; movdqa        0x21f7(%rip),%xmm0        # 5440 <_sk_callback_sse2+0xc3d>
+  DB  102,15,111,5,242,33,0,0             ; movdqa        0x21f2(%rip),%xmm0        # 5460 <_sk_callback_sse2+0xc38>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,249,33,0,0                  ; mulps         0x21f9(%rip),%xmm0        # 5450 <_sk_callback_sse2+0xc4d>
-  DB  102,15,111,13,1,34,0,0              ; movdqa        0x2201(%rip),%xmm1        # 5460 <_sk_callback_sse2+0xc5d>
+  DB  15,89,5,244,33,0,0                  ; mulps         0x21f4(%rip),%xmm0        # 5470 <_sk_callback_sse2+0xc48>
+  DB  102,15,111,13,252,33,0,0            ; movdqa        0x21fc(%rip),%xmm1        # 5480 <_sk_callback_sse2+0xc58>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,3,34,0,0                   ; mulps         0x2203(%rip),%xmm1        # 5470 <_sk_callback_sse2+0xc6d>
-  DB  102,15,111,21,11,34,0,0             ; movdqa        0x220b(%rip),%xmm2        # 5480 <_sk_callback_sse2+0xc7d>
+  DB  15,89,13,254,33,0,0                 ; mulps         0x21fe(%rip),%xmm1        # 5490 <_sk_callback_sse2+0xc68>
+  DB  102,15,111,21,6,34,0,0              ; movdqa        0x2206(%rip),%xmm2        # 54a0 <_sk_callback_sse2+0xc78>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,13,34,0,0                  ; mulps         0x220d(%rip),%xmm2        # 5490 <_sk_callback_sse2+0xc8d>
-  DB  102,15,219,29,21,34,0,0             ; pand          0x2215(%rip),%xmm3        # 54a0 <_sk_callback_sse2+0xc9d>
+  DB  15,89,21,8,34,0,0                   ; mulps         0x2208(%rip),%xmm2        # 54b0 <_sk_callback_sse2+0xc88>
+  DB  102,15,219,29,16,34,0,0             ; pand          0x2210(%rip),%xmm3        # 54c0 <_sk_callback_sse2+0xc98>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,27,34,0,0                  ; mulps         0x221b(%rip),%xmm3        # 54b0 <_sk_callback_sse2+0xcad>
+  DB  15,89,29,22,34,0,0                  ; mulps         0x2216(%rip),%xmm3        # 54d0 <_sk_callback_sse2+0xca8>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -20258,7 +20285,7 @@ PUBLIC _sk_store_4444_sse2
 _sk_store_4444_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,26,34,0,0                ; movaps        0x221a(%rip),%xmm8        # 54c0 <_sk_callback_sse2+0xcbd>
+  DB  68,15,40,5,21,34,0,0                ; movaps        0x2215(%rip),%xmm8        # 54e0 <_sk_callback_sse2+0xcb8>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -20288,11 +20315,11 @@ _sk_load_8888_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  68,15,16,12,184                     ; movups        (%rax,%rdi,4),%xmm9
-  DB  15,40,21,173,33,0,0                 ; movaps        0x21ad(%rip),%xmm2        # 54d0 <_sk_callback_sse2+0xccd>
+  DB  15,40,21,168,33,0,0                 ; movaps        0x21a8(%rip),%xmm2        # 54f0 <_sk_callback_sse2+0xcc8>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  15,84,194                           ; andps         %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,171,33,0,0               ; movaps        0x21ab(%rip),%xmm8        # 54e0 <_sk_callback_sse2+0xcdd>
+  DB  68,15,40,5,166,33,0,0               ; movaps        0x21a6(%rip),%xmm8        # 5500 <_sk_callback_sse2+0xcd8>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,40,201                        ; movaps        %xmm9,%xmm1
   DB  102,15,114,209,8                    ; psrld         $0x8,%xmm1
@@ -20339,11 +20366,11 @@ _sk_gather_8888_sse2 LABEL PROC
   DB  102,67,15,110,12,129                ; movd          (%r9,%r8,4),%xmm1
   DB  102,68,15,98,201                    ; punpckldq     %xmm1,%xmm9
   DB  102,68,15,98,200                    ; punpckldq     %xmm0,%xmm9
-  DB  102,15,111,21,252,32,0,0            ; movdqa        0x20fc(%rip),%xmm2        # 54f0 <_sk_callback_sse2+0xced>
+  DB  102,15,111,21,247,32,0,0            ; movdqa        0x20f7(%rip),%xmm2        # 5510 <_sk_callback_sse2+0xce8>
   DB  102,65,15,111,193                   ; movdqa        %xmm9,%xmm0
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,248,32,0,0               ; movaps        0x20f8(%rip),%xmm8        # 5500 <_sk_callback_sse2+0xcfd>
+  DB  68,15,40,5,243,32,0,0               ; movaps        0x20f3(%rip),%xmm8        # 5520 <_sk_callback_sse2+0xcf8>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,114,209,8                    ; psrld         $0x8,%xmm1
@@ -20365,7 +20392,7 @@ PUBLIC _sk_store_8888_sse2
 _sk_store_8888_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,187,32,0,0               ; movaps        0x20bb(%rip),%xmm8        # 5510 <_sk_callback_sse2+0xd0d>
+  DB  68,15,40,5,182,32,0,0               ; movaps        0x20b6(%rip),%xmm8        # 5530 <_sk_callback_sse2+0xd08>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -20402,7 +20429,7 @@ _sk_load_f16_sse2 LABEL PROC
   DB  102,69,15,239,210                   ; pxor          %xmm10,%xmm10
   DB  102,65,15,111,206                   ; movdqa        %xmm14,%xmm1
   DB  102,65,15,97,202                    ; punpcklwd     %xmm10,%xmm1
-  DB  102,68,15,111,13,43,32,0,0          ; movdqa        0x202b(%rip),%xmm9        # 5520 <_sk_callback_sse2+0xd1d>
+  DB  102,68,15,111,13,38,32,0,0          ; movdqa        0x2026(%rip),%xmm9        # 5540 <_sk_callback_sse2+0xd18>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,193                   ; pand          %xmm9,%xmm0
   DB  102,15,239,200                      ; pxor          %xmm0,%xmm1
@@ -20410,11 +20437,11 @@ _sk_load_f16_sse2 LABEL PROC
   DB  102,68,15,111,233                   ; movdqa        %xmm1,%xmm13
   DB  102,65,15,114,245,13                ; pslld         $0xd,%xmm13
   DB  102,68,15,235,232                   ; por           %xmm0,%xmm13
-  DB  102,68,15,111,29,16,32,0,0          ; movdqa        0x2010(%rip),%xmm11        # 5530 <_sk_callback_sse2+0xd2d>
+  DB  102,68,15,111,29,11,32,0,0          ; movdqa        0x200b(%rip),%xmm11        # 5550 <_sk_callback_sse2+0xd28>
   DB  102,69,15,254,235                   ; paddd         %xmm11,%xmm13
-  DB  102,68,15,111,37,18,32,0,0          ; movdqa        0x2012(%rip),%xmm12        # 5540 <_sk_callback_sse2+0xd3d>
+  DB  102,68,15,111,37,13,32,0,0          ; movdqa        0x200d(%rip),%xmm12        # 5560 <_sk_callback_sse2+0xd38>
   DB  102,65,15,239,204                   ; pxor          %xmm12,%xmm1
-  DB  102,15,111,29,21,32,0,0             ; movdqa        0x2015(%rip),%xmm3        # 5550 <_sk_callback_sse2+0xd4d>
+  DB  102,15,111,29,16,32,0,0             ; movdqa        0x2010(%rip),%xmm3        # 5570 <_sk_callback_sse2+0xd48>
   DB  102,15,111,195                      ; movdqa        %xmm3,%xmm0
   DB  102,15,102,193                      ; pcmpgtd       %xmm1,%xmm0
   DB  102,65,15,223,197                   ; pandn         %xmm13,%xmm0
@@ -20498,7 +20525,7 @@ _sk_gather_f16_sse2 LABEL PROC
   DB  102,69,15,239,210                   ; pxor          %xmm10,%xmm10
   DB  102,65,15,111,206                   ; movdqa        %xmm14,%xmm1
   DB  102,65,15,97,202                    ; punpcklwd     %xmm10,%xmm1
-  DB  102,68,15,111,13,163,30,0,0         ; movdqa        0x1ea3(%rip),%xmm9        # 5560 <_sk_callback_sse2+0xd5d>
+  DB  102,68,15,111,13,158,30,0,0         ; movdqa        0x1e9e(%rip),%xmm9        # 5580 <_sk_callback_sse2+0xd58>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,193                   ; pand          %xmm9,%xmm0
   DB  102,15,239,200                      ; pxor          %xmm0,%xmm1
@@ -20506,11 +20533,11 @@ _sk_gather_f16_sse2 LABEL PROC
   DB  102,68,15,111,233                   ; movdqa        %xmm1,%xmm13
   DB  102,65,15,114,245,13                ; pslld         $0xd,%xmm13
   DB  102,68,15,235,232                   ; por           %xmm0,%xmm13
-  DB  102,68,15,111,29,136,30,0,0         ; movdqa        0x1e88(%rip),%xmm11        # 5570 <_sk_callback_sse2+0xd6d>
+  DB  102,68,15,111,29,131,30,0,0         ; movdqa        0x1e83(%rip),%xmm11        # 5590 <_sk_callback_sse2+0xd68>
   DB  102,69,15,254,235                   ; paddd         %xmm11,%xmm13
-  DB  102,68,15,111,37,138,30,0,0         ; movdqa        0x1e8a(%rip),%xmm12        # 5580 <_sk_callback_sse2+0xd7d>
+  DB  102,68,15,111,37,133,30,0,0         ; movdqa        0x1e85(%rip),%xmm12        # 55a0 <_sk_callback_sse2+0xd78>
   DB  102,65,15,239,204                   ; pxor          %xmm12,%xmm1
-  DB  102,15,111,29,141,30,0,0            ; movdqa        0x1e8d(%rip),%xmm3        # 5590 <_sk_callback_sse2+0xd8d>
+  DB  102,15,111,29,136,30,0,0            ; movdqa        0x1e88(%rip),%xmm3        # 55b0 <_sk_callback_sse2+0xd88>
   DB  102,15,111,195                      ; movdqa        %xmm3,%xmm0
   DB  102,15,102,193                      ; pcmpgtd       %xmm1,%xmm0
   DB  102,65,15,223,197                   ; pandn         %xmm13,%xmm0
@@ -20561,17 +20588,17 @@ PUBLIC _sk_store_f16_sse2
 _sk_store_f16_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  102,68,15,111,21,181,29,0,0         ; movdqa        0x1db5(%rip),%xmm10        # 55a0 <_sk_callback_sse2+0xd9d>
+  DB  102,68,15,111,21,176,29,0,0         ; movdqa        0x1db0(%rip),%xmm10        # 55c0 <_sk_callback_sse2+0xd98>
   DB  102,68,15,111,224                   ; movdqa        %xmm0,%xmm12
   DB  102,68,15,111,232                   ; movdqa        %xmm0,%xmm13
   DB  102,69,15,219,234                   ; pand          %xmm10,%xmm13
   DB  102,69,15,239,229                   ; pxor          %xmm13,%xmm12
-  DB  102,68,15,111,13,168,29,0,0         ; movdqa        0x1da8(%rip),%xmm9        # 55b0 <_sk_callback_sse2+0xdad>
+  DB  102,68,15,111,13,163,29,0,0         ; movdqa        0x1da3(%rip),%xmm9        # 55d0 <_sk_callback_sse2+0xda8>
   DB  102,65,15,114,213,16                ; psrld         $0x10,%xmm13
   DB  102,69,15,111,193                   ; movdqa        %xmm9,%xmm8
   DB  102,69,15,102,196                   ; pcmpgtd       %xmm12,%xmm8
   DB  102,65,15,114,212,13                ; psrld         $0xd,%xmm12
-  DB  102,68,15,111,29,153,29,0,0         ; movdqa        0x1d99(%rip),%xmm11        # 55c0 <_sk_callback_sse2+0xdbd>
+  DB  102,68,15,111,29,148,29,0,0         ; movdqa        0x1d94(%rip),%xmm11        # 55e0 <_sk_callback_sse2+0xdb8>
   DB  102,69,15,235,235                   ; por           %xmm11,%xmm13
   DB  102,69,15,254,236                   ; paddd         %xmm12,%xmm13
   DB  102,65,15,114,245,16                ; pslld         $0x10,%xmm13
@@ -20648,7 +20675,7 @@ _sk_load_u16_be_sse2 LABEL PROC
   DB  102,69,15,239,201                   ; pxor          %xmm9,%xmm9
   DB  102,65,15,97,201                    ; punpcklwd     %xmm9,%xmm1
   DB  15,91,193                           ; cvtdq2ps      %xmm1,%xmm0
-  DB  68,15,40,5,55,28,0,0                ; movaps        0x1c37(%rip),%xmm8        # 55d0 <_sk_callback_sse2+0xdcd>
+  DB  68,15,40,5,50,28,0,0                ; movaps        0x1c32(%rip),%xmm8        # 55f0 <_sk_callback_sse2+0xdc8>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -20699,7 +20726,7 @@ _sk_load_rgb_u16_be_sse2 LABEL PROC
   DB  102,69,15,239,192                   ; pxor          %xmm8,%xmm8
   DB  102,65,15,97,192                    ; punpcklwd     %xmm8,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,13,115,27,0,0              ; movaps        0x1b73(%rip),%xmm9        # 55e0 <_sk_callback_sse2+0xddd>
+  DB  68,15,40,13,110,27,0,0              ; movaps        0x1b6e(%rip),%xmm9        # 5600 <_sk_callback_sse2+0xdd8>
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -20716,14 +20743,14 @@ _sk_load_rgb_u16_be_sse2 LABEL PROC
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,58,27,0,0                  ; movaps        0x1b3a(%rip),%xmm3        # 55f0 <_sk_callback_sse2+0xded>
+  DB  15,40,29,53,27,0,0                  ; movaps        0x1b35(%rip),%xmm3        # 5610 <_sk_callback_sse2+0xde8>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_u16_be_sse2
 _sk_store_u16_be_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,13,59,27,0,0               ; movaps        0x1b3b(%rip),%xmm9        # 5600 <_sk_callback_sse2+0xdfd>
+  DB  68,15,40,13,54,27,0,0               ; movaps        0x1b36(%rip),%xmm9        # 5620 <_sk_callback_sse2+0xdf8>
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
@@ -20863,7 +20890,7 @@ _sk_repeat_x_sse2 LABEL PROC
   DB  243,69,15,91,209                    ; cvttps2dq     %xmm9,%xmm10
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
   DB  69,15,194,202,1                     ; cmpltps       %xmm10,%xmm9
-  DB  68,15,84,13,37,25,0,0               ; andps         0x1925(%rip),%xmm9        # 5610 <_sk_callback_sse2+0xe0d>
+  DB  68,15,84,13,32,25,0,0               ; andps         0x1920(%rip),%xmm9        # 5630 <_sk_callback_sse2+0xe08>
   DB  69,15,92,209                        ; subps         %xmm9,%xmm10
   DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
   DB  65,15,92,194                        ; subps         %xmm10,%xmm0
@@ -20883,7 +20910,7 @@ _sk_repeat_y_sse2 LABEL PROC
   DB  243,69,15,91,209                    ; cvttps2dq     %xmm9,%xmm10
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
   DB  69,15,194,202,1                     ; cmpltps       %xmm10,%xmm9
-  DB  68,15,84,13,237,24,0,0              ; andps         0x18ed(%rip),%xmm9        # 5620 <_sk_callback_sse2+0xe1d>
+  DB  68,15,84,13,232,24,0,0              ; andps         0x18e8(%rip),%xmm9        # 5640 <_sk_callback_sse2+0xe18>
   DB  69,15,92,209                        ; subps         %xmm9,%xmm10
   DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
   DB  65,15,92,202                        ; subps         %xmm10,%xmm1
@@ -20907,7 +20934,7 @@ _sk_mirror_x_sse2 LABEL PROC
   DB  243,69,15,91,218                    ; cvttps2dq     %xmm10,%xmm11
   DB  69,15,91,219                        ; cvtdq2ps      %xmm11,%xmm11
   DB  69,15,194,211,1                     ; cmpltps       %xmm11,%xmm10
-  DB  68,15,84,21,163,24,0,0              ; andps         0x18a3(%rip),%xmm10        # 5630 <_sk_callback_sse2+0xe2d>
+  DB  68,15,84,21,158,24,0,0              ; andps         0x189e(%rip),%xmm10        # 5650 <_sk_callback_sse2+0xe28>
   DB  69,15,87,228                        ; xorps         %xmm12,%xmm12
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  69,15,89,216                        ; mulps         %xmm8,%xmm11
@@ -20935,7 +20962,7 @@ _sk_mirror_y_sse2 LABEL PROC
   DB  243,69,15,91,218                    ; cvttps2dq     %xmm10,%xmm11
   DB  69,15,91,219                        ; cvtdq2ps      %xmm11,%xmm11
   DB  69,15,194,211,1                     ; cmpltps       %xmm11,%xmm10
-  DB  68,15,84,21,73,24,0,0               ; andps         0x1849(%rip),%xmm10        # 5640 <_sk_callback_sse2+0xe3d>
+  DB  68,15,84,21,68,24,0,0               ; andps         0x1844(%rip),%xmm10        # 5660 <_sk_callback_sse2+0xe38>
   DB  69,15,87,228                        ; xorps         %xmm12,%xmm12
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  69,15,89,216                        ; mulps         %xmm8,%xmm11
@@ -20952,10 +20979,10 @@ _sk_mirror_y_sse2 LABEL PROC
 PUBLIC _sk_luminance_to_alpha_sse2
 _sk_luminance_to_alpha_sse2 LABEL PROC
   DB  15,40,218                           ; movaps        %xmm2,%xmm3
-  DB  15,89,5,33,24,0,0                   ; mulps         0x1821(%rip),%xmm0        # 5650 <_sk_callback_sse2+0xe4d>
-  DB  15,89,13,42,24,0,0                  ; mulps         0x182a(%rip),%xmm1        # 5660 <_sk_callback_sse2+0xe5d>
+  DB  15,89,5,28,24,0,0                   ; mulps         0x181c(%rip),%xmm0        # 5670 <_sk_callback_sse2+0xe48>
+  DB  15,89,13,37,24,0,0                  ; mulps         0x1825(%rip),%xmm1        # 5680 <_sk_callback_sse2+0xe58>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  15,89,29,48,24,0,0                  ; mulps         0x1830(%rip),%xmm3        # 5670 <_sk_callback_sse2+0xe6d>
+  DB  15,89,29,43,24,0,0                  ; mulps         0x182b(%rip),%xmm3        # 5690 <_sk_callback_sse2+0xe68>
   DB  15,88,217                           ; addps         %xmm1,%xmm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
@@ -21178,7 +21205,7 @@ _sk_linear_gradient_sse2 LABEL PROC
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
   DB  72,139,8                            ; mov           (%rax),%rcx
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,132,15,1,0,0                     ; je            42f3 <_sk_linear_gradient_sse2+0x149>
+  DB  15,132,15,1,0,0                     ; je            4318 <_sk_linear_gradient_sse2+0x149>
   DB  72,139,64,8                         ; mov           0x8(%rax),%rax
   DB  72,131,192,32                       ; add           $0x20,%rax
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
@@ -21239,8 +21266,8 @@ _sk_linear_gradient_sse2 LABEL PROC
   DB  69,15,86,231                        ; orps          %xmm15,%xmm12
   DB  72,131,192,36                       ; add           $0x24,%rax
   DB  72,255,201                          ; dec           %rcx
-  DB  15,133,8,255,255,255                ; jne           41f9 <_sk_linear_gradient_sse2+0x4f>
-  DB  235,13                              ; jmp           4300 <_sk_linear_gradient_sse2+0x156>
+  DB  15,133,8,255,255,255                ; jne           421e <_sk_linear_gradient_sse2+0x4f>
+  DB  235,13                              ; jmp           4325 <_sk_linear_gradient_sse2+0x156>
   DB  15,87,201                           ; xorps         %xmm1,%xmm1
   DB  15,87,210                           ; xorps         %xmm2,%xmm2
   DB  15,87,219                           ; xorps         %xmm3,%xmm3
@@ -21305,29 +21332,29 @@ _sk_xy_to_polar_unit_sse2 LABEL PROC
   DB  69,15,94,220                        ; divps         %xmm12,%xmm11
   DB  69,15,40,227                        ; movaps        %xmm11,%xmm12
   DB  69,15,89,228                        ; mulps         %xmm12,%xmm12
-  DB  68,15,40,45,168,18,0,0              ; movaps        0x12a8(%rip),%xmm13        # 5680 <_sk_callback_sse2+0xe7d>
+  DB  68,15,40,45,163,18,0,0              ; movaps        0x12a3(%rip),%xmm13        # 56a0 <_sk_callback_sse2+0xe78>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
-  DB  68,15,88,45,172,18,0,0              ; addps         0x12ac(%rip),%xmm13        # 5690 <_sk_callback_sse2+0xe8d>
+  DB  68,15,88,45,167,18,0,0              ; addps         0x12a7(%rip),%xmm13        # 56b0 <_sk_callback_sse2+0xe88>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
-  DB  68,15,88,45,176,18,0,0              ; addps         0x12b0(%rip),%xmm13        # 56a0 <_sk_callback_sse2+0xe9d>
+  DB  68,15,88,45,171,18,0,0              ; addps         0x12ab(%rip),%xmm13        # 56c0 <_sk_callback_sse2+0xe98>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
-  DB  68,15,88,45,180,18,0,0              ; addps         0x12b4(%rip),%xmm13        # 56b0 <_sk_callback_sse2+0xead>
+  DB  68,15,88,45,175,18,0,0              ; addps         0x12af(%rip),%xmm13        # 56d0 <_sk_callback_sse2+0xea8>
   DB  69,15,89,235                        ; mulps         %xmm11,%xmm13
   DB  69,15,194,202,1                     ; cmpltps       %xmm10,%xmm9
-  DB  68,15,40,21,179,18,0,0              ; movaps        0x12b3(%rip),%xmm10        # 56c0 <_sk_callback_sse2+0xebd>
+  DB  68,15,40,21,174,18,0,0              ; movaps        0x12ae(%rip),%xmm10        # 56e0 <_sk_callback_sse2+0xeb8>
   DB  69,15,92,213                        ; subps         %xmm13,%xmm10
   DB  69,15,84,209                        ; andps         %xmm9,%xmm10
   DB  69,15,85,205                        ; andnps        %xmm13,%xmm9
   DB  69,15,86,202                        ; orps          %xmm10,%xmm9
   DB  68,15,194,192,1                     ; cmpltps       %xmm0,%xmm8
-  DB  68,15,40,21,166,18,0,0              ; movaps        0x12a6(%rip),%xmm10        # 56d0 <_sk_callback_sse2+0xecd>
+  DB  68,15,40,21,161,18,0,0              ; movaps        0x12a1(%rip),%xmm10        # 56f0 <_sk_callback_sse2+0xec8>
   DB  69,15,92,209                        ; subps         %xmm9,%xmm10
   DB  69,15,84,208                        ; andps         %xmm8,%xmm10
   DB  69,15,85,193                        ; andnps        %xmm9,%xmm8
   DB  69,15,86,194                        ; orps          %xmm10,%xmm8
   DB  68,15,40,201                        ; movaps        %xmm1,%xmm9
   DB  68,15,194,200,1                     ; cmpltps       %xmm0,%xmm9
-  DB  68,15,40,21,149,18,0,0              ; movaps        0x1295(%rip),%xmm10        # 56e0 <_sk_callback_sse2+0xedd>
+  DB  68,15,40,21,144,18,0,0              ; movaps        0x1290(%rip),%xmm10        # 5700 <_sk_callback_sse2+0xed8>
   DB  69,15,92,208                        ; subps         %xmm8,%xmm10
   DB  69,15,84,209                        ; andps         %xmm9,%xmm10
   DB  69,15,85,200                        ; andnps        %xmm8,%xmm9
@@ -21351,7 +21378,7 @@ _sk_xy_to_radius_sse2 LABEL PROC
 PUBLIC _sk_save_xy_sse2
 _sk_save_xy_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,100,18,0,0               ; movaps        0x1264(%rip),%xmm8        # 56f0 <_sk_callback_sse2+0xeed>
+  DB  68,15,40,5,95,18,0,0                ; movaps        0x125f(%rip),%xmm8        # 5710 <_sk_callback_sse2+0xee8>
   DB  15,17,0                             ; movups        %xmm0,(%rax)
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,88,200                        ; addps         %xmm8,%xmm9
@@ -21359,7 +21386,7 @@ _sk_save_xy_sse2 LABEL PROC
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
   DB  69,15,40,217                        ; movaps        %xmm9,%xmm11
   DB  69,15,194,218,1                     ; cmpltps       %xmm10,%xmm11
-  DB  68,15,40,37,79,18,0,0               ; movaps        0x124f(%rip),%xmm12        # 5700 <_sk_callback_sse2+0xefd>
+  DB  68,15,40,37,74,18,0,0               ; movaps        0x124a(%rip),%xmm12        # 5720 <_sk_callback_sse2+0xef8>
   DB  69,15,84,220                        ; andps         %xmm12,%xmm11
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
   DB  69,15,92,202                        ; subps         %xmm10,%xmm9
@@ -21402,8 +21429,8 @@ _sk_bilinear_nx_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,200,17,0,0                  ; addps         0x11c8(%rip),%xmm0        # 5710 <_sk_callback_sse2+0xf0d>
-  DB  68,15,40,13,208,17,0,0              ; movaps        0x11d0(%rip),%xmm9        # 5720 <_sk_callback_sse2+0xf1d>
+  DB  15,88,5,195,17,0,0                  ; addps         0x11c3(%rip),%xmm0        # 5730 <_sk_callback_sse2+0xf08>
+  DB  68,15,40,13,203,17,0,0              ; movaps        0x11cb(%rip),%xmm9        # 5740 <_sk_callback_sse2+0xf18>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -21414,7 +21441,7 @@ _sk_bilinear_px_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,191,17,0,0                  ; addps         0x11bf(%rip),%xmm0        # 5730 <_sk_callback_sse2+0xf2d>
+  DB  15,88,5,186,17,0,0                  ; addps         0x11ba(%rip),%xmm0        # 5750 <_sk_callback_sse2+0xf28>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -21424,8 +21451,8 @@ _sk_bilinear_ny_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,177,17,0,0                 ; addps         0x11b1(%rip),%xmm1        # 5740 <_sk_callback_sse2+0xf3d>
-  DB  68,15,40,13,185,17,0,0              ; movaps        0x11b9(%rip),%xmm9        # 5750 <_sk_callback_sse2+0xf4d>
+  DB  15,88,13,172,17,0,0                 ; addps         0x11ac(%rip),%xmm1        # 5760 <_sk_callback_sse2+0xf38>
+  DB  68,15,40,13,180,17,0,0              ; movaps        0x11b4(%rip),%xmm9        # 5770 <_sk_callback_sse2+0xf48>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -21436,7 +21463,7 @@ _sk_bilinear_py_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,167,17,0,0                 ; addps         0x11a7(%rip),%xmm1        # 5760 <_sk_callback_sse2+0xf5d>
+  DB  15,88,13,162,17,0,0                 ; addps         0x11a2(%rip),%xmm1        # 5780 <_sk_callback_sse2+0xf58>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -21446,13 +21473,13 @@ _sk_bicubic_n3x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,154,17,0,0                  ; addps         0x119a(%rip),%xmm0        # 5770 <_sk_callback_sse2+0xf6d>
-  DB  68,15,40,13,162,17,0,0              ; movaps        0x11a2(%rip),%xmm9        # 5780 <_sk_callback_sse2+0xf7d>
+  DB  15,88,5,149,17,0,0                  ; addps         0x1195(%rip),%xmm0        # 5790 <_sk_callback_sse2+0xf68>
+  DB  68,15,40,13,157,17,0,0              ; movaps        0x119d(%rip),%xmm9        # 57a0 <_sk_callback_sse2+0xf78>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,158,17,0,0              ; mulps         0x119e(%rip),%xmm9        # 5790 <_sk_callback_sse2+0xf8d>
-  DB  68,15,88,13,166,17,0,0              ; addps         0x11a6(%rip),%xmm9        # 57a0 <_sk_callback_sse2+0xf9d>
+  DB  68,15,89,13,153,17,0,0              ; mulps         0x1199(%rip),%xmm9        # 57b0 <_sk_callback_sse2+0xf88>
+  DB  68,15,88,13,161,17,0,0              ; addps         0x11a1(%rip),%xmm9        # 57c0 <_sk_callback_sse2+0xf98>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -21463,16 +21490,16 @@ _sk_bicubic_n1x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,149,17,0,0                  ; addps         0x1195(%rip),%xmm0        # 57b0 <_sk_callback_sse2+0xfad>
-  DB  68,15,40,13,157,17,0,0              ; movaps        0x119d(%rip),%xmm9        # 57c0 <_sk_callback_sse2+0xfbd>
+  DB  15,88,5,144,17,0,0                  ; addps         0x1190(%rip),%xmm0        # 57d0 <_sk_callback_sse2+0xfa8>
+  DB  68,15,40,13,152,17,0,0              ; movaps        0x1198(%rip),%xmm9        # 57e0 <_sk_callback_sse2+0xfb8>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,161,17,0,0               ; movaps        0x11a1(%rip),%xmm8        # 57d0 <_sk_callback_sse2+0xfcd>
+  DB  68,15,40,5,156,17,0,0               ; movaps        0x119c(%rip),%xmm8        # 57f0 <_sk_callback_sse2+0xfc8>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,165,17,0,0               ; addps         0x11a5(%rip),%xmm8        # 57e0 <_sk_callback_sse2+0xfdd>
+  DB  68,15,88,5,160,17,0,0               ; addps         0x11a0(%rip),%xmm8        # 5800 <_sk_callback_sse2+0xfd8>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,169,17,0,0               ; addps         0x11a9(%rip),%xmm8        # 57f0 <_sk_callback_sse2+0xfed>
+  DB  68,15,88,5,164,17,0,0               ; addps         0x11a4(%rip),%xmm8        # 5810 <_sk_callback_sse2+0xfe8>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,173,17,0,0               ; addps         0x11ad(%rip),%xmm8        # 5800 <_sk_callback_sse2+0xffd>
+  DB  68,15,88,5,168,17,0,0               ; addps         0x11a8(%rip),%xmm8        # 5820 <_sk_callback_sse2+0xff8>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -21480,17 +21507,17 @@ _sk_bicubic_n1x_sse2 LABEL PROC
 PUBLIC _sk_bicubic_p1x_sse2
 _sk_bicubic_p1x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,167,17,0,0               ; movaps        0x11a7(%rip),%xmm8        # 5810 <_sk_callback_sse2+0x100d>
+  DB  68,15,40,5,162,17,0,0               ; movaps        0x11a2(%rip),%xmm8        # 5830 <_sk_callback_sse2+0x1008>
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,72,64                      ; movups        0x40(%rax),%xmm9
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
-  DB  68,15,40,21,163,17,0,0              ; movaps        0x11a3(%rip),%xmm10        # 5820 <_sk_callback_sse2+0x101d>
+  DB  68,15,40,21,158,17,0,0              ; movaps        0x119e(%rip),%xmm10        # 5840 <_sk_callback_sse2+0x1018>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,167,17,0,0              ; addps         0x11a7(%rip),%xmm10        # 5830 <_sk_callback_sse2+0x102d>
+  DB  68,15,88,21,162,17,0,0              ; addps         0x11a2(%rip),%xmm10        # 5850 <_sk_callback_sse2+0x1028>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,163,17,0,0              ; addps         0x11a3(%rip),%xmm10        # 5840 <_sk_callback_sse2+0x103d>
+  DB  68,15,88,21,158,17,0,0              ; addps         0x119e(%rip),%xmm10        # 5860 <_sk_callback_sse2+0x1038>
   DB  68,15,17,144,128,0,0,0              ; movups        %xmm10,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -21500,11 +21527,11 @@ _sk_bicubic_p3x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,150,17,0,0                  ; addps         0x1196(%rip),%xmm0        # 5850 <_sk_callback_sse2+0x104d>
+  DB  15,88,5,145,17,0,0                  ; addps         0x1191(%rip),%xmm0        # 5870 <_sk_callback_sse2+0x1048>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,150,17,0,0               ; mulps         0x1196(%rip),%xmm8        # 5860 <_sk_callback_sse2+0x105d>
-  DB  68,15,88,5,158,17,0,0               ; addps         0x119e(%rip),%xmm8        # 5870 <_sk_callback_sse2+0x106d>
+  DB  68,15,89,5,145,17,0,0               ; mulps         0x1191(%rip),%xmm8        # 5880 <_sk_callback_sse2+0x1058>
+  DB  68,15,88,5,153,17,0,0               ; addps         0x1199(%rip),%xmm8        # 5890 <_sk_callback_sse2+0x1068>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -21515,13 +21542,13 @@ _sk_bicubic_n3y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,140,17,0,0                 ; addps         0x118c(%rip),%xmm1        # 5880 <_sk_callback_sse2+0x107d>
-  DB  68,15,40,13,148,17,0,0              ; movaps        0x1194(%rip),%xmm9        # 5890 <_sk_callback_sse2+0x108d>
+  DB  15,88,13,135,17,0,0                 ; addps         0x1187(%rip),%xmm1        # 58a0 <_sk_callback_sse2+0x1078>
+  DB  68,15,40,13,143,17,0,0              ; movaps        0x118f(%rip),%xmm9        # 58b0 <_sk_callback_sse2+0x1088>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,144,17,0,0              ; mulps         0x1190(%rip),%xmm9        # 58a0 <_sk_callback_sse2+0x109d>
-  DB  68,15,88,13,152,17,0,0              ; addps         0x1198(%rip),%xmm9        # 58b0 <_sk_callback_sse2+0x10ad>
+  DB  68,15,89,13,139,17,0,0              ; mulps         0x118b(%rip),%xmm9        # 58c0 <_sk_callback_sse2+0x1098>
+  DB  68,15,88,13,147,17,0,0              ; addps         0x1193(%rip),%xmm9        # 58d0 <_sk_callback_sse2+0x10a8>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -21532,16 +21559,16 @@ _sk_bicubic_n1y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,134,17,0,0                 ; addps         0x1186(%rip),%xmm1        # 58c0 <_sk_callback_sse2+0x10bd>
-  DB  68,15,40,13,142,17,0,0              ; movaps        0x118e(%rip),%xmm9        # 58d0 <_sk_callback_sse2+0x10cd>
+  DB  15,88,13,129,17,0,0                 ; addps         0x1181(%rip),%xmm1        # 58e0 <_sk_callback_sse2+0x10b8>
+  DB  68,15,40,13,137,17,0,0              ; movaps        0x1189(%rip),%xmm9        # 58f0 <_sk_callback_sse2+0x10c8>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,146,17,0,0               ; movaps        0x1192(%rip),%xmm8        # 58e0 <_sk_callback_sse2+0x10dd>
+  DB  68,15,40,5,141,17,0,0               ; movaps        0x118d(%rip),%xmm8        # 5900 <_sk_callback_sse2+0x10d8>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,150,17,0,0               ; addps         0x1196(%rip),%xmm8        # 58f0 <_sk_callback_sse2+0x10ed>
+  DB  68,15,88,5,145,17,0,0               ; addps         0x1191(%rip),%xmm8        # 5910 <_sk_callback_sse2+0x10e8>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,154,17,0,0               ; addps         0x119a(%rip),%xmm8        # 5900 <_sk_callback_sse2+0x10fd>
+  DB  68,15,88,5,149,17,0,0               ; addps         0x1195(%rip),%xmm8        # 5920 <_sk_callback_sse2+0x10f8>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,158,17,0,0               ; addps         0x119e(%rip),%xmm8        # 5910 <_sk_callback_sse2+0x110d>
+  DB  68,15,88,5,153,17,0,0               ; addps         0x1199(%rip),%xmm8        # 5930 <_sk_callback_sse2+0x1108>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -21549,17 +21576,17 @@ _sk_bicubic_n1y_sse2 LABEL PROC
 PUBLIC _sk_bicubic_p1y_sse2
 _sk_bicubic_p1y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,152,17,0,0               ; movaps        0x1198(%rip),%xmm8        # 5920 <_sk_callback_sse2+0x111d>
+  DB  68,15,40,5,147,17,0,0               ; movaps        0x1193(%rip),%xmm8        # 5940 <_sk_callback_sse2+0x1118>
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,72,96                      ; movups        0x60(%rax),%xmm9
   DB  65,15,88,200                        ; addps         %xmm8,%xmm1
-  DB  68,15,40,21,147,17,0,0              ; movaps        0x1193(%rip),%xmm10        # 5930 <_sk_callback_sse2+0x112d>
+  DB  68,15,40,21,142,17,0,0              ; movaps        0x118e(%rip),%xmm10        # 5950 <_sk_callback_sse2+0x1128>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,151,17,0,0              ; addps         0x1197(%rip),%xmm10        # 5940 <_sk_callback_sse2+0x113d>
+  DB  68,15,88,21,146,17,0,0              ; addps         0x1192(%rip),%xmm10        # 5960 <_sk_callback_sse2+0x1138>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,147,17,0,0              ; addps         0x1193(%rip),%xmm10        # 5950 <_sk_callback_sse2+0x114d>
+  DB  68,15,88,21,142,17,0,0              ; addps         0x118e(%rip),%xmm10        # 5970 <_sk_callback_sse2+0x1148>
   DB  68,15,17,144,160,0,0,0              ; movups        %xmm10,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -21569,11 +21596,11 @@ _sk_bicubic_p3y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,133,17,0,0                 ; addps         0x1185(%rip),%xmm1        # 5960 <_sk_callback_sse2+0x115d>
+  DB  15,88,13,128,17,0,0                 ; addps         0x1180(%rip),%xmm1        # 5980 <_sk_callback_sse2+0x1158>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,133,17,0,0               ; mulps         0x1185(%rip),%xmm8        # 5970 <_sk_callback_sse2+0x116d>
-  DB  68,15,88,5,141,17,0,0               ; addps         0x118d(%rip),%xmm8        # 5980 <_sk_callback_sse2+0x117d>
+  DB  68,15,89,5,128,17,0,0               ; mulps         0x1180(%rip),%xmm8        # 5990 <_sk_callback_sse2+0x1168>
+  DB  68,15,88,5,136,17,0,0               ; addps         0x1188(%rip),%xmm8        # 59a0 <_sk_callback_sse2+0x1178>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -21778,11 +21805,11 @@ ALIGN 16
   DB  128,191,0,0,128,191,0               ; cmpb          $0x0,-0x40800000(%rdi)
   DB  0,224                               ; add           %ah,%al
   DB  64,0,0                              ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4a88 <.literal16+0x1d8>
+  DB  224,64                              ; loopne        4ab8 <.literal16+0x1d8>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4a8c <.literal16+0x1dc>
+  DB  224,64                              ; loopne        4abc <.literal16+0x1dc>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4a90 <.literal16+0x1e0>
+  DB  224,64                              ; loopne        4ac0 <.literal16+0x1e0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -21807,13 +21834,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4ac1 <.literal16+0x211>
+  DB  71,225,61                           ; rex.RXB       loope 4af1 <.literal16+0x211>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4ac5 <.literal16+0x215>
+  DB  71,225,61                           ; rex.RXB       loope 4af5 <.literal16+0x215>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4ac9 <.literal16+0x219>
+  DB  71,225,61                           ; rex.RXB       loope 4af9 <.literal16+0x219>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4acd <.literal16+0x21d>
+  DB  71,225,61                           ; rex.RXB       loope 4afd <.literal16+0x21d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -21838,13 +21865,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b01 <.literal16+0x251>
+  DB  71,225,61                           ; rex.RXB       loope 4b31 <.literal16+0x251>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b05 <.literal16+0x255>
+  DB  71,225,61                           ; rex.RXB       loope 4b35 <.literal16+0x255>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b09 <.literal16+0x259>
+  DB  71,225,61                           ; rex.RXB       loope 4b39 <.literal16+0x259>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b0d <.literal16+0x25d>
+  DB  71,225,61                           ; rex.RXB       loope 4b3d <.literal16+0x25d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -21869,13 +21896,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b41 <.literal16+0x291>
+  DB  71,225,61                           ; rex.RXB       loope 4b71 <.literal16+0x291>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b45 <.literal16+0x295>
+  DB  71,225,61                           ; rex.RXB       loope 4b75 <.literal16+0x295>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b49 <.literal16+0x299>
+  DB  71,225,61                           ; rex.RXB       loope 4b79 <.literal16+0x299>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b4d <.literal16+0x29d>
+  DB  71,225,61                           ; rex.RXB       loope 4b7d <.literal16+0x29d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -21900,13 +21927,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b81 <.literal16+0x2d1>
+  DB  71,225,61                           ; rex.RXB       loope 4bb1 <.literal16+0x2d1>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b85 <.literal16+0x2d5>
+  DB  71,225,61                           ; rex.RXB       loope 4bb5 <.literal16+0x2d5>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b89 <.literal16+0x2d9>
+  DB  71,225,61                           ; rex.RXB       loope 4bb9 <.literal16+0x2d9>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4b8d <.literal16+0x2dd>
+  DB  71,225,61                           ; rex.RXB       loope 4bbd <.literal16+0x2dd>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -22130,13 +22157,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4d59 <.literal16+0x4a9>
+  DB  224,7                               ; loopne        4d89 <.literal16+0x4a9>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4d5d <.literal16+0x4ad>
+  DB  224,7                               ; loopne        4d8d <.literal16+0x4ad>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4d61 <.literal16+0x4b1>
+  DB  224,7                               ; loopne        4d91 <.literal16+0x4b1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4d65 <.literal16+0x4b5>
+  DB  224,7                               ; loopne        4d95 <.literal16+0x4b5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -22160,22 +22187,18 @@ ALIGN 16
   DB  4,61                                ; add           $0x3d,%al
   DB  8,33                                ; or            %ah,(%rcx)
   DB  4,61                                ; add           $0x3d,%al
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  128,63,0                            ; cmpb          $0x0,(%rdi)
-  DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
-  DB  63                                  ; (bad)
-  DB  0,0                                 ; add           %al,(%rax)
-  DB  128,63,255                          ; cmpb          $0xff,(%rdi)
-  DB  0,0                                 ; add           %al,(%rax)
-  DB  0,255                               ; add           %bh,%bh
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  0,255                               ; add           %bh,%bh
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  0,255                               ; add           %bh,%bh
+  DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  0,129,128,128,59,129                ; add           %al,-0x7ec47f80(%rcx)
-  DB  128,128,59,129,128,128,59           ; addb          $0x3b,-0x7f7f7ec5(%rax)
-  DB  129,128,128,59,255,0,255,0,255,0    ; addl          $0xff00ff,0xff3b80(%rax)
+  DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
+  DB  128,59,129                          ; cmpb          $0x81,(%rbx)
+  DB  128,128,59,255,0,255,0              ; addb          $0x0,-0xff00c5(%rax)
+  DB  255,0                               ; incl          (%rax)
   DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,0                                 ; add           %al,(%rax)
@@ -22205,11 +22228,11 @@ ALIGN 16
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4e4b <.literal16+0x59b>
+  DB  127,67                              ; jg            4e6b <.literal16+0x58b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4e4f <.literal16+0x59f>
+  DB  127,67                              ; jg            4e6f <.literal16+0x58f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4e53 <.literal16+0x5a3>
+  DB  127,67                              ; jg            4e73 <.literal16+0x593>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,129,128,128,59           ; addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -22224,16 +22247,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e44 <.literal16+0x594>
+  DB  127,0                               ; jg            4e64 <.literal16+0x584>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e48 <.literal16+0x598>
+  DB  127,0                               ; jg            4e68 <.literal16+0x588>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e4c <.literal16+0x59c>
+  DB  127,0                               ; jg            4e6c <.literal16+0x58c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e50 <.literal16+0x5a0>
+  DB  127,0                               ; jg            4e70 <.literal16+0x590>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -22242,7 +22265,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4ed5 <.literal16+0x625>
+  DB  119,115                             ; ja            4ef5 <.literal16+0x615>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -22253,7 +22276,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4e39 <.literal16+0x589>
+  DB  117,191                             ; jne           4e59 <.literal16+0x579>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -22265,7 +22288,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38e7a <_sk_callback_sse2+0xffffffffe9a34677>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38e9a <_sk_callback_sse2+0xffffffffe9a34672>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -22319,16 +22342,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4f14 <.literal16+0x664>
+  DB  127,0                               ; jg            4f34 <.literal16+0x654>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4f18 <.literal16+0x668>
+  DB  127,0                               ; jg            4f38 <.literal16+0x658>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4f1c <.literal16+0x66c>
+  DB  127,0                               ; jg            4f3c <.literal16+0x65c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4f20 <.literal16+0x670>
+  DB  127,0                               ; jg            4f40 <.literal16+0x660>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -22337,7 +22360,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4fa5 <.literal16+0x6f5>
+  DB  119,115                             ; ja            4fc5 <.literal16+0x6e5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -22348,7 +22371,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4f09 <.literal16+0x659>
+  DB  117,191                             ; jne           4f29 <.literal16+0x649>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -22360,7 +22383,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38f4a <_sk_callback_sse2+0xffffffffe9a34747>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38f6a <_sk_callback_sse2+0xffffffffe9a34742>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -22414,16 +22437,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fe4 <.literal16+0x734>
+  DB  127,0                               ; jg            5004 <.literal16+0x724>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fe8 <.literal16+0x738>
+  DB  127,0                               ; jg            5008 <.literal16+0x728>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fec <.literal16+0x73c>
+  DB  127,0                               ; jg            500c <.literal16+0x72c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4ff0 <.literal16+0x740>
+  DB  127,0                               ; jg            5010 <.literal16+0x730>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -22432,7 +22455,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5075 <.literal16+0x7c5>
+  DB  119,115                             ; ja            5095 <.literal16+0x7b5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -22443,7 +22466,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4fd9 <.literal16+0x729>
+  DB  117,191                             ; jne           4ff9 <.literal16+0x719>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -22455,7 +22478,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3901a <_sk_callback_sse2+0xffffffffe9a34817>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3903a <_sk_callback_sse2+0xffffffffe9a34812>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -22509,16 +22532,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            50b4 <.literal16+0x804>
+  DB  127,0                               ; jg            50d4 <.literal16+0x7f4>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            50b8 <.literal16+0x808>
+  DB  127,0                               ; jg            50d8 <.literal16+0x7f8>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            50bc <.literal16+0x80c>
+  DB  127,0                               ; jg            50dc <.literal16+0x7fc>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            50c0 <.literal16+0x810>
+  DB  127,0                               ; jg            50e0 <.literal16+0x800>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -22527,7 +22550,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5145 <.literal16+0x895>
+  DB  119,115                             ; ja            5165 <.literal16+0x885>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -22538,7 +22561,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           50a9 <.literal16+0x7f9>
+  DB  117,191                             ; jne           50c9 <.literal16+0x7e9>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -22550,7 +22573,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a390ea <_sk_callback_sse2+0xffffffffe9a348e7>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3910a <_sk_callback_sse2+0xffffffffe9a348e2>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -22600,13 +22623,13 @@ ALIGN 16
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
-  DB  127,67                              ; jg            51c7 <.literal16+0x917>
+  DB  127,67                              ; jg            51e7 <.literal16+0x907>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            51cb <.literal16+0x91b>
+  DB  127,67                              ; jg            51eb <.literal16+0x90b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            51cf <.literal16+0x91f>
+  DB  127,67                              ; jg            51ef <.literal16+0x90f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            51d3 <.literal16+0x923>
+  DB  127,67                              ; jg            51f3 <.literal16+0x913>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -22653,16 +22676,16 @@ ALIGN 16
   DB  128,3,62                            ; addb          $0x3e,(%rbx)
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5253 <.literal16+0x9a3>
+  DB  118,63                              ; jbe           5273 <.literal16+0x993>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5257 <.literal16+0x9a7>
+  DB  118,63                              ; jbe           5277 <.literal16+0x997>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           525b <.literal16+0x9ab>
+  DB  118,63                              ; jbe           527b <.literal16+0x99b>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           525f <.literal16+0x9af>
+  DB  118,63                              ; jbe           527f <.literal16+0x99f>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
@@ -22674,11 +22697,11 @@ ALIGN 16
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            529b <.literal16+0x9eb>
+  DB  127,67                              ; jg            52bb <.literal16+0x9db>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            529f <.literal16+0x9ef>
+  DB  127,67                              ; jg            52bf <.literal16+0x9df>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            52a3 <.literal16+0x9f3>
+  DB  127,67                              ; jg            52c3 <.literal16+0x9e3>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,0,0,128,63               ; addb          $0x3f,-0x7fffffc5(%rax)
@@ -22718,13 +22741,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        52e9 <.literal16+0xa39>
+  DB  224,7                               ; loopne        5309 <.literal16+0xa29>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        52ed <.literal16+0xa3d>
+  DB  224,7                               ; loopne        530d <.literal16+0xa2d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        52f1 <.literal16+0xa41>
+  DB  224,7                               ; loopne        5311 <.literal16+0xa31>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        52f5 <.literal16+0xa45>
+  DB  224,7                               ; loopne        5315 <.literal16+0xa35>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -22770,13 +22793,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5359 <.literal16+0xaa9>
+  DB  224,7                               ; loopne        5379 <.literal16+0xa99>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        535d <.literal16+0xaad>
+  DB  224,7                               ; loopne        537d <.literal16+0xa9d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5361 <.literal16+0xab1>
+  DB  224,7                               ; loopne        5381 <.literal16+0xaa1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5365 <.literal16+0xab5>
+  DB  224,7                               ; loopne        5385 <.literal16+0xaa5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -22814,13 +22837,13 @@ ALIGN 16
   DB  65,0,0                              ; add           %al,(%r8)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            53f6 <.literal16+0xb46>
+  DB  124,66                              ; jl            5416 <.literal16+0xb36>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            53fa <.literal16+0xb4a>
+  DB  124,66                              ; jl            541a <.literal16+0xb3a>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            53fe <.literal16+0xb4e>
+  DB  124,66                              ; jl            541e <.literal16+0xb3e>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            5402 <.literal16+0xb52>
+  DB  124,66                              ; jl            5422 <.literal16+0xb42>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,240                               ; add           %dh,%al
@@ -22910,13 +22933,13 @@ ALIGN 16
   DB  136,136,61,137,136,136              ; mov           %cl,-0x777776c3(%rax)
   DB  61,137,136,136,61                   ; cmp           $0x3d888889,%eax
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5505 <.literal16+0xc55>
+  DB  112,65                              ; jo            5525 <.literal16+0xc45>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5509 <.literal16+0xc59>
+  DB  112,65                              ; jo            5529 <.literal16+0xc49>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            550d <.literal16+0xc5d>
+  DB  112,65                              ; jo            552d <.literal16+0xc4d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5511 <.literal16+0xc61>
+  DB  112,65                              ; jo            5531 <.literal16+0xc51>
   DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  255,0                               ; incl          (%rax)
@@ -22938,11 +22961,11 @@ ALIGN 16
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,0,0,127,67               ; addb          $0x43,0x7f00003b(%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            555b <.literal16+0xcab>
+  DB  127,67                              ; jg            557b <.literal16+0xc9b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            555f <.literal16+0xcaf>
+  DB  127,67                              ; jg            557f <.literal16+0xc9f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            5563 <.literal16+0xcb3>
+  DB  127,67                              ; jg            5583 <.literal16+0xca3>
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
@@ -23018,13 +23041,13 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  255                                 ; (bad)
-  DB  127,71                              ; jg            564b <.literal16+0xd9b>
+  DB  127,71                              ; jg            566b <.literal16+0xd8b>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            564f <.literal16+0xd9f>
+  DB  127,71                              ; jg            566f <.literal16+0xd8f>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            5653 <.literal16+0xda3>
+  DB  127,71                              ; jg            5673 <.literal16+0xd93>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            5657 <.literal16+0xda7>
+  DB  127,71                              ; jg            5677 <.literal16+0xd97>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -23177,11 +23200,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         57b2 <.literal16+0xf02>
+  DB  62,114,28                           ; jb,pt         57d2 <.literal16+0xef2>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         57b6 <.literal16+0xf06>
+  DB  62,114,28                           ; jb,pt         57d6 <.literal16+0xef6>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         57ba <.literal16+0xf0a>
+  DB  62,114,28                           ; jb,pt         57da <.literal16+0xefa>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -23225,7 +23248,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e645 <_sk_callback_sse2+0x3d639e42>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e665 <_sk_callback_sse2+0x3d639e3d>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -23251,7 +23274,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e685 <_sk_callback_sse2+0x3d639e82>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e6a5 <_sk_callback_sse2+0x3d639e7d>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -23260,13 +23283,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            587e <.literal16+0xfce>
+  DB  114,28                              ; jb            589e <.literal16+0xfbe>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5882 <.literal16+0xfd2>
+  DB  62,114,28                           ; jb,pt         58a2 <.literal16+0xfc2>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5886 <.literal16+0xfd6>
+  DB  62,114,28                           ; jb,pt         58a6 <.literal16+0xfc6>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         588a <.literal16+0xfda>
+  DB  62,114,28                           ; jb,pt         58aa <.literal16+0xfca>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -23287,11 +23310,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         58c2 <.literal16+0x1012>
+  DB  62,114,28                           ; jb,pt         58e2 <.literal16+0x1002>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         58c6 <.literal16+0x1016>
+  DB  62,114,28                           ; jb,pt         58e6 <.literal16+0x1006>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         58ca <.literal16+0x101a>
+  DB  62,114,28                           ; jb,pt         58ea <.literal16+0x100a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -23335,7 +23358,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e755 <_sk_callback_sse2+0x3d639f52>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e775 <_sk_callback_sse2+0x3d639f4d>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -23361,7 +23384,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e795 <_sk_callback_sse2+0x3d639f92>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e7b5 <_sk_callback_sse2+0x3d639f8d>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -23370,13 +23393,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            598e <.literal16+0x10de>
+  DB  114,28                              ; jb            59ae <.literal16+0x10ce>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5992 <_sk_callback_sse2+0x118f>
+  DB  62,114,28                           ; jb,pt         59b2 <_sk_callback_sse2+0x118a>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5996 <_sk_callback_sse2+0x1193>
+  DB  62,114,28                           ; jb,pt         59b6 <_sk_callback_sse2+0x118e>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         599a <_sk_callback_sse2+0x1197>
+  DB  62,114,28                           ; jb,pt         59ba <_sk_callback_sse2+0x1192>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
index 8c4c842..a78da65 100644 (file)
@@ -703,7 +703,7 @@ STAGE(lerp_565) {
     r = lerp(dr, r, cr);
     g = lerp(dg, g, cg);
     b = lerp(db, b, cb);
-    a = 1.0f;
+    a = max(lerp(da, a, cr), lerp(da, a, cg), lerp(da, a, cb));
 }
 
 STAGE(load_tables) {