clamp to premul in dither
authorMike Klein <mtklein@chromium.org>
Tue, 16 May 2017 16:03:15 +0000 (12:03 -0400)
committerSkia Commit-Bot <skia-commit-bot@chromium.org>
Tue, 16 May 2017 21:10:14 +0000 (21:10 +0000)
Dither can bump color values above alpha (duh), or below
zero (duh), so clamp back to premul after dithering.

BUG=skia:6644,skia:6643

Change-Id: Ida107e866380e06130af0d01467117bca929ba44
Reviewed-on: https://skia-review.googlesource.com/17070
Reviewed-by: Herb Derby <herb@google.com>
Reviewed-by: Florin Malita <fmalita@chromium.org>
Commit-Queue: Mike Klein <mtklein@chromium.org>

gm/bug6643.cpp [new file with mode: 0644]
gn/gm.gni
src/jumper/SkJumper_generated.S
src/jumper/SkJumper_generated_win.S
src/jumper/SkJumper_stages.cpp

diff --git a/gm/bug6643.cpp b/gm/bug6643.cpp
new file mode 100644 (file)
index 0000000..abe79fa
--- /dev/null
@@ -0,0 +1,29 @@
+/*
+ * Copyright 2017 Google Inc.
+ *
+ * Use of this source code is governed by a BSD-style license that can be
+ * found in the LICENSE file.
+ */
+
+#include "SkGradientShader.h"
+#include "SkPictureRecorder.h"
+#include "gm.h"
+
+DEF_SIMPLE_GM(bug6643, canvas, 200, 200) {
+    SkColor colors[] = { SK_ColorTRANSPARENT, SK_ColorGREEN, SK_ColorTRANSPARENT };
+
+    SkPaint p;
+    p.setAntiAlias(true);
+    p.setShader(SkGradientShader::MakeSweep(100, 100, colors, nullptr, SK_ARRAY_COUNT(colors),
+                                            SkGradientShader::kInterpolateColorsInPremul_Flag,
+                                            nullptr));
+
+    SkPictureRecorder recorder;
+    recorder.beginRecording(200, 200)->drawPaint(p);
+
+    p.setShader(SkShader::MakePictureShader(recorder.finishRecordingAsPicture(),
+                                            SkShader::kRepeat_TileMode, SkShader::kRepeat_TileMode,
+                                            nullptr, nullptr));
+    canvas->drawColor(SK_ColorWHITE);
+    canvas->drawPaint(p);
+}
index 2e1c8af..5e33641 100644 (file)
--- a/gn/gm.gni
+++ b/gn/gm.gni
@@ -49,6 +49,7 @@ gm_sources = [
   "$_gm/blurs.cpp",
   "$_gm/bmpfilterqualityrepeat.cpp",
   "$_gm/bug5252.cpp",
+  "$_gm/bug6643.cpp",
   "$_gm/bug530095.cpp",
   "$_gm/bug615686.cpp",
   "$_gm/cgm.c",
index a5a5add..d9e1d05 100644 (file)
@@ -119,10 +119,10 @@ _sk_dither_aarch64:
   .long  0x4e341e14                          // and           v20.16b, v16.16b, v20.16b
   .long  0x4f255652                          // shl           v18.4s, v18.4s, #5
   .long  0x4e331e10                          // and           v16.16b, v16.16b, v19.16b
-  .long  0x4f225694                          // shl           v20.4s, v20.4s, #2
   .long  0x4eb21e31                          // orr           v17.16b, v17.16b, v18.16b
+  .long  0x4f225694                          // shl           v20.4s, v20.4s, #2
   .long  0x52a79009                          // mov           w9, #0x3c800000
-  .long  0xbd400913                          // ldr           s19, [x8, #8]
+  .long  0xbd400912                          // ldr           s18, [x8, #8]
   .long  0x6f3f0610                          // ushr          v16.4s, v16.4s, #1
   .long  0x4eb41e31                          // orr           v17.16b, v17.16b, v20.16b
   .long  0x4e040d36                          // dup           v22.4s, w9
@@ -131,10 +131,17 @@ _sk_dither_aarch64:
   .long  0x4e040d35                          // dup           v21.4s, w9
   .long  0x4e21da10                          // scvtf         v16.4s, v16.4s
   .long  0x4e30ced5                          // fmla          v21.4s, v22.4s, v16.4s
-  .long  0x4f9392b0                          // fmul          v16.4s, v21.4s, v19.s[0]
+  .long  0x4f9292b0                          // fmul          v16.4s, v21.4s, v18.s[0]
   .long  0x4e20d600                          // fadd          v0.4s, v16.4s, v0.4s
   .long  0x4e21d601                          // fadd          v1.4s, v16.4s, v1.4s
   .long  0x4e22d602                          // fadd          v2.4s, v16.4s, v2.4s
+  .long  0x6f00e413                          // movi          v19.2d, #0x0
+  .long  0x4ea3f400                          // fmin          v0.4s, v0.4s, v3.4s
+  .long  0x4ea3f421                          // fmin          v1.4s, v1.4s, v3.4s
+  .long  0x4ea3f442                          // fmin          v2.4s, v2.4s, v3.4s
+  .long  0x4e20f660                          // fmax          v0.4s, v19.4s, v0.4s
+  .long  0x4e21f661                          // fmax          v1.4s, v19.4s, v1.4s
+  .long  0x4e22f662                          // fmax          v2.4s, v19.4s, v2.4s
   .long  0xd61f0060                          // br            x3
 
 HIDDEN _sk_constant_color_aarch64
@@ -2706,9 +2713,9 @@ FUNCTION(_sk_gather_i8_aarch64)
 _sk_gather_i8_aarch64:
   .long  0xaa0103e8                          // mov           x8, x1
   .long  0xf8408429                          // ldr           x9, [x1], #8
-  .long  0xb4000069                          // cbz           x9, 2444 <sk_gather_i8_aarch64+0x14>
+  .long  0xb4000069                          // cbz           x9, 2460 <sk_gather_i8_aarch64+0x14>
   .long  0xaa0903ea                          // mov           x10, x9
-  .long  0x14000003                          // b             244c <sk_gather_i8_aarch64+0x1c>
+  .long  0x14000003                          // b             2468 <sk_gather_i8_aarch64+0x1c>
   .long  0xf940050a                          // ldr           x10, [x8, #8]
   .long  0x91004101                          // add           x1, x8, #0x10
   .long  0xf8410548                          // ldr           x8, [x10], #16
@@ -3640,7 +3647,7 @@ _sk_gradient_aarch64:
   .long  0x6f00e411                          // movi          v17.2d, #0x0
   .long  0xf9400109                          // ldr           x9, [x8]
   .long  0xf100093f                          // cmp           x9, #0x2
-  .long  0x540001c3                          // b.cc          30b0 <sk_gradient_aarch64+0x58>  // b.lo, b.ul, b.last
+  .long  0x540001c3                          // b.cc          30cc <sk_gradient_aarch64+0x58>  // b.lo, b.ul, b.last
   .long  0xf940250a                          // ldr           x10, [x8, #72]
   .long  0xd1000529                          // sub           x9, x9, #0x1
   .long  0x6f00e401                          // movi          v1.2d, #0x0
@@ -3651,7 +3658,7 @@ _sk_gradient_aarch64:
   .long  0x6e23e403                          // fcmge         v3.4s, v0.4s, v3.4s
   .long  0x4e221c63                          // and           v3.16b, v3.16b, v2.16b
   .long  0x4ea18461                          // add           v1.4s, v3.4s, v1.4s
-  .long  0xb5ffff69                          // cbnz          x9, 3090 <sk_gradient_aarch64+0x38>
+  .long  0xb5ffff69                          // cbnz          x9, 30ac <sk_gradient_aarch64+0x38>
   .long  0x6f20a431                          // uxtl2         v17.2d, v1.4s
   .long  0x2f20a421                          // uxtl          v1.2d, v1.2s
   .long  0xa940b10a                          // ldp           x10, x12, [x8, #8]
@@ -4239,19 +4246,27 @@ _sk_dither_vfp4:
   .long  0xf2e22532                          // vshl.s32      d18, d18, #2
   .long  0xf3ff1033                          // vshr.u32      d17, d19, #1
   .long  0xf26001b2                          // vorr          d16, d16, d18
+  .long  0xf2c03010                          // vmov.i32      d19, #0
   .long  0xf26001b1                          // vorr          d16, d16, d17
   .long  0xee813b90                          // vdup.32       d17, r3
   .long  0xf3fb0620                          // vcvt.f32.s32  d16, d16
   .long  0xf3400db1                          // vmul.f32      d16, d16, d17
-  .long  0xeddf1b07                          // vldr          d17, [pc, #28]
+  .long  0xeddf1b0e                          // vldr          d17, [pc, #56]
   .long  0xf2400da1                          // vadd.f32      d16, d16, d17
   .long  0xf4e41c9f                          // vld1.32       {d17[]}, [r4 :32]
   .long  0xf3410db0                          // vmul.f32      d16, d17, d16
-  .long  0xf2000d80                          // vadd.f32      d0, d16, d0
-  .long  0xf2001d81                          // vadd.f32      d1, d16, d1
-  .long  0xf2002d82                          // vadd.f32      d2, d16, d2
+  .long  0xf2401d80                          // vadd.f32      d17, d16, d0
+  .long  0xf2402d81                          // vadd.f32      d18, d16, d1
+  .long  0xf2400d82                          // vadd.f32      d16, d16, d2
+  .long  0xf2611f83                          // vmin.f32      d17, d17, d3
+  .long  0xf2622f83                          // vmin.f32      d18, d18, d3
+  .long  0xf2600f83                          // vmin.f32      d16, d16, d3
+  .long  0xf2030fa1                          // vmax.f32      d0, d19, d17
+  .long  0xf2031fa2                          // vmax.f32      d1, d19, d18
+  .long  0xf2032fa0                          // vmax.f32      d2, d19, d16
   .long  0xe8bd4010                          // pop           {r4, lr}
   .long  0xe12fff1c                          // bx            ip
+  .long  0xe320f000                          // nop           {0}
   .long  0xbefc0000                          // .word         0xbefc0000
   .long  0xbefc0000                          // .word         0xbefc0000
 
@@ -8117,7 +8132,7 @@ _sk_gradient_vfp4:
   .long  0xf2c00010                          // vmov.i32      d16, #0
   .long  0xe59c3000                          // ldr           r3, [ip]
   .long  0xe3530002                          // cmp           r3, #2
-  .long  0x3a00000b                          // bcc           3644 <sk_gradient_vfp4+0x50>
+  .long  0x3a00000b                          // bcc           3664 <sk_gradient_vfp4+0x50>
   .long  0xe59c4024                          // ldr           r4, [ip, #36]
   .long  0xf2c01010                          // vmov.i32      d17, #0
   .long  0xf2c02011                          // vmov.i32      d18, #1
@@ -8129,7 +8144,7 @@ _sk_gradient_vfp4:
   .long  0xf3403e23                          // vcge.f32      d19, d0, d19
   .long  0xf35231b1                          // vbsl          d19, d18, d17
   .long  0xf26308a0                          // vadd.i32      d16, d19, d16
-  .long  0x1afffff9                          // bne           362c <sk_gradient_vfp4+0x38>
+  .long  0x1afffff9                          // bne           364c <sk_gradient_vfp4+0x38>
   .long  0xee303b90                          // vmov.32       r3, d16[1]
   .long  0xe59c7010                          // ldr           r7, [ip, #16]
   .long  0xee10eb90                          // vmov.32       lr, d16[0]
@@ -8722,14 +8737,14 @@ _sk_seed_shader_hsw:
   .byte  197,249,110,199                     // vmovd         %edi,%xmm0
   .byte  196,226,125,88,192                  // vpbroadcastd  %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,185,70,0,0        // vbroadcastss  0x46b9(%rip),%ymm1        # 477c <_sk_callback_hsw+0x128>
+  .byte  196,226,125,24,13,213,70,0,0        // vbroadcastss  0x46d5(%rip),%ymm1        # 4798 <_sk_callback_hsw+0x128>
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
   .byte  197,252,88,2                        // vaddps        (%rdx),%ymm0,%ymm0
   .byte  196,226,125,24,16                   // vbroadcastss  (%rax),%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  197,236,88,201                      // vaddps        %ymm1,%ymm2,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,21,157,70,0,0        // vbroadcastss  0x469d(%rip),%ymm2        # 4780 <_sk_callback_hsw+0x12c>
+  .byte  196,226,125,24,21,185,70,0,0        // vbroadcastss  0x46b9(%rip),%ymm2        # 479c <_sk_callback_hsw+0x12c>
   .byte  197,228,87,219                      // vxorps        %ymm3,%ymm3,%ymm3
   .byte  197,220,87,228                      // vxorps        %ymm4,%ymm4,%ymm4
   .byte  197,212,87,237                      // vxorps        %ymm5,%ymm5,%ymm5
@@ -8750,13 +8765,13 @@ _sk_dither_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  196,66,125,88,8                     // vpbroadcastd  (%r8),%ymm9
   .byte  196,65,61,239,201                   // vpxor         %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,88,21,92,70,0,0          // vpbroadcastd  0x465c(%rip),%ymm10        # 4784 <_sk_callback_hsw+0x130>
+  .byte  196,98,125,88,21,120,70,0,0         // vpbroadcastd  0x4678(%rip),%ymm10        # 47a0 <_sk_callback_hsw+0x130>
   .byte  196,65,53,219,218                   // vpand         %ymm10,%ymm9,%ymm11
   .byte  196,193,37,114,243,5                // vpslld        $0x5,%ymm11,%ymm11
   .byte  196,65,61,219,210                   // vpand         %ymm10,%ymm8,%ymm10
   .byte  196,193,45,114,242,4                // vpslld        $0x4,%ymm10,%ymm10
-  .byte  196,98,125,88,37,65,70,0,0          // vpbroadcastd  0x4641(%rip),%ymm12        # 4788 <_sk_callback_hsw+0x134>
-  .byte  196,98,125,88,45,60,70,0,0          // vpbroadcastd  0x463c(%rip),%ymm13        # 478c <_sk_callback_hsw+0x138>
+  .byte  196,98,125,88,37,93,70,0,0          // vpbroadcastd  0x465d(%rip),%ymm12        # 47a4 <_sk_callback_hsw+0x134>
+  .byte  196,98,125,88,45,88,70,0,0          // vpbroadcastd  0x4658(%rip),%ymm13        # 47a8 <_sk_callback_hsw+0x138>
   .byte  196,65,53,219,245                   // vpand         %ymm13,%ymm9,%ymm14
   .byte  196,193,13,114,246,2                // vpslld        $0x2,%ymm14,%ymm14
   .byte  196,65,61,219,237                   // vpand         %ymm13,%ymm8,%ymm13
@@ -8771,14 +8786,21 @@ _sk_dither_hsw:
   .byte  196,65,61,235,194                   // vpor          %ymm10,%ymm8,%ymm8
   .byte  196,65,61,235,193                   // vpor          %ymm9,%ymm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,238,69,0,0         // vbroadcastss  0x45ee(%rip),%ymm9        # 4790 <_sk_callback_hsw+0x13c>
-  .byte  196,98,125,24,21,233,69,0,0         // vbroadcastss  0x45e9(%rip),%ymm10        # 4794 <_sk_callback_hsw+0x140>
+  .byte  196,98,125,24,13,10,70,0,0          // vbroadcastss  0x460a(%rip),%ymm9        # 47ac <_sk_callback_hsw+0x13c>
+  .byte  196,98,125,24,21,5,70,0,0           // vbroadcastss  0x4605(%rip),%ymm10        # 47b0 <_sk_callback_hsw+0x140>
   .byte  196,66,61,184,209                   // vfmadd231ps   %ymm9,%ymm8,%ymm10
   .byte  196,98,125,24,64,8                  // vbroadcastss  0x8(%rax),%ymm8
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
   .byte  197,188,88,192                      // vaddps        %ymm0,%ymm8,%ymm0
   .byte  197,188,88,201                      // vaddps        %ymm1,%ymm8,%ymm1
   .byte  197,188,88,210                      // vaddps        %ymm2,%ymm8,%ymm2
+  .byte  197,252,93,195                      // vminps        %ymm3,%ymm0,%ymm0
+  .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
+  .byte  197,188,95,192                      // vmaxps        %ymm0,%ymm8,%ymm0
+  .byte  197,244,93,203                      // vminps        %ymm3,%ymm1,%ymm1
+  .byte  197,188,95,201                      // vmaxps        %ymm1,%ymm8,%ymm1
+  .byte  197,236,93,211                      // vminps        %ymm3,%ymm2,%ymm2
+  .byte  197,188,95,210                      // vmaxps        %ymm2,%ymm8,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -8834,7 +8856,7 @@ HIDDEN _sk_srcatop_hsw
 FUNCTION(_sk_srcatop_hsw)
 _sk_srcatop_hsw:
   .byte  197,252,89,199                      // vmulps        %ymm7,%ymm0,%ymm0
-  .byte  196,98,125,24,5,93,69,0,0           // vbroadcastss  0x455d(%rip),%ymm8        # 4798 <_sk_callback_hsw+0x144>
+  .byte  196,98,125,24,5,92,69,0,0           // vbroadcastss  0x455c(%rip),%ymm8        # 47b4 <_sk_callback_hsw+0x144>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,226,61,184,196                  // vfmadd231ps   %ymm4,%ymm8,%ymm0
   .byte  197,244,89,207                      // vmulps        %ymm7,%ymm1,%ymm1
@@ -8850,7 +8872,7 @@ HIDDEN _sk_dstatop_hsw
 .globl _sk_dstatop_hsw
 FUNCTION(_sk_dstatop_hsw)
 _sk_dstatop_hsw:
-  .byte  196,98,125,24,5,48,69,0,0           // vbroadcastss  0x4530(%rip),%ymm8        # 479c <_sk_callback_hsw+0x148>
+  .byte  196,98,125,24,5,47,69,0,0           // vbroadcastss  0x452f(%rip),%ymm8        # 47b8 <_sk_callback_hsw+0x148>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  196,226,101,184,196                 // vfmadd231ps   %ymm4,%ymm3,%ymm0
@@ -8889,7 +8911,7 @@ HIDDEN _sk_srcout_hsw
 .globl _sk_srcout_hsw
 FUNCTION(_sk_srcout_hsw)
 _sk_srcout_hsw:
-  .byte  196,98,125,24,5,215,68,0,0          // vbroadcastss  0x44d7(%rip),%ymm8        # 47a0 <_sk_callback_hsw+0x14c>
+  .byte  196,98,125,24,5,214,68,0,0          // vbroadcastss  0x44d6(%rip),%ymm8        # 47bc <_sk_callback_hsw+0x14c>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -8902,7 +8924,7 @@ HIDDEN _sk_dstout_hsw
 .globl _sk_dstout_hsw
 FUNCTION(_sk_dstout_hsw)
 _sk_dstout_hsw:
-  .byte  196,226,125,24,5,186,68,0,0         // vbroadcastss  0x44ba(%rip),%ymm0        # 47a4 <_sk_callback_hsw+0x150>
+  .byte  196,226,125,24,5,185,68,0,0         // vbroadcastss  0x44b9(%rip),%ymm0        # 47c0 <_sk_callback_hsw+0x150>
   .byte  197,252,92,219                      // vsubps        %ymm3,%ymm0,%ymm3
   .byte  197,228,89,196                      // vmulps        %ymm4,%ymm3,%ymm0
   .byte  197,228,89,205                      // vmulps        %ymm5,%ymm3,%ymm1
@@ -8915,7 +8937,7 @@ HIDDEN _sk_srcover_hsw
 .globl _sk_srcover_hsw
 FUNCTION(_sk_srcover_hsw)
 _sk_srcover_hsw:
-  .byte  196,98,125,24,5,157,68,0,0          // vbroadcastss  0x449d(%rip),%ymm8        # 47a8 <_sk_callback_hsw+0x154>
+  .byte  196,98,125,24,5,156,68,0,0          // vbroadcastss  0x449c(%rip),%ymm8        # 47c4 <_sk_callback_hsw+0x154>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,93,184,192                  // vfmadd231ps   %ymm8,%ymm4,%ymm0
   .byte  196,194,85,184,200                  // vfmadd231ps   %ymm8,%ymm5,%ymm1
@@ -8928,7 +8950,7 @@ HIDDEN _sk_dstover_hsw
 .globl _sk_dstover_hsw
 FUNCTION(_sk_dstover_hsw)
 _sk_dstover_hsw:
-  .byte  196,98,125,24,5,124,68,0,0          // vbroadcastss  0x447c(%rip),%ymm8        # 47ac <_sk_callback_hsw+0x158>
+  .byte  196,98,125,24,5,123,68,0,0          // vbroadcastss  0x447b(%rip),%ymm8        # 47c8 <_sk_callback_hsw+0x158>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  196,226,61,168,196                  // vfmadd213ps   %ymm4,%ymm8,%ymm0
   .byte  196,226,61,168,205                  // vfmadd213ps   %ymm5,%ymm8,%ymm1
@@ -8952,7 +8974,7 @@ HIDDEN _sk_multiply_hsw
 .globl _sk_multiply_hsw
 FUNCTION(_sk_multiply_hsw)
 _sk_multiply_hsw:
-  .byte  196,98,125,24,5,71,68,0,0           // vbroadcastss  0x4447(%rip),%ymm8        # 47b0 <_sk_callback_hsw+0x15c>
+  .byte  196,98,125,24,5,70,68,0,0           // vbroadcastss  0x4446(%rip),%ymm8        # 47cc <_sk_callback_hsw+0x15c>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,208                       // vmulps        %ymm0,%ymm9,%ymm10
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -9000,7 +9022,7 @@ HIDDEN _sk_xor__hsw
 .globl _sk_xor__hsw
 FUNCTION(_sk_xor__hsw)
 _sk_xor__hsw:
-  .byte  196,98,125,24,5,194,67,0,0          // vbroadcastss  0x43c2(%rip),%ymm8        # 47b4 <_sk_callback_hsw+0x160>
+  .byte  196,98,125,24,5,193,67,0,0          // vbroadcastss  0x43c1(%rip),%ymm8        # 47d0 <_sk_callback_hsw+0x160>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -9034,7 +9056,7 @@ _sk_darken_hsw:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,95,209                  // vmaxps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,74,67,0,0           // vbroadcastss  0x434a(%rip),%ymm8        # 47b8 <_sk_callback_hsw+0x164>
+  .byte  196,98,125,24,5,73,67,0,0           // vbroadcastss  0x4349(%rip),%ymm8        # 47d4 <_sk_callback_hsw+0x164>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -9059,7 +9081,7 @@ _sk_lighten_hsw:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,249,66,0,0          // vbroadcastss  0x42f9(%rip),%ymm8        # 47bc <_sk_callback_hsw+0x168>
+  .byte  196,98,125,24,5,248,66,0,0          // vbroadcastss  0x42f8(%rip),%ymm8        # 47d8 <_sk_callback_hsw+0x168>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -9087,7 +9109,7 @@ _sk_difference_hsw:
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,156,66,0,0          // vbroadcastss  0x429c(%rip),%ymm8        # 47c0 <_sk_callback_hsw+0x16c>
+  .byte  196,98,125,24,5,155,66,0,0          // vbroadcastss  0x429b(%rip),%ymm8        # 47dc <_sk_callback_hsw+0x16c>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -9109,7 +9131,7 @@ _sk_exclusion_hsw:
   .byte  197,236,89,214                      // vmulps        %ymm6,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,90,66,0,0           // vbroadcastss  0x425a(%rip),%ymm8        # 47c4 <_sk_callback_hsw+0x170>
+  .byte  196,98,125,24,5,89,66,0,0           // vbroadcastss  0x4259(%rip),%ymm8        # 47e0 <_sk_callback_hsw+0x170>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  196,194,69,184,216                  // vfmadd231ps   %ymm8,%ymm7,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -9119,7 +9141,7 @@ HIDDEN _sk_colorburn_hsw
 .globl _sk_colorburn_hsw
 FUNCTION(_sk_colorburn_hsw)
 _sk_colorburn_hsw:
-  .byte  196,98,125,24,5,72,66,0,0           // vbroadcastss  0x4248(%rip),%ymm8        # 47c8 <_sk_callback_hsw+0x174>
+  .byte  196,98,125,24,5,71,66,0,0           // vbroadcastss  0x4247(%rip),%ymm8        # 47e4 <_sk_callback_hsw+0x174>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,216                       // vmulps        %ymm0,%ymm9,%ymm11
   .byte  196,65,44,87,210                    // vxorps        %ymm10,%ymm10,%ymm10
@@ -9177,7 +9199,7 @@ HIDDEN _sk_colordodge_hsw
 FUNCTION(_sk_colordodge_hsw)
 _sk_colordodge_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
-  .byte  196,98,125,24,13,83,65,0,0          // vbroadcastss  0x4153(%rip),%ymm9        # 47cc <_sk_callback_hsw+0x178>
+  .byte  196,98,125,24,13,82,65,0,0          // vbroadcastss  0x4152(%rip),%ymm9        # 47e8 <_sk_callback_hsw+0x178>
   .byte  197,52,92,215                       // vsubps        %ymm7,%ymm9,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,52,92,203                       // vsubps        %ymm3,%ymm9,%ymm9
@@ -9230,7 +9252,7 @@ HIDDEN _sk_hardlight_hsw
 .globl _sk_hardlight_hsw
 FUNCTION(_sk_hardlight_hsw)
 _sk_hardlight_hsw:
-  .byte  196,98,125,24,5,116,64,0,0          // vbroadcastss  0x4074(%rip),%ymm8        # 47d0 <_sk_callback_hsw+0x17c>
+  .byte  196,98,125,24,5,115,64,0,0          // vbroadcastss  0x4073(%rip),%ymm8        # 47ec <_sk_callback_hsw+0x17c>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -9281,7 +9303,7 @@ HIDDEN _sk_overlay_hsw
 .globl _sk_overlay_hsw
 FUNCTION(_sk_overlay_hsw)
 _sk_overlay_hsw:
-  .byte  196,98,125,24,5,172,63,0,0          // vbroadcastss  0x3fac(%rip),%ymm8        # 47d4 <_sk_callback_hsw+0x180>
+  .byte  196,98,125,24,5,171,63,0,0          // vbroadcastss  0x3fab(%rip),%ymm8        # 47f0 <_sk_callback_hsw+0x180>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -9342,10 +9364,10 @@ _sk_softlight_hsw:
   .byte  196,65,20,88,197                    // vaddps        %ymm13,%ymm13,%ymm8
   .byte  196,65,60,88,192                    // vaddps        %ymm8,%ymm8,%ymm8
   .byte  196,66,61,168,192                   // vfmadd213ps   %ymm8,%ymm8,%ymm8
-  .byte  196,98,125,24,29,183,62,0,0         // vbroadcastss  0x3eb7(%rip),%ymm11        # 47dc <_sk_callback_hsw+0x188>
+  .byte  196,98,125,24,29,182,62,0,0         // vbroadcastss  0x3eb6(%rip),%ymm11        # 47f8 <_sk_callback_hsw+0x188>
   .byte  196,65,20,88,227                    // vaddps        %ymm11,%ymm13,%ymm12
   .byte  196,65,28,89,192                    // vmulps        %ymm8,%ymm12,%ymm8
-  .byte  196,98,125,24,37,168,62,0,0         // vbroadcastss  0x3ea8(%rip),%ymm12        # 47e0 <_sk_callback_hsw+0x18c>
+  .byte  196,98,125,24,37,167,62,0,0         // vbroadcastss  0x3ea7(%rip),%ymm12        # 47fc <_sk_callback_hsw+0x18c>
   .byte  196,66,21,184,196                   // vfmadd231ps   %ymm12,%ymm13,%ymm8
   .byte  196,65,124,82,245                   // vrsqrtps      %ymm13,%ymm14
   .byte  196,65,124,83,246                   // vrcpps        %ymm14,%ymm14
@@ -9355,7 +9377,7 @@ _sk_softlight_hsw:
   .byte  197,4,194,255,2                     // vcmpleps      %ymm7,%ymm15,%ymm15
   .byte  196,67,13,74,240,240                // vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   .byte  197,116,88,249                      // vaddps        %ymm1,%ymm1,%ymm15
-  .byte  196,98,125,24,5,107,62,0,0          // vbroadcastss  0x3e6b(%rip),%ymm8        # 47d8 <_sk_callback_hsw+0x184>
+  .byte  196,98,125,24,5,106,62,0,0          // vbroadcastss  0x3e6a(%rip),%ymm8        # 47f4 <_sk_callback_hsw+0x184>
   .byte  196,65,60,92,237                    // vsubps        %ymm13,%ymm8,%ymm13
   .byte  197,132,92,195                      // vsubps        %ymm3,%ymm15,%ymm0
   .byte  196,98,125,168,235                  // vfmadd213ps   %ymm3,%ymm0,%ymm13
@@ -9468,11 +9490,11 @@ _sk_hue_hsw:
   .byte  196,65,28,89,210                    // vmulps        %ymm10,%ymm12,%ymm10
   .byte  196,65,44,94,214                    // vdivps        %ymm14,%ymm10,%ymm10
   .byte  196,67,45,74,224,240                // vblendvps     %ymm15,%ymm8,%ymm10,%ymm12
-  .byte  196,98,125,24,53,111,60,0,0         // vbroadcastss  0x3c6f(%rip),%ymm14        # 47e4 <_sk_callback_hsw+0x190>
-  .byte  196,98,125,24,61,106,60,0,0         // vbroadcastss  0x3c6a(%rip),%ymm15        # 47e8 <_sk_callback_hsw+0x194>
+  .byte  196,98,125,24,53,110,60,0,0         // vbroadcastss  0x3c6e(%rip),%ymm14        # 4800 <_sk_callback_hsw+0x190>
+  .byte  196,98,125,24,61,105,60,0,0         // vbroadcastss  0x3c69(%rip),%ymm15        # 4804 <_sk_callback_hsw+0x194>
   .byte  196,65,84,89,239                    // vmulps        %ymm15,%ymm5,%ymm13
   .byte  196,66,93,184,238                   // vfmadd231ps   %ymm14,%ymm4,%ymm13
-  .byte  196,226,125,24,5,91,60,0,0          // vbroadcastss  0x3c5b(%rip),%ymm0        # 47ec <_sk_callback_hsw+0x198>
+  .byte  196,226,125,24,5,90,60,0,0          // vbroadcastss  0x3c5a(%rip),%ymm0        # 4808 <_sk_callback_hsw+0x198>
   .byte  196,98,77,184,232                   // vfmadd231ps   %ymm0,%ymm6,%ymm13
   .byte  196,65,116,89,215                   // vmulps        %ymm15,%ymm1,%ymm10
   .byte  196,66,53,184,214                   // vfmadd231ps   %ymm14,%ymm9,%ymm10
@@ -9527,7 +9549,7 @@ _sk_hue_hsw:
   .byte  196,193,124,95,192                  // vmaxps        %ymm8,%ymm0,%ymm0
   .byte  196,65,36,95,200                    // vmaxps        %ymm8,%ymm11,%ymm9
   .byte  196,65,116,95,192                   // vmaxps        %ymm8,%ymm1,%ymm8
-  .byte  196,226,125,24,13,72,59,0,0         // vbroadcastss  0x3b48(%rip),%ymm1        # 47f0 <_sk_callback_hsw+0x19c>
+  .byte  196,226,125,24,13,71,59,0,0         // vbroadcastss  0x3b47(%rip),%ymm1        # 480c <_sk_callback_hsw+0x19c>
   .byte  197,116,92,215                      // vsubps        %ymm7,%ymm1,%ymm10
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  197,116,92,219                      // vsubps        %ymm3,%ymm1,%ymm11
@@ -9581,11 +9603,11 @@ _sk_saturation_hsw:
   .byte  196,65,28,89,210                    // vmulps        %ymm10,%ymm12,%ymm10
   .byte  196,65,44,94,214                    // vdivps        %ymm14,%ymm10,%ymm10
   .byte  196,67,45,74,224,240                // vblendvps     %ymm15,%ymm8,%ymm10,%ymm12
-  .byte  196,98,125,24,53,95,58,0,0          // vbroadcastss  0x3a5f(%rip),%ymm14        # 47f4 <_sk_callback_hsw+0x1a0>
-  .byte  196,98,125,24,61,90,58,0,0          // vbroadcastss  0x3a5a(%rip),%ymm15        # 47f8 <_sk_callback_hsw+0x1a4>
+  .byte  196,98,125,24,53,94,58,0,0          // vbroadcastss  0x3a5e(%rip),%ymm14        # 4810 <_sk_callback_hsw+0x1a0>
+  .byte  196,98,125,24,61,89,58,0,0          // vbroadcastss  0x3a59(%rip),%ymm15        # 4814 <_sk_callback_hsw+0x1a4>
   .byte  196,65,84,89,239                    // vmulps        %ymm15,%ymm5,%ymm13
   .byte  196,66,93,184,238                   // vfmadd231ps   %ymm14,%ymm4,%ymm13
-  .byte  196,226,125,24,5,75,58,0,0          // vbroadcastss  0x3a4b(%rip),%ymm0        # 47fc <_sk_callback_hsw+0x1a8>
+  .byte  196,226,125,24,5,74,58,0,0          // vbroadcastss  0x3a4a(%rip),%ymm0        # 4818 <_sk_callback_hsw+0x1a8>
   .byte  196,98,77,184,232                   // vfmadd231ps   %ymm0,%ymm6,%ymm13
   .byte  196,65,116,89,215                   // vmulps        %ymm15,%ymm1,%ymm10
   .byte  196,66,53,184,214                   // vfmadd231ps   %ymm14,%ymm9,%ymm10
@@ -9640,7 +9662,7 @@ _sk_saturation_hsw:
   .byte  196,193,124,95,192                  // vmaxps        %ymm8,%ymm0,%ymm0
   .byte  196,65,36,95,200                    // vmaxps        %ymm8,%ymm11,%ymm9
   .byte  196,65,116,95,192                   // vmaxps        %ymm8,%ymm1,%ymm8
-  .byte  196,226,125,24,13,56,57,0,0         // vbroadcastss  0x3938(%rip),%ymm1        # 4800 <_sk_callback_hsw+0x1ac>
+  .byte  196,226,125,24,13,55,57,0,0         // vbroadcastss  0x3937(%rip),%ymm1        # 481c <_sk_callback_hsw+0x1ac>
   .byte  197,116,92,215                      // vsubps        %ymm7,%ymm1,%ymm10
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  197,116,92,219                      // vsubps        %ymm3,%ymm1,%ymm11
@@ -9668,11 +9690,11 @@ _sk_color_hsw:
   .byte  197,108,89,199                      // vmulps        %ymm7,%ymm2,%ymm8
   .byte  197,116,89,215                      // vmulps        %ymm7,%ymm1,%ymm10
   .byte  197,52,89,223                       // vmulps        %ymm7,%ymm9,%ymm11
-  .byte  196,98,125,24,45,209,56,0,0         // vbroadcastss  0x38d1(%rip),%ymm13        # 4804 <_sk_callback_hsw+0x1b0>
-  .byte  196,98,125,24,53,204,56,0,0         // vbroadcastss  0x38cc(%rip),%ymm14        # 4808 <_sk_callback_hsw+0x1b4>
+  .byte  196,98,125,24,45,208,56,0,0         // vbroadcastss  0x38d0(%rip),%ymm13        # 4820 <_sk_callback_hsw+0x1b0>
+  .byte  196,98,125,24,53,203,56,0,0         // vbroadcastss  0x38cb(%rip),%ymm14        # 4824 <_sk_callback_hsw+0x1b4>
   .byte  196,65,84,89,230                    // vmulps        %ymm14,%ymm5,%ymm12
   .byte  196,66,93,184,229                   // vfmadd231ps   %ymm13,%ymm4,%ymm12
-  .byte  196,98,125,24,61,189,56,0,0         // vbroadcastss  0x38bd(%rip),%ymm15        # 480c <_sk_callback_hsw+0x1b8>
+  .byte  196,98,125,24,61,188,56,0,0         // vbroadcastss  0x38bc(%rip),%ymm15        # 4828 <_sk_callback_hsw+0x1b8>
   .byte  196,66,77,184,231                   // vfmadd231ps   %ymm15,%ymm6,%ymm12
   .byte  196,65,44,89,206                    // vmulps        %ymm14,%ymm10,%ymm9
   .byte  196,66,61,184,205                   // vfmadd231ps   %ymm13,%ymm8,%ymm9
@@ -9728,7 +9750,7 @@ _sk_color_hsw:
   .byte  196,193,116,95,206                  // vmaxps        %ymm14,%ymm1,%ymm1
   .byte  196,65,44,95,198                    // vmaxps        %ymm14,%ymm10,%ymm8
   .byte  196,65,124,95,206                   // vmaxps        %ymm14,%ymm0,%ymm9
-  .byte  196,226,125,24,5,159,55,0,0         // vbroadcastss  0x379f(%rip),%ymm0        # 4810 <_sk_callback_hsw+0x1bc>
+  .byte  196,226,125,24,5,158,55,0,0         // vbroadcastss  0x379e(%rip),%ymm0        # 482c <_sk_callback_hsw+0x1bc>
   .byte  197,124,92,215                      // vsubps        %ymm7,%ymm0,%ymm10
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  197,124,92,219                      // vsubps        %ymm3,%ymm0,%ymm11
@@ -9756,11 +9778,11 @@ _sk_luminosity_hsw:
   .byte  197,100,89,196                      // vmulps        %ymm4,%ymm3,%ymm8
   .byte  197,100,89,213                      // vmulps        %ymm5,%ymm3,%ymm10
   .byte  197,100,89,222                      // vmulps        %ymm6,%ymm3,%ymm11
-  .byte  196,98,125,24,45,56,55,0,0          // vbroadcastss  0x3738(%rip),%ymm13        # 4814 <_sk_callback_hsw+0x1c0>
-  .byte  196,98,125,24,53,51,55,0,0          // vbroadcastss  0x3733(%rip),%ymm14        # 4818 <_sk_callback_hsw+0x1c4>
+  .byte  196,98,125,24,45,55,55,0,0          // vbroadcastss  0x3737(%rip),%ymm13        # 4830 <_sk_callback_hsw+0x1c0>
+  .byte  196,98,125,24,53,50,55,0,0          // vbroadcastss  0x3732(%rip),%ymm14        # 4834 <_sk_callback_hsw+0x1c4>
   .byte  196,65,116,89,230                   // vmulps        %ymm14,%ymm1,%ymm12
   .byte  196,66,109,184,229                  // vfmadd231ps   %ymm13,%ymm2,%ymm12
-  .byte  196,98,125,24,61,36,55,0,0          // vbroadcastss  0x3724(%rip),%ymm15        # 481c <_sk_callback_hsw+0x1c8>
+  .byte  196,98,125,24,61,35,55,0,0          // vbroadcastss  0x3723(%rip),%ymm15        # 4838 <_sk_callback_hsw+0x1c8>
   .byte  196,66,53,184,231                   // vfmadd231ps   %ymm15,%ymm9,%ymm12
   .byte  196,65,44,89,206                    // vmulps        %ymm14,%ymm10,%ymm9
   .byte  196,66,61,184,205                   // vfmadd231ps   %ymm13,%ymm8,%ymm9
@@ -9816,7 +9838,7 @@ _sk_luminosity_hsw:
   .byte  196,193,116,95,206                  // vmaxps        %ymm14,%ymm1,%ymm1
   .byte  196,65,44,95,198                    // vmaxps        %ymm14,%ymm10,%ymm8
   .byte  196,65,124,95,206                   // vmaxps        %ymm14,%ymm0,%ymm9
-  .byte  196,226,125,24,5,6,54,0,0           // vbroadcastss  0x3606(%rip),%ymm0        # 4820 <_sk_callback_hsw+0x1cc>
+  .byte  196,226,125,24,5,5,54,0,0           // vbroadcastss  0x3605(%rip),%ymm0        # 483c <_sk_callback_hsw+0x1cc>
   .byte  197,124,92,215                      // vsubps        %ymm7,%ymm0,%ymm10
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  197,124,92,219                      // vsubps        %ymm3,%ymm0,%ymm11
@@ -9849,7 +9871,7 @@ HIDDEN _sk_clamp_1_hsw
 .globl _sk_clamp_1_hsw
 FUNCTION(_sk_clamp_1_hsw)
 _sk_clamp_1_hsw:
-  .byte  196,98,125,24,5,162,53,0,0          // vbroadcastss  0x35a2(%rip),%ymm8        # 4824 <_sk_callback_hsw+0x1d0>
+  .byte  196,98,125,24,5,161,53,0,0          // vbroadcastss  0x35a1(%rip),%ymm8        # 4840 <_sk_callback_hsw+0x1d0>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
@@ -9861,7 +9883,7 @@ HIDDEN _sk_clamp_a_hsw
 .globl _sk_clamp_a_hsw
 FUNCTION(_sk_clamp_a_hsw)
 _sk_clamp_a_hsw:
-  .byte  196,98,125,24,5,133,53,0,0          // vbroadcastss  0x3585(%rip),%ymm8        # 4828 <_sk_callback_hsw+0x1d4>
+  .byte  196,98,125,24,5,132,53,0,0          // vbroadcastss  0x3584(%rip),%ymm8        # 4844 <_sk_callback_hsw+0x1d4>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  197,252,93,195                      // vminps        %ymm3,%ymm0,%ymm0
   .byte  197,244,93,203                      // vminps        %ymm3,%ymm1,%ymm1
@@ -9947,7 +9969,7 @@ FUNCTION(_sk_unpremul_hsw)
 _sk_unpremul_hsw:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,200,0                // vcmpeqps      %ymm8,%ymm3,%ymm9
-  .byte  196,98,125,24,21,205,52,0,0         // vbroadcastss  0x34cd(%rip),%ymm10        # 482c <_sk_callback_hsw+0x1d8>
+  .byte  196,98,125,24,21,204,52,0,0         // vbroadcastss  0x34cc(%rip),%ymm10        # 4848 <_sk_callback_hsw+0x1d8>
   .byte  197,44,94,211                       // vdivps        %ymm3,%ymm10,%ymm10
   .byte  196,67,45,74,192,144                // vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
@@ -9960,16 +9982,16 @@ HIDDEN _sk_from_srgb_hsw
 .globl _sk_from_srgb_hsw
 FUNCTION(_sk_from_srgb_hsw)
 _sk_from_srgb_hsw:
-  .byte  196,98,125,24,5,174,52,0,0          // vbroadcastss  0x34ae(%rip),%ymm8        # 4830 <_sk_callback_hsw+0x1dc>
+  .byte  196,98,125,24,5,173,52,0,0          // vbroadcastss  0x34ad(%rip),%ymm8        # 484c <_sk_callback_hsw+0x1dc>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  197,124,89,208                      // vmulps        %ymm0,%ymm0,%ymm10
-  .byte  196,98,125,24,29,160,52,0,0         // vbroadcastss  0x34a0(%rip),%ymm11        # 4834 <_sk_callback_hsw+0x1e0>
-  .byte  196,98,125,24,37,155,52,0,0         // vbroadcastss  0x349b(%rip),%ymm12        # 4838 <_sk_callback_hsw+0x1e4>
+  .byte  196,98,125,24,29,159,52,0,0         // vbroadcastss  0x349f(%rip),%ymm11        # 4850 <_sk_callback_hsw+0x1e0>
+  .byte  196,98,125,24,37,154,52,0,0         // vbroadcastss  0x349a(%rip),%ymm12        # 4854 <_sk_callback_hsw+0x1e4>
   .byte  196,65,124,40,236                   // vmovaps       %ymm12,%ymm13
   .byte  196,66,125,168,235                  // vfmadd213ps   %ymm11,%ymm0,%ymm13
-  .byte  196,98,125,24,53,140,52,0,0         // vbroadcastss  0x348c(%rip),%ymm14        # 483c <_sk_callback_hsw+0x1e8>
+  .byte  196,98,125,24,53,139,52,0,0         // vbroadcastss  0x348b(%rip),%ymm14        # 4858 <_sk_callback_hsw+0x1e8>
   .byte  196,66,45,168,238                   // vfmadd213ps   %ymm14,%ymm10,%ymm13
-  .byte  196,98,125,24,21,130,52,0,0         // vbroadcastss  0x3482(%rip),%ymm10        # 4840 <_sk_callback_hsw+0x1ec>
+  .byte  196,98,125,24,21,129,52,0,0         // vbroadcastss  0x3481(%rip),%ymm10        # 485c <_sk_callback_hsw+0x1ec>
   .byte  196,193,124,194,194,1               // vcmpltps      %ymm10,%ymm0,%ymm0
   .byte  196,195,21,74,193,0                 // vblendvps     %ymm0,%ymm9,%ymm13,%ymm0
   .byte  196,65,116,89,200                   // vmulps        %ymm8,%ymm1,%ymm9
@@ -9995,16 +10017,16 @@ _sk_to_srgb_hsw:
   .byte  197,124,82,192                      // vrsqrtps      %ymm0,%ymm8
   .byte  196,65,124,83,200                   // vrcpps        %ymm8,%ymm9
   .byte  196,65,124,82,208                   // vrsqrtps      %ymm8,%ymm10
-  .byte  196,98,125,24,5,28,52,0,0           // vbroadcastss  0x341c(%rip),%ymm8        # 4844 <_sk_callback_hsw+0x1f0>
+  .byte  196,98,125,24,5,27,52,0,0           // vbroadcastss  0x341b(%rip),%ymm8        # 4860 <_sk_callback_hsw+0x1f0>
   .byte  196,65,124,89,216                   // vmulps        %ymm8,%ymm0,%ymm11
-  .byte  196,98,125,24,37,18,52,0,0          // vbroadcastss  0x3412(%rip),%ymm12        # 4848 <_sk_callback_hsw+0x1f4>
-  .byte  196,98,125,24,45,13,52,0,0          // vbroadcastss  0x340d(%rip),%ymm13        # 484c <_sk_callback_hsw+0x1f8>
+  .byte  196,98,125,24,37,17,52,0,0          // vbroadcastss  0x3411(%rip),%ymm12        # 4864 <_sk_callback_hsw+0x1f4>
+  .byte  196,98,125,24,45,12,52,0,0          // vbroadcastss  0x340c(%rip),%ymm13        # 4868 <_sk_callback_hsw+0x1f8>
   .byte  196,66,21,168,204                   // vfmadd213ps   %ymm12,%ymm13,%ymm9
-  .byte  196,98,125,24,53,3,52,0,0           // vbroadcastss  0x3403(%rip),%ymm14        # 4850 <_sk_callback_hsw+0x1fc>
+  .byte  196,98,125,24,53,2,52,0,0           // vbroadcastss  0x3402(%rip),%ymm14        # 486c <_sk_callback_hsw+0x1fc>
   .byte  196,66,13,184,202                   // vfmadd231ps   %ymm10,%ymm14,%ymm9
-  .byte  196,98,125,24,21,249,51,0,0         // vbroadcastss  0x33f9(%rip),%ymm10        # 4854 <_sk_callback_hsw+0x200>
+  .byte  196,98,125,24,21,248,51,0,0         // vbroadcastss  0x33f8(%rip),%ymm10        # 4870 <_sk_callback_hsw+0x200>
   .byte  196,65,44,93,201                    // vminps        %ymm9,%ymm10,%ymm9
-  .byte  196,98,125,24,61,239,51,0,0         // vbroadcastss  0x33ef(%rip),%ymm15        # 4858 <_sk_callback_hsw+0x204>
+  .byte  196,98,125,24,61,238,51,0,0         // vbroadcastss  0x33ee(%rip),%ymm15        # 4874 <_sk_callback_hsw+0x204>
   .byte  196,193,124,194,199,1               // vcmpltps      %ymm15,%ymm0,%ymm0
   .byte  196,195,53,74,195,0                 // vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   .byte  197,124,82,201                      // vrsqrtps      %ymm1,%ymm9
@@ -10037,26 +10059,26 @@ _sk_rgb_to_hsl_hsw:
   .byte  197,124,93,201                      // vminps        %ymm1,%ymm0,%ymm9
   .byte  197,52,93,202                       // vminps        %ymm2,%ymm9,%ymm9
   .byte  196,65,60,92,209                    // vsubps        %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,29,105,51,0,0         // vbroadcastss  0x3369(%rip),%ymm11        # 485c <_sk_callback_hsw+0x208>
+  .byte  196,98,125,24,29,104,51,0,0         // vbroadcastss  0x3368(%rip),%ymm11        # 4878 <_sk_callback_hsw+0x208>
   .byte  196,65,36,94,218                    // vdivps        %ymm10,%ymm11,%ymm11
   .byte  197,116,92,226                      // vsubps        %ymm2,%ymm1,%ymm12
   .byte  197,116,194,234,1                   // vcmpltps      %ymm2,%ymm1,%ymm13
-  .byte  196,98,125,24,53,86,51,0,0          // vbroadcastss  0x3356(%rip),%ymm14        # 4860 <_sk_callback_hsw+0x20c>
+  .byte  196,98,125,24,53,85,51,0,0          // vbroadcastss  0x3355(%rip),%ymm14        # 487c <_sk_callback_hsw+0x20c>
   .byte  196,65,4,87,255                     // vxorps        %ymm15,%ymm15,%ymm15
   .byte  196,67,5,74,238,208                 // vblendvps     %ymm13,%ymm14,%ymm15,%ymm13
   .byte  196,66,37,168,229                   // vfmadd213ps   %ymm13,%ymm11,%ymm12
   .byte  197,236,92,208                      // vsubps        %ymm0,%ymm2,%ymm2
   .byte  197,124,92,233                      // vsubps        %ymm1,%ymm0,%ymm13
-  .byte  196,98,125,24,53,61,51,0,0          // vbroadcastss  0x333d(%rip),%ymm14        # 4868 <_sk_callback_hsw+0x214>
+  .byte  196,98,125,24,53,60,51,0,0          // vbroadcastss  0x333c(%rip),%ymm14        # 4884 <_sk_callback_hsw+0x214>
   .byte  196,66,37,168,238                   // vfmadd213ps   %ymm14,%ymm11,%ymm13
-  .byte  196,98,125,24,53,43,51,0,0          // vbroadcastss  0x332b(%rip),%ymm14        # 4864 <_sk_callback_hsw+0x210>
+  .byte  196,98,125,24,53,42,51,0,0          // vbroadcastss  0x332a(%rip),%ymm14        # 4880 <_sk_callback_hsw+0x210>
   .byte  196,194,37,168,214                  // vfmadd213ps   %ymm14,%ymm11,%ymm2
   .byte  197,188,194,201,0                   // vcmpeqps      %ymm1,%ymm8,%ymm1
   .byte  196,227,21,74,202,16                // vblendvps     %ymm1,%ymm2,%ymm13,%ymm1
   .byte  197,188,194,192,0                   // vcmpeqps      %ymm0,%ymm8,%ymm0
   .byte  196,195,117,74,196,0                // vblendvps     %ymm0,%ymm12,%ymm1,%ymm0
   .byte  196,193,60,88,201                   // vaddps        %ymm9,%ymm8,%ymm1
-  .byte  196,98,125,24,29,14,51,0,0          // vbroadcastss  0x330e(%rip),%ymm11        # 4870 <_sk_callback_hsw+0x21c>
+  .byte  196,98,125,24,29,13,51,0,0          // vbroadcastss  0x330d(%rip),%ymm11        # 488c <_sk_callback_hsw+0x21c>
   .byte  196,193,116,89,211                  // vmulps        %ymm11,%ymm1,%ymm2
   .byte  197,36,194,218,1                    // vcmpltps      %ymm2,%ymm11,%ymm11
   .byte  196,65,12,92,224                    // vsubps        %ymm8,%ymm14,%ymm12
@@ -10066,7 +10088,7 @@ _sk_rgb_to_hsl_hsw:
   .byte  197,172,94,201                      // vdivps        %ymm1,%ymm10,%ymm1
   .byte  196,195,125,74,199,128              // vblendvps     %ymm8,%ymm15,%ymm0,%ymm0
   .byte  196,195,117,74,207,128              // vblendvps     %ymm8,%ymm15,%ymm1,%ymm1
-  .byte  196,98,125,24,5,209,50,0,0          // vbroadcastss  0x32d1(%rip),%ymm8        # 486c <_sk_callback_hsw+0x218>
+  .byte  196,98,125,24,5,208,50,0,0          // vbroadcastss  0x32d0(%rip),%ymm8        # 4888 <_sk_callback_hsw+0x218>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10083,30 +10105,30 @@ _sk_hsl_to_rgb_hsw:
   .byte  197,252,17,92,36,128                // vmovups       %ymm3,-0x80(%rsp)
   .byte  197,252,40,233                      // vmovaps       %ymm1,%ymm5
   .byte  197,252,40,224                      // vmovaps       %ymm0,%ymm4
-  .byte  196,98,125,24,5,158,50,0,0          // vbroadcastss  0x329e(%rip),%ymm8        # 4874 <_sk_callback_hsw+0x220>
+  .byte  196,98,125,24,5,157,50,0,0          // vbroadcastss  0x329d(%rip),%ymm8        # 4890 <_sk_callback_hsw+0x220>
   .byte  197,60,194,202,2                    // vcmpleps      %ymm2,%ymm8,%ymm9
   .byte  197,84,89,210                       // vmulps        %ymm2,%ymm5,%ymm10
   .byte  196,65,84,92,218                    // vsubps        %ymm10,%ymm5,%ymm11
   .byte  196,67,45,74,203,144                // vblendvps     %ymm9,%ymm11,%ymm10,%ymm9
   .byte  197,52,88,210                       // vaddps        %ymm2,%ymm9,%ymm10
-  .byte  196,98,125,24,13,129,50,0,0         // vbroadcastss  0x3281(%rip),%ymm9        # 4878 <_sk_callback_hsw+0x224>
+  .byte  196,98,125,24,13,128,50,0,0         // vbroadcastss  0x3280(%rip),%ymm9        # 4894 <_sk_callback_hsw+0x224>
   .byte  196,66,109,170,202                  // vfmsub213ps   %ymm10,%ymm2,%ymm9
-  .byte  196,98,125,24,29,119,50,0,0         // vbroadcastss  0x3277(%rip),%ymm11        # 487c <_sk_callback_hsw+0x228>
+  .byte  196,98,125,24,29,118,50,0,0         // vbroadcastss  0x3276(%rip),%ymm11        # 4898 <_sk_callback_hsw+0x228>
   .byte  196,65,92,88,219                    // vaddps        %ymm11,%ymm4,%ymm11
   .byte  196,67,125,8,227,1                  // vroundps      $0x1,%ymm11,%ymm12
   .byte  196,65,36,92,252                    // vsubps        %ymm12,%ymm11,%ymm15
   .byte  196,65,44,92,217                    // vsubps        %ymm9,%ymm10,%ymm11
-  .byte  196,98,125,24,45,97,50,0,0          // vbroadcastss  0x3261(%rip),%ymm13        # 4884 <_sk_callback_hsw+0x230>
+  .byte  196,98,125,24,45,96,50,0,0          // vbroadcastss  0x3260(%rip),%ymm13        # 48a0 <_sk_callback_hsw+0x230>
   .byte  196,193,4,89,197                    // vmulps        %ymm13,%ymm15,%ymm0
-  .byte  196,98,125,24,53,87,50,0,0          // vbroadcastss  0x3257(%rip),%ymm14        # 4888 <_sk_callback_hsw+0x234>
+  .byte  196,98,125,24,53,86,50,0,0          // vbroadcastss  0x3256(%rip),%ymm14        # 48a4 <_sk_callback_hsw+0x234>
   .byte  197,12,92,224                       // vsubps        %ymm0,%ymm14,%ymm12
   .byte  196,66,37,168,225                   // vfmadd213ps   %ymm9,%ymm11,%ymm12
-  .byte  196,226,125,24,29,61,50,0,0         // vbroadcastss  0x323d(%rip),%ymm3        # 4880 <_sk_callback_hsw+0x22c>
+  .byte  196,226,125,24,29,60,50,0,0         // vbroadcastss  0x323c(%rip),%ymm3        # 489c <_sk_callback_hsw+0x22c>
   .byte  196,193,100,194,255,2               // vcmpleps      %ymm15,%ymm3,%ymm7
   .byte  196,195,29,74,249,112               // vblendvps     %ymm7,%ymm9,%ymm12,%ymm7
   .byte  196,65,60,194,231,2                 // vcmpleps      %ymm15,%ymm8,%ymm12
   .byte  196,227,45,74,255,192               // vblendvps     %ymm12,%ymm7,%ymm10,%ymm7
-  .byte  196,98,125,24,37,40,50,0,0          // vbroadcastss  0x3228(%rip),%ymm12        # 488c <_sk_callback_hsw+0x238>
+  .byte  196,98,125,24,37,39,50,0,0          // vbroadcastss  0x3227(%rip),%ymm12        # 48a8 <_sk_callback_hsw+0x238>
   .byte  196,65,28,194,255,2                 // vcmpleps      %ymm15,%ymm12,%ymm15
   .byte  196,194,37,168,193                  // vfmadd213ps   %ymm9,%ymm11,%ymm0
   .byte  196,99,125,74,255,240               // vblendvps     %ymm15,%ymm7,%ymm0,%ymm15
@@ -10122,7 +10144,7 @@ _sk_hsl_to_rgb_hsw:
   .byte  197,156,194,192,2                   // vcmpleps      %ymm0,%ymm12,%ymm0
   .byte  196,194,37,168,249                  // vfmadd213ps   %ymm9,%ymm11,%ymm7
   .byte  196,227,69,74,201,0                 // vblendvps     %ymm0,%ymm1,%ymm7,%ymm1
-  .byte  196,226,125,24,5,212,49,0,0         // vbroadcastss  0x31d4(%rip),%ymm0        # 4890 <_sk_callback_hsw+0x23c>
+  .byte  196,226,125,24,5,211,49,0,0         // vbroadcastss  0x31d3(%rip),%ymm0        # 48ac <_sk_callback_hsw+0x23c>
   .byte  197,220,88,192                      // vaddps        %ymm0,%ymm4,%ymm0
   .byte  196,227,125,8,224,1                 // vroundps      $0x1,%ymm0,%ymm4
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
@@ -10172,11 +10194,11 @@ _sk_scale_u8_hsw:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,51                              // jne           179c <_sk_scale_u8_hsw+0x43>
+  .byte  117,51                              // jne           17b9 <_sk_scale_u8_hsw+0x43>
   .byte  197,122,126,0                       // vmovq         (%rax),%xmm8
   .byte  196,66,125,49,192                   // vpmovzxbd     %xmm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,20,49,0,0          // vbroadcastss  0x3114(%rip),%ymm9        # 4894 <_sk_callback_hsw+0x240>
+  .byte  196,98,125,24,13,19,49,0,0          // vbroadcastss  0x3113(%rip),%ymm9        # 48b0 <_sk_callback_hsw+0x240>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -10194,9 +10216,9 @@ _sk_scale_u8_hsw:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           17a4 <_sk_scale_u8_hsw+0x4b>
+  .byte  117,234                             // jne           17c1 <_sk_scale_u8_hsw+0x4b>
   .byte  196,65,249,110,193                  // vmovq         %r9,%xmm8
-  .byte  235,172                             // jmp           176d <_sk_scale_u8_hsw+0x14>
+  .byte  235,172                             // jmp           178a <_sk_scale_u8_hsw+0x14>
 
 HIDDEN _sk_lerp_1_float_hsw
 .globl _sk_lerp_1_float_hsw
@@ -10224,11 +10246,11 @@ _sk_lerp_u8_hsw:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,71                              // jne           1847 <_sk_lerp_u8_hsw+0x57>
+  .byte  117,71                              // jne           1864 <_sk_lerp_u8_hsw+0x57>
   .byte  197,122,126,0                       // vmovq         (%rax),%xmm8
   .byte  196,66,125,49,192                   // vpmovzxbd     %xmm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,129,48,0,0         // vbroadcastss  0x3081(%rip),%ymm9        # 4898 <_sk_callback_hsw+0x244>
+  .byte  196,98,125,24,13,128,48,0,0         // vbroadcastss  0x3080(%rip),%ymm9        # 48b4 <_sk_callback_hsw+0x244>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
   .byte  196,226,61,168,196                  // vfmadd213ps   %ymm4,%ymm8,%ymm0
@@ -10250,9 +10272,9 @@ _sk_lerp_u8_hsw:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           184f <_sk_lerp_u8_hsw+0x5f>
+  .byte  117,234                             // jne           186c <_sk_lerp_u8_hsw+0x5f>
   .byte  196,65,249,110,193                  // vmovq         %r9,%xmm8
-  .byte  235,152                             // jmp           1804 <_sk_lerp_u8_hsw+0x14>
+  .byte  235,152                             // jmp           1821 <_sk_lerp_u8_hsw+0x14>
 
 HIDDEN _sk_lerp_565_hsw
 .globl _sk_lerp_565_hsw
@@ -10261,23 +10283,23 @@ _sk_lerp_565_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,169,0,0,0                    // jne           1923 <_sk_lerp_565_hsw+0xb7>
+  .byte  15,133,169,0,0,0                    // jne           1940 <_sk_lerp_565_hsw+0xb7>
   .byte  196,65,122,111,4,122                // vmovdqu       (%r10,%rdi,2),%xmm8
   .byte  196,66,125,51,192                   // vpmovzxwd     %xmm8,%ymm8
-  .byte  196,98,125,88,13,14,48,0,0          // vpbroadcastd  0x300e(%rip),%ymm9        # 489c <_sk_callback_hsw+0x248>
+  .byte  196,98,125,88,13,13,48,0,0          // vpbroadcastd  0x300d(%rip),%ymm9        # 48b8 <_sk_callback_hsw+0x248>
   .byte  196,65,61,219,201                   // vpand         %ymm9,%ymm8,%ymm9
   .byte  196,65,124,91,201                   // vcvtdq2ps     %ymm9,%ymm9
-  .byte  196,98,125,24,21,255,47,0,0         // vbroadcastss  0x2fff(%rip),%ymm10        # 48a0 <_sk_callback_hsw+0x24c>
+  .byte  196,98,125,24,21,254,47,0,0         // vbroadcastss  0x2ffe(%rip),%ymm10        # 48bc <_sk_callback_hsw+0x24c>
   .byte  196,65,52,89,202                    // vmulps        %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,88,21,245,47,0,0         // vpbroadcastd  0x2ff5(%rip),%ymm10        # 48a4 <_sk_callback_hsw+0x250>
+  .byte  196,98,125,88,21,244,47,0,0         // vpbroadcastd  0x2ff4(%rip),%ymm10        # 48c0 <_sk_callback_hsw+0x250>
   .byte  196,65,61,219,210                   // vpand         %ymm10,%ymm8,%ymm10
   .byte  196,65,124,91,210                   // vcvtdq2ps     %ymm10,%ymm10
-  .byte  196,98,125,24,29,230,47,0,0         // vbroadcastss  0x2fe6(%rip),%ymm11        # 48a8 <_sk_callback_hsw+0x254>
+  .byte  196,98,125,24,29,229,47,0,0         // vbroadcastss  0x2fe5(%rip),%ymm11        # 48c4 <_sk_callback_hsw+0x254>
   .byte  196,65,44,89,211                    // vmulps        %ymm11,%ymm10,%ymm10
-  .byte  196,98,125,88,29,220,47,0,0         // vpbroadcastd  0x2fdc(%rip),%ymm11        # 48ac <_sk_callback_hsw+0x258>
+  .byte  196,98,125,88,29,219,47,0,0         // vpbroadcastd  0x2fdb(%rip),%ymm11        # 48c8 <_sk_callback_hsw+0x258>
   .byte  196,65,61,219,195                   // vpand         %ymm11,%ymm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,29,205,47,0,0         // vbroadcastss  0x2fcd(%rip),%ymm11        # 48b0 <_sk_callback_hsw+0x25c>
+  .byte  196,98,125,24,29,204,47,0,0         // vbroadcastss  0x2fcc(%rip),%ymm11        # 48cc <_sk_callback_hsw+0x25c>
   .byte  196,65,60,89,195                    // vmulps        %ymm11,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
   .byte  196,226,53,168,196                  // vfmadd213ps   %ymm4,%ymm9,%ymm0
@@ -10298,9 +10320,9 @@ _sk_lerp_565_hsw:
   .byte  196,65,57,239,192                   // vpxor         %xmm8,%xmm8,%xmm8
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,68,255,255,255               // ja            1880 <_sk_lerp_565_hsw+0x14>
+  .byte  15,135,68,255,255,255               // ja            189d <_sk_lerp_565_hsw+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,77,0,0,0                  // lea           0x4d(%rip),%r9        # 1994 <_sk_lerp_565_hsw+0x128>
+  .byte  76,141,13,76,0,0,0                  // lea           0x4c(%rip),%r9        # 19b0 <_sk_lerp_565_hsw+0x127>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10312,26 +10334,28 @@ _sk_lerp_565_hsw:
   .byte  196,65,57,196,68,122,4,2            // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,2,1            // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,4,122,0               // vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
-  .byte  233,239,254,255,255                 // jmpq          1880 <_sk_lerp_565_hsw+0x14>
-  .byte  15,31,0                             // nopl          (%rax)
-  .byte  241                                 // icebp
+  .byte  233,239,254,255,255                 // jmpq          189d <_sk_lerp_565_hsw+0x14>
+  .byte  102,144                             // xchg          %ax,%ax
+  .byte  242,255                             // repnz         (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
+  .byte  234                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  233,255,255,255,225                 // jmpq          ffffffffe200199c <_sk_callback_hsw+0xffffffffe1ffd348>
   .byte  255                                 // (bad)
+  .byte  255,226                             // jmpq          *%rdx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  217,255                             // fcos
   .byte  255                                 // (bad)
-  .byte  255,209                             // callq         *%rcx
+  .byte  218,255                             // (bad)
   .byte  255                                 // (bad)
+  .byte  255,210                             // callq         *%rdx
   .byte  255                                 // (bad)
-  .byte  255,201                             // dec           %ecx
+  .byte  255                                 // (bad)
+  .byte  255,202                             // dec           %edx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  188                                 // .byte         0xbc
+  .byte  189                                 // .byte         0xbd
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
@@ -10345,23 +10369,23 @@ _sk_load_tables_hsw:
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  76,3,8                              // add           (%rax),%r9
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,105                             // jne           1a2e <_sk_load_tables_hsw+0x7e>
+  .byte  117,105                             // jne           1a4a <_sk_load_tables_hsw+0x7e>
   .byte  196,193,126,111,25                  // vmovdqu       (%r9),%ymm3
-  .byte  197,229,219,13,142,49,0,0           // vpand         0x318e(%rip),%ymm3,%ymm1        # 4b60 <_sk_callback_hsw+0x50c>
+  .byte  197,229,219,13,146,49,0,0           // vpand         0x3192(%rip),%ymm3,%ymm1        # 4b80 <_sk_callback_hsw+0x510>
   .byte  196,65,61,118,192                   // vpcmpeqd      %ymm8,%ymm8,%ymm8
   .byte  72,139,72,8                         // mov           0x8(%rax),%rcx
   .byte  76,139,72,16                        // mov           0x10(%rax),%r9
   .byte  197,237,118,210                     // vpcmpeqd      %ymm2,%ymm2,%ymm2
   .byte  196,226,109,146,4,137               // vgatherdps    %ymm2,(%rcx,%ymm1,4),%ymm0
-  .byte  196,226,101,0,21,142,49,0,0         // vpshufb       0x318e(%rip),%ymm3,%ymm2        # 4b80 <_sk_callback_hsw+0x52c>
+  .byte  196,226,101,0,21,146,49,0,0         // vpshufb       0x3192(%rip),%ymm3,%ymm2        # 4ba0 <_sk_callback_hsw+0x530>
   .byte  196,65,53,118,201                   // vpcmpeqd      %ymm9,%ymm9,%ymm9
   .byte  196,194,53,146,12,145               // vgatherdps    %ymm9,(%r9,%ymm2,4),%ymm1
   .byte  72,139,64,24                        // mov           0x18(%rax),%rax
-  .byte  196,98,101,0,13,150,49,0,0          // vpshufb       0x3196(%rip),%ymm3,%ymm9        # 4ba0 <_sk_callback_hsw+0x54c>
+  .byte  196,98,101,0,13,154,49,0,0          // vpshufb       0x319a(%rip),%ymm3,%ymm9        # 4bc0 <_sk_callback_hsw+0x550>
   .byte  196,162,61,146,20,136               // vgatherdps    %ymm8,(%rax,%ymm9,4),%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,146,46,0,0          // vbroadcastss  0x2e92(%rip),%ymm8        # 48b4 <_sk_callback_hsw+0x260>
+  .byte  196,98,125,24,5,146,46,0,0          // vbroadcastss  0x2e92(%rip),%ymm8        # 48d0 <_sk_callback_hsw+0x260>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,137,193                          // mov           %r8,%rcx
@@ -10374,7 +10398,7 @@ _sk_load_tables_hsw:
   .byte  196,193,249,110,194                 // vmovq         %r10,%xmm0
   .byte  196,226,125,33,192                  // vpmovsxbd     %xmm0,%ymm0
   .byte  196,194,125,140,25                  // vpmaskmovd    (%r9),%ymm0,%ymm3
-  .byte  233,115,255,255,255                 // jmpq          19ca <_sk_load_tables_hsw+0x1a>
+  .byte  233,115,255,255,255                 // jmpq          19e6 <_sk_load_tables_hsw+0x1a>
 
 HIDDEN _sk_load_tables_u16_be_hsw
 .globl _sk_load_tables_u16_be_hsw
@@ -10384,7 +10408,7 @@ _sk_load_tables_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,201,0,0,0                    // jne           1b36 <_sk_load_tables_u16_be_hsw+0xdf>
+  .byte  15,133,201,0,0,0                    // jne           1b52 <_sk_load_tables_u16_be_hsw+0xdf>
   .byte  196,1,121,16,4,72                   // vmovupd       (%r8,%r9,2),%xmm8
   .byte  196,129,121,16,84,72,16             // vmovupd       0x10(%r8,%r9,2),%xmm2
   .byte  196,129,121,16,92,72,32             // vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -10400,7 +10424,7 @@ _sk_load_tables_u16_be_hsw:
   .byte  197,185,108,200                     // vpunpcklqdq   %xmm0,%xmm8,%xmm1
   .byte  197,185,109,208                     // vpunpckhqdq   %xmm0,%xmm8,%xmm2
   .byte  197,49,108,195                      // vpunpcklqdq   %xmm3,%xmm9,%xmm8
-  .byte  197,121,111,21,34,50,0,0            // vmovdqa       0x3222(%rip),%xmm10        # 4ce0 <_sk_callback_hsw+0x68c>
+  .byte  197,121,111,21,38,50,0,0            // vmovdqa       0x3226(%rip),%xmm10        # 4d00 <_sk_callback_hsw+0x690>
   .byte  196,193,113,219,194                 // vpand         %xmm10,%xmm1,%xmm0
   .byte  196,226,125,51,200                  // vpmovzxwd     %xmm0,%ymm1
   .byte  196,65,37,118,219                   // vpcmpeqd      %ymm11,%ymm11,%ymm11
@@ -10422,36 +10446,36 @@ _sk_load_tables_u16_be_hsw:
   .byte  197,185,235,219                     // vpor          %xmm3,%xmm8,%xmm3
   .byte  196,226,125,51,219                  // vpmovzxwd     %xmm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,139,45,0,0          // vbroadcastss  0x2d8b(%rip),%ymm8        # 48b8 <_sk_callback_hsw+0x264>
+  .byte  196,98,125,24,5,139,45,0,0          // vbroadcastss  0x2d8b(%rip),%ymm8        # 48d4 <_sk_callback_hsw+0x264>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
   .byte  196,1,123,16,4,72                   // vmovsd        (%r8,%r9,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            1b9c <_sk_load_tables_u16_be_hsw+0x145>
+  .byte  116,85                              // je            1bb8 <_sk_load_tables_u16_be_hsw+0x145>
   .byte  196,1,57,22,68,72,8                 // vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            1b9c <_sk_load_tables_u16_be_hsw+0x145>
+  .byte  114,72                              // jb            1bb8 <_sk_load_tables_u16_be_hsw+0x145>
   .byte  196,129,123,16,84,72,16             // vmovsd        0x10(%r8,%r9,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            1ba9 <_sk_load_tables_u16_be_hsw+0x152>
+  .byte  116,72                              // je            1bc5 <_sk_load_tables_u16_be_hsw+0x152>
   .byte  196,129,105,22,84,72,24             // vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            1ba9 <_sk_load_tables_u16_be_hsw+0x152>
+  .byte  114,59                              // jb            1bc5 <_sk_load_tables_u16_be_hsw+0x152>
   .byte  196,129,123,16,92,72,32             // vmovsd        0x20(%r8,%r9,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,9,255,255,255                // je            1a88 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  15,132,9,255,255,255                // je            1aa4 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  196,129,97,22,92,72,40              // vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,248,254,255,255              // jb            1a88 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  15,130,248,254,255,255              // jb            1aa4 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  196,1,122,126,76,72,48              // vmovq         0x30(%r8,%r9,2),%xmm9
-  .byte  233,236,254,255,255                 // jmpq          1a88 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  233,236,254,255,255                 // jmpq          1aa4 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,223,254,255,255                 // jmpq          1a88 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  233,223,254,255,255                 // jmpq          1aa4 <_sk_load_tables_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,214,254,255,255                 // jmpq          1a88 <_sk_load_tables_u16_be_hsw+0x31>
+  .byte  233,214,254,255,255                 // jmpq          1aa4 <_sk_load_tables_u16_be_hsw+0x31>
 
 HIDDEN _sk_load_tables_rgb_u16_be_hsw
 .globl _sk_load_tables_rgb_u16_be_hsw
@@ -10461,7 +10485,7 @@ _sk_load_tables_rgb_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,127                       // lea           (%rdi,%rdi,2),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,193,0,0,0                    // jne           1c85 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
+  .byte  15,133,193,0,0,0                    // jne           1ca1 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
   .byte  196,129,122,111,4,72                // vmovdqu       (%r8,%r9,2),%xmm0
   .byte  196,129,122,111,84,72,12            // vmovdqu       0xc(%r8,%r9,2),%xmm2
   .byte  196,129,122,111,76,72,24            // vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -10482,7 +10506,7 @@ _sk_load_tables_rgb_u16_be_hsw:
   .byte  197,185,108,218                     // vpunpcklqdq   %xmm2,%xmm8,%xmm3
   .byte  197,185,109,210                     // vpunpckhqdq   %xmm2,%xmm8,%xmm2
   .byte  197,121,108,193                     // vpunpcklqdq   %xmm1,%xmm0,%xmm8
-  .byte  197,121,111,13,194,48,0,0           // vmovdqa       0x30c2(%rip),%xmm9        # 4cf0 <_sk_callback_hsw+0x69c>
+  .byte  197,121,111,13,198,48,0,0           // vmovdqa       0x30c6(%rip),%xmm9        # 4d10 <_sk_callback_hsw+0x6a0>
   .byte  196,193,97,219,193                  // vpand         %xmm9,%xmm3,%xmm0
   .byte  196,226,125,51,200                  // vpmovzxwd     %xmm0,%ymm1
   .byte  197,229,118,219                     // vpcmpeqd      %ymm3,%ymm3,%ymm3
@@ -10499,41 +10523,41 @@ _sk_load_tables_rgb_u16_be_hsw:
   .byte  196,98,125,51,194                   // vpmovzxwd     %xmm2,%ymm8
   .byte  196,162,101,146,20,128              // vgatherdps    %ymm3,(%rax,%ymm8,4),%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,57,44,0,0         // vbroadcastss  0x2c39(%rip),%ymm3        # 48bc <_sk_callback_hsw+0x268>
+  .byte  196,226,125,24,29,57,44,0,0         // vbroadcastss  0x2c39(%rip),%ymm3        # 48d8 <_sk_callback_hsw+0x268>
   .byte  255,224                             // jmpq          *%rax
   .byte  196,129,121,110,4,72                // vmovd         (%r8,%r9,2),%xmm0
   .byte  196,129,121,196,68,72,4,2           // vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           1c9e <_sk_load_tables_rgb_u16_be_hsw+0xec>
-  .byte  233,90,255,255,255                  // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  117,5                               // jne           1cba <_sk_load_tables_rgb_u16_be_hsw+0xec>
+  .byte  233,90,255,255,255                  // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   .byte  196,129,121,110,76,72,6             // vmovd         0x6(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,68,72,10,2            // vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            1ccd <_sk_load_tables_rgb_u16_be_hsw+0x11b>
+  .byte  114,26                              // jb            1ce9 <_sk_load_tables_rgb_u16_be_hsw+0x11b>
   .byte  196,129,121,110,76,72,12            // vmovd         0xc(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,84,72,16,2          // vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           1cd2 <_sk_load_tables_rgb_u16_be_hsw+0x120>
-  .byte  233,43,255,255,255                  // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  .byte  233,38,255,255,255                  // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           1cee <_sk_load_tables_rgb_u16_be_hsw+0x120>
+  .byte  233,43,255,255,255                  // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,38,255,255,255                  // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   .byte  196,129,121,110,76,72,18            // vmovd         0x12(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,76,72,22,2            // vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            1d01 <_sk_load_tables_rgb_u16_be_hsw+0x14f>
+  .byte  114,26                              // jb            1d1d <_sk_load_tables_rgb_u16_be_hsw+0x14f>
   .byte  196,129,121,110,76,72,24            // vmovd         0x18(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,76,72,28,2          // vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           1d06 <_sk_load_tables_rgb_u16_be_hsw+0x154>
-  .byte  233,247,254,255,255                 // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  .byte  233,242,254,255,255                 // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           1d22 <_sk_load_tables_rgb_u16_be_hsw+0x154>
+  .byte  233,247,254,255,255                 // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,242,254,255,255                 // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   .byte  196,129,121,110,92,72,30            // vmovd         0x1e(%r8,%r9,2),%xmm3
   .byte  196,1,97,196,92,72,34,2             // vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            1d2f <_sk_load_tables_rgb_u16_be_hsw+0x17d>
+  .byte  114,20                              // jb            1d4b <_sk_load_tables_rgb_u16_be_hsw+0x17d>
   .byte  196,129,121,110,92,72,36            // vmovd         0x24(%r8,%r9,2),%xmm3
   .byte  196,129,97,196,92,72,40,2           // vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  .byte  233,201,254,255,255                 // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  .byte  233,196,254,255,255                 // jmpq          1bf8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,201,254,255,255                 // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  .byte  233,196,254,255,255                 // jmpq          1c14 <_sk_load_tables_rgb_u16_be_hsw+0x46>
 
 HIDDEN _sk_byte_tables_hsw
 .globl _sk_byte_tables_hsw
@@ -10546,7 +10570,7 @@ _sk_byte_tables_hsw:
   .byte  65,84                               // push          %r12
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,119,43,0,0          // vbroadcastss  0x2b77(%rip),%ymm8        # 48c0 <_sk_callback_hsw+0x26c>
+  .byte  196,98,125,24,5,119,43,0,0          // vbroadcastss  0x2b77(%rip),%ymm8        # 48dc <_sk_callback_hsw+0x26c>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,195,249,22,192,1                // vpextrq       $0x1,%xmm0,%r8
@@ -10583,7 +10607,7 @@ _sk_byte_tables_hsw:
   .byte  196,227,121,32,197,7                // vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,200,42,0,0         // vbroadcastss  0x2ac8(%rip),%ymm9        # 48c4 <_sk_callback_hsw+0x270>
+  .byte  196,98,125,24,13,200,42,0,0         // vbroadcastss  0x2ac8(%rip),%ymm9        # 48e0 <_sk_callback_hsw+0x270>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -10744,7 +10768,7 @@ _sk_byte_tables_rgb_hsw:
   .byte  196,227,121,32,197,7                // vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,1,40,0,0           // vbroadcastss  0x2801(%rip),%ymm9        # 48c8 <_sk_callback_hsw+0x274>
+  .byte  196,98,125,24,13,1,40,0,0           // vbroadcastss  0x2801(%rip),%ymm9        # 48e4 <_sk_callback_hsw+0x274>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -10907,33 +10931,33 @@ _sk_parametric_r_hsw:
   .byte  196,66,125,168,211                  // vfmadd213ps   %ymm11,%ymm0,%ymm10
   .byte  196,226,125,24,0                    // vbroadcastss  (%rax),%ymm0
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,180,37,0,0         // vbroadcastss  0x25b4(%rip),%ymm12        # 48cc <_sk_callback_hsw+0x278>
-  .byte  196,98,125,24,45,175,37,0,0         // vbroadcastss  0x25af(%rip),%ymm13        # 48d0 <_sk_callback_hsw+0x27c>
+  .byte  196,98,125,24,37,180,37,0,0         // vbroadcastss  0x25b4(%rip),%ymm12        # 48e8 <_sk_callback_hsw+0x278>
+  .byte  196,98,125,24,45,175,37,0,0         // vbroadcastss  0x25af(%rip),%ymm13        # 48ec <_sk_callback_hsw+0x27c>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,165,37,0,0         // vbroadcastss  0x25a5(%rip),%ymm13        # 48d4 <_sk_callback_hsw+0x280>
+  .byte  196,98,125,24,45,165,37,0,0         // vbroadcastss  0x25a5(%rip),%ymm13        # 48f0 <_sk_callback_hsw+0x280>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,155,37,0,0         // vbroadcastss  0x259b(%rip),%ymm13        # 48d8 <_sk_callback_hsw+0x284>
+  .byte  196,98,125,24,45,155,37,0,0         // vbroadcastss  0x259b(%rip),%ymm13        # 48f4 <_sk_callback_hsw+0x284>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,145,37,0,0         // vbroadcastss  0x2591(%rip),%ymm11        # 48dc <_sk_callback_hsw+0x288>
+  .byte  196,98,125,24,29,145,37,0,0         // vbroadcastss  0x2591(%rip),%ymm11        # 48f8 <_sk_callback_hsw+0x288>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,135,37,0,0         // vbroadcastss  0x2587(%rip),%ymm12        # 48e0 <_sk_callback_hsw+0x28c>
+  .byte  196,98,125,24,37,135,37,0,0         // vbroadcastss  0x2587(%rip),%ymm12        # 48fc <_sk_callback_hsw+0x28c>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,125,37,0,0         // vbroadcastss  0x257d(%rip),%ymm12        # 48e4 <_sk_callback_hsw+0x290>
+  .byte  196,98,125,24,37,125,37,0,0         // vbroadcastss  0x257d(%rip),%ymm12        # 4900 <_sk_callback_hsw+0x290>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  196,99,125,8,208,1                  // vroundps      $0x1,%ymm0,%ymm10
   .byte  196,65,124,92,210                   // vsubps        %ymm10,%ymm0,%ymm10
-  .byte  196,98,125,24,29,94,37,0,0          // vbroadcastss  0x255e(%rip),%ymm11        # 48e8 <_sk_callback_hsw+0x294>
+  .byte  196,98,125,24,29,94,37,0,0          // vbroadcastss  0x255e(%rip),%ymm11        # 4904 <_sk_callback_hsw+0x294>
   .byte  196,193,124,88,195                  // vaddps        %ymm11,%ymm0,%ymm0
-  .byte  196,98,125,24,29,84,37,0,0          // vbroadcastss  0x2554(%rip),%ymm11        # 48ec <_sk_callback_hsw+0x298>
+  .byte  196,98,125,24,29,84,37,0,0          // vbroadcastss  0x2554(%rip),%ymm11        # 4908 <_sk_callback_hsw+0x298>
   .byte  196,98,45,172,216                   // vfnmadd213ps  %ymm0,%ymm10,%ymm11
-  .byte  196,226,125,24,5,74,37,0,0          // vbroadcastss  0x254a(%rip),%ymm0        # 48f0 <_sk_callback_hsw+0x29c>
+  .byte  196,226,125,24,5,74,37,0,0          // vbroadcastss  0x254a(%rip),%ymm0        # 490c <_sk_callback_hsw+0x29c>
   .byte  196,193,124,92,194                  // vsubps        %ymm10,%ymm0,%ymm0
-  .byte  196,98,125,24,21,64,37,0,0          // vbroadcastss  0x2540(%rip),%ymm10        # 48f4 <_sk_callback_hsw+0x2a0>
+  .byte  196,98,125,24,21,64,37,0,0          // vbroadcastss  0x2540(%rip),%ymm10        # 4910 <_sk_callback_hsw+0x2a0>
   .byte  197,172,94,192                      // vdivps        %ymm0,%ymm10,%ymm0
   .byte  197,164,88,192                      // vaddps        %ymm0,%ymm11,%ymm0
-  .byte  196,98,125,24,21,51,37,0,0          // vbroadcastss  0x2533(%rip),%ymm10        # 48f8 <_sk_callback_hsw+0x2a4>
+  .byte  196,98,125,24,21,51,37,0,0          // vbroadcastss  0x2533(%rip),%ymm10        # 4914 <_sk_callback_hsw+0x2a4>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -10941,7 +10965,7 @@ _sk_parametric_r_hsw:
   .byte  196,195,125,74,193,128              // vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,124,95,192                  // vmaxps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,10,37,0,0           // vbroadcastss  0x250a(%rip),%ymm8        # 48fc <_sk_callback_hsw+0x2a8>
+  .byte  196,98,125,24,5,10,37,0,0           // vbroadcastss  0x250a(%rip),%ymm8        # 4918 <_sk_callback_hsw+0x2a8>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -10961,33 +10985,33 @@ _sk_parametric_g_hsw:
   .byte  196,66,117,168,211                  // vfmadd213ps   %ymm11,%ymm1,%ymm10
   .byte  196,226,125,24,8                    // vbroadcastss  (%rax),%ymm1
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,194,36,0,0         // vbroadcastss  0x24c2(%rip),%ymm12        # 4900 <_sk_callback_hsw+0x2ac>
-  .byte  196,98,125,24,45,189,36,0,0         // vbroadcastss  0x24bd(%rip),%ymm13        # 4904 <_sk_callback_hsw+0x2b0>
+  .byte  196,98,125,24,37,194,36,0,0         // vbroadcastss  0x24c2(%rip),%ymm12        # 491c <_sk_callback_hsw+0x2ac>
+  .byte  196,98,125,24,45,189,36,0,0         // vbroadcastss  0x24bd(%rip),%ymm13        # 4920 <_sk_callback_hsw+0x2b0>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,179,36,0,0         // vbroadcastss  0x24b3(%rip),%ymm13        # 4908 <_sk_callback_hsw+0x2b4>
+  .byte  196,98,125,24,45,179,36,0,0         // vbroadcastss  0x24b3(%rip),%ymm13        # 4924 <_sk_callback_hsw+0x2b4>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,169,36,0,0         // vbroadcastss  0x24a9(%rip),%ymm13        # 490c <_sk_callback_hsw+0x2b8>
+  .byte  196,98,125,24,45,169,36,0,0         // vbroadcastss  0x24a9(%rip),%ymm13        # 4928 <_sk_callback_hsw+0x2b8>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,159,36,0,0         // vbroadcastss  0x249f(%rip),%ymm11        # 4910 <_sk_callback_hsw+0x2bc>
+  .byte  196,98,125,24,29,159,36,0,0         // vbroadcastss  0x249f(%rip),%ymm11        # 492c <_sk_callback_hsw+0x2bc>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,149,36,0,0         // vbroadcastss  0x2495(%rip),%ymm12        # 4914 <_sk_callback_hsw+0x2c0>
+  .byte  196,98,125,24,37,149,36,0,0         // vbroadcastss  0x2495(%rip),%ymm12        # 4930 <_sk_callback_hsw+0x2c0>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,139,36,0,0         // vbroadcastss  0x248b(%rip),%ymm12        # 4918 <_sk_callback_hsw+0x2c4>
+  .byte  196,98,125,24,37,139,36,0,0         // vbroadcastss  0x248b(%rip),%ymm12        # 4934 <_sk_callback_hsw+0x2c4>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  196,99,125,8,209,1                  // vroundps      $0x1,%ymm1,%ymm10
   .byte  196,65,116,92,210                   // vsubps        %ymm10,%ymm1,%ymm10
-  .byte  196,98,125,24,29,108,36,0,0         // vbroadcastss  0x246c(%rip),%ymm11        # 491c <_sk_callback_hsw+0x2c8>
+  .byte  196,98,125,24,29,108,36,0,0         // vbroadcastss  0x246c(%rip),%ymm11        # 4938 <_sk_callback_hsw+0x2c8>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,98,36,0,0          // vbroadcastss  0x2462(%rip),%ymm11        # 4920 <_sk_callback_hsw+0x2cc>
+  .byte  196,98,125,24,29,98,36,0,0          // vbroadcastss  0x2462(%rip),%ymm11        # 493c <_sk_callback_hsw+0x2cc>
   .byte  196,98,45,172,217                   // vfnmadd213ps  %ymm1,%ymm10,%ymm11
-  .byte  196,226,125,24,13,88,36,0,0         // vbroadcastss  0x2458(%rip),%ymm1        # 4924 <_sk_callback_hsw+0x2d0>
+  .byte  196,226,125,24,13,88,36,0,0         // vbroadcastss  0x2458(%rip),%ymm1        # 4940 <_sk_callback_hsw+0x2d0>
   .byte  196,193,116,92,202                  // vsubps        %ymm10,%ymm1,%ymm1
-  .byte  196,98,125,24,21,78,36,0,0          // vbroadcastss  0x244e(%rip),%ymm10        # 4928 <_sk_callback_hsw+0x2d4>
+  .byte  196,98,125,24,21,78,36,0,0          // vbroadcastss  0x244e(%rip),%ymm10        # 4944 <_sk_callback_hsw+0x2d4>
   .byte  197,172,94,201                      // vdivps        %ymm1,%ymm10,%ymm1
   .byte  197,164,88,201                      // vaddps        %ymm1,%ymm11,%ymm1
-  .byte  196,98,125,24,21,65,36,0,0          // vbroadcastss  0x2441(%rip),%ymm10        # 492c <_sk_callback_hsw+0x2d8>
+  .byte  196,98,125,24,21,65,36,0,0          // vbroadcastss  0x2441(%rip),%ymm10        # 4948 <_sk_callback_hsw+0x2d8>
   .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -10995,7 +11019,7 @@ _sk_parametric_g_hsw:
   .byte  196,195,117,74,201,128              // vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,116,95,200                  // vmaxps        %ymm8,%ymm1,%ymm1
-  .byte  196,98,125,24,5,24,36,0,0           // vbroadcastss  0x2418(%rip),%ymm8        # 4930 <_sk_callback_hsw+0x2dc>
+  .byte  196,98,125,24,5,24,36,0,0           // vbroadcastss  0x2418(%rip),%ymm8        # 494c <_sk_callback_hsw+0x2dc>
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11015,33 +11039,33 @@ _sk_parametric_b_hsw:
   .byte  196,66,109,168,211                  // vfmadd213ps   %ymm11,%ymm2,%ymm10
   .byte  196,226,125,24,16                   // vbroadcastss  (%rax),%ymm2
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,208,35,0,0         // vbroadcastss  0x23d0(%rip),%ymm12        # 4934 <_sk_callback_hsw+0x2e0>
-  .byte  196,98,125,24,45,203,35,0,0         // vbroadcastss  0x23cb(%rip),%ymm13        # 4938 <_sk_callback_hsw+0x2e4>
+  .byte  196,98,125,24,37,208,35,0,0         // vbroadcastss  0x23d0(%rip),%ymm12        # 4950 <_sk_callback_hsw+0x2e0>
+  .byte  196,98,125,24,45,203,35,0,0         // vbroadcastss  0x23cb(%rip),%ymm13        # 4954 <_sk_callback_hsw+0x2e4>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,193,35,0,0         // vbroadcastss  0x23c1(%rip),%ymm13        # 493c <_sk_callback_hsw+0x2e8>
+  .byte  196,98,125,24,45,193,35,0,0         // vbroadcastss  0x23c1(%rip),%ymm13        # 4958 <_sk_callback_hsw+0x2e8>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,183,35,0,0         // vbroadcastss  0x23b7(%rip),%ymm13        # 4940 <_sk_callback_hsw+0x2ec>
+  .byte  196,98,125,24,45,183,35,0,0         // vbroadcastss  0x23b7(%rip),%ymm13        # 495c <_sk_callback_hsw+0x2ec>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,173,35,0,0         // vbroadcastss  0x23ad(%rip),%ymm11        # 4944 <_sk_callback_hsw+0x2f0>
+  .byte  196,98,125,24,29,173,35,0,0         // vbroadcastss  0x23ad(%rip),%ymm11        # 4960 <_sk_callback_hsw+0x2f0>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,163,35,0,0         // vbroadcastss  0x23a3(%rip),%ymm12        # 4948 <_sk_callback_hsw+0x2f4>
+  .byte  196,98,125,24,37,163,35,0,0         // vbroadcastss  0x23a3(%rip),%ymm12        # 4964 <_sk_callback_hsw+0x2f4>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,153,35,0,0         // vbroadcastss  0x2399(%rip),%ymm12        # 494c <_sk_callback_hsw+0x2f8>
+  .byte  196,98,125,24,37,153,35,0,0         // vbroadcastss  0x2399(%rip),%ymm12        # 4968 <_sk_callback_hsw+0x2f8>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  196,99,125,8,210,1                  // vroundps      $0x1,%ymm2,%ymm10
   .byte  196,65,108,92,210                   // vsubps        %ymm10,%ymm2,%ymm10
-  .byte  196,98,125,24,29,122,35,0,0         // vbroadcastss  0x237a(%rip),%ymm11        # 4950 <_sk_callback_hsw+0x2fc>
+  .byte  196,98,125,24,29,122,35,0,0         // vbroadcastss  0x237a(%rip),%ymm11        # 496c <_sk_callback_hsw+0x2fc>
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
-  .byte  196,98,125,24,29,112,35,0,0         // vbroadcastss  0x2370(%rip),%ymm11        # 4954 <_sk_callback_hsw+0x300>
+  .byte  196,98,125,24,29,112,35,0,0         // vbroadcastss  0x2370(%rip),%ymm11        # 4970 <_sk_callback_hsw+0x300>
   .byte  196,98,45,172,218                   // vfnmadd213ps  %ymm2,%ymm10,%ymm11
-  .byte  196,226,125,24,21,102,35,0,0        // vbroadcastss  0x2366(%rip),%ymm2        # 4958 <_sk_callback_hsw+0x304>
+  .byte  196,226,125,24,21,102,35,0,0        // vbroadcastss  0x2366(%rip),%ymm2        # 4974 <_sk_callback_hsw+0x304>
   .byte  196,193,108,92,210                  // vsubps        %ymm10,%ymm2,%ymm2
-  .byte  196,98,125,24,21,92,35,0,0          // vbroadcastss  0x235c(%rip),%ymm10        # 495c <_sk_callback_hsw+0x308>
+  .byte  196,98,125,24,21,92,35,0,0          // vbroadcastss  0x235c(%rip),%ymm10        # 4978 <_sk_callback_hsw+0x308>
   .byte  197,172,94,210                      // vdivps        %ymm2,%ymm10,%ymm2
   .byte  197,164,88,210                      // vaddps        %ymm2,%ymm11,%ymm2
-  .byte  196,98,125,24,21,79,35,0,0          // vbroadcastss  0x234f(%rip),%ymm10        # 4960 <_sk_callback_hsw+0x30c>
+  .byte  196,98,125,24,21,79,35,0,0          // vbroadcastss  0x234f(%rip),%ymm10        # 497c <_sk_callback_hsw+0x30c>
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  197,253,91,210                      // vcvtps2dq     %ymm2,%ymm2
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -11049,7 +11073,7 @@ _sk_parametric_b_hsw:
   .byte  196,195,109,74,209,128              // vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,108,95,208                  // vmaxps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,38,35,0,0           // vbroadcastss  0x2326(%rip),%ymm8        # 4964 <_sk_callback_hsw+0x310>
+  .byte  196,98,125,24,5,38,35,0,0           // vbroadcastss  0x2326(%rip),%ymm8        # 4980 <_sk_callback_hsw+0x310>
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11069,33 +11093,33 @@ _sk_parametric_a_hsw:
   .byte  196,66,101,168,211                  // vfmadd213ps   %ymm11,%ymm3,%ymm10
   .byte  196,226,125,24,24                   // vbroadcastss  (%rax),%ymm3
   .byte  196,65,124,91,218                   // vcvtdq2ps     %ymm10,%ymm11
-  .byte  196,98,125,24,37,222,34,0,0         // vbroadcastss  0x22de(%rip),%ymm12        # 4968 <_sk_callback_hsw+0x314>
-  .byte  196,98,125,24,45,217,34,0,0         // vbroadcastss  0x22d9(%rip),%ymm13        # 496c <_sk_callback_hsw+0x318>
+  .byte  196,98,125,24,37,222,34,0,0         // vbroadcastss  0x22de(%rip),%ymm12        # 4984 <_sk_callback_hsw+0x314>
+  .byte  196,98,125,24,45,217,34,0,0         // vbroadcastss  0x22d9(%rip),%ymm13        # 4988 <_sk_callback_hsw+0x318>
   .byte  196,65,44,84,213                    // vandps        %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,207,34,0,0         // vbroadcastss  0x22cf(%rip),%ymm13        # 4970 <_sk_callback_hsw+0x31c>
+  .byte  196,98,125,24,45,207,34,0,0         // vbroadcastss  0x22cf(%rip),%ymm13        # 498c <_sk_callback_hsw+0x31c>
   .byte  196,65,44,86,213                    // vorps         %ymm13,%ymm10,%ymm10
-  .byte  196,98,125,24,45,197,34,0,0         // vbroadcastss  0x22c5(%rip),%ymm13        # 4974 <_sk_callback_hsw+0x320>
+  .byte  196,98,125,24,45,197,34,0,0         // vbroadcastss  0x22c5(%rip),%ymm13        # 4990 <_sk_callback_hsw+0x320>
   .byte  196,66,37,184,236                   // vfmadd231ps   %ymm12,%ymm11,%ymm13
-  .byte  196,98,125,24,29,187,34,0,0         // vbroadcastss  0x22bb(%rip),%ymm11        # 4978 <_sk_callback_hsw+0x324>
+  .byte  196,98,125,24,29,187,34,0,0         // vbroadcastss  0x22bb(%rip),%ymm11        # 4994 <_sk_callback_hsw+0x324>
   .byte  196,66,45,172,221                   // vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  .byte  196,98,125,24,37,177,34,0,0         // vbroadcastss  0x22b1(%rip),%ymm12        # 497c <_sk_callback_hsw+0x328>
+  .byte  196,98,125,24,37,177,34,0,0         // vbroadcastss  0x22b1(%rip),%ymm12        # 4998 <_sk_callback_hsw+0x328>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,167,34,0,0         // vbroadcastss  0x22a7(%rip),%ymm12        # 4980 <_sk_callback_hsw+0x32c>
+  .byte  196,98,125,24,37,167,34,0,0         // vbroadcastss  0x22a7(%rip),%ymm12        # 499c <_sk_callback_hsw+0x32c>
   .byte  196,65,28,94,210                    // vdivps        %ymm10,%ymm12,%ymm10
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
   .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
   .byte  196,99,125,8,211,1                  // vroundps      $0x1,%ymm3,%ymm10
   .byte  196,65,100,92,210                   // vsubps        %ymm10,%ymm3,%ymm10
-  .byte  196,98,125,24,29,136,34,0,0         // vbroadcastss  0x2288(%rip),%ymm11        # 4984 <_sk_callback_hsw+0x330>
+  .byte  196,98,125,24,29,136,34,0,0         // vbroadcastss  0x2288(%rip),%ymm11        # 49a0 <_sk_callback_hsw+0x330>
   .byte  196,193,100,88,219                  // vaddps        %ymm11,%ymm3,%ymm3
-  .byte  196,98,125,24,29,126,34,0,0         // vbroadcastss  0x227e(%rip),%ymm11        # 4988 <_sk_callback_hsw+0x334>
+  .byte  196,98,125,24,29,126,34,0,0         // vbroadcastss  0x227e(%rip),%ymm11        # 49a4 <_sk_callback_hsw+0x334>
   .byte  196,98,45,172,219                   // vfnmadd213ps  %ymm3,%ymm10,%ymm11
-  .byte  196,226,125,24,29,116,34,0,0        // vbroadcastss  0x2274(%rip),%ymm3        # 498c <_sk_callback_hsw+0x338>
+  .byte  196,226,125,24,29,116,34,0,0        // vbroadcastss  0x2274(%rip),%ymm3        # 49a8 <_sk_callback_hsw+0x338>
   .byte  196,193,100,92,218                  // vsubps        %ymm10,%ymm3,%ymm3
-  .byte  196,98,125,24,21,106,34,0,0         // vbroadcastss  0x226a(%rip),%ymm10        # 4990 <_sk_callback_hsw+0x33c>
+  .byte  196,98,125,24,21,106,34,0,0         // vbroadcastss  0x226a(%rip),%ymm10        # 49ac <_sk_callback_hsw+0x33c>
   .byte  197,172,94,219                      // vdivps        %ymm3,%ymm10,%ymm3
   .byte  197,164,88,219                      // vaddps        %ymm3,%ymm11,%ymm3
-  .byte  196,98,125,24,21,93,34,0,0          // vbroadcastss  0x225d(%rip),%ymm10        # 4994 <_sk_callback_hsw+0x340>
+  .byte  196,98,125,24,21,93,34,0,0          // vbroadcastss  0x225d(%rip),%ymm10        # 49b0 <_sk_callback_hsw+0x340>
   .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
   .byte  197,253,91,219                      // vcvtps2dq     %ymm3,%ymm3
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -11103,7 +11127,7 @@ _sk_parametric_a_hsw:
   .byte  196,195,101,74,217,128              // vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,100,95,216                  // vmaxps        %ymm8,%ymm3,%ymm3
-  .byte  196,98,125,24,5,52,34,0,0           // vbroadcastss  0x2234(%rip),%ymm8        # 4998 <_sk_callback_hsw+0x344>
+  .byte  196,98,125,24,5,52,34,0,0           // vbroadcastss  0x2234(%rip),%ymm8        # 49b4 <_sk_callback_hsw+0x344>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11112,26 +11136,26 @@ HIDDEN _sk_lab_to_xyz_hsw
 .globl _sk_lab_to_xyz_hsw
 FUNCTION(_sk_lab_to_xyz_hsw)
 _sk_lab_to_xyz_hsw:
-  .byte  196,98,125,24,5,38,34,0,0           // vbroadcastss  0x2226(%rip),%ymm8        # 499c <_sk_callback_hsw+0x348>
-  .byte  196,98,125,24,13,33,34,0,0          // vbroadcastss  0x2221(%rip),%ymm9        # 49a0 <_sk_callback_hsw+0x34c>
-  .byte  196,98,125,24,21,28,34,0,0          // vbroadcastss  0x221c(%rip),%ymm10        # 49a4 <_sk_callback_hsw+0x350>
+  .byte  196,98,125,24,5,38,34,0,0           // vbroadcastss  0x2226(%rip),%ymm8        # 49b8 <_sk_callback_hsw+0x348>
+  .byte  196,98,125,24,13,33,34,0,0          // vbroadcastss  0x2221(%rip),%ymm9        # 49bc <_sk_callback_hsw+0x34c>
+  .byte  196,98,125,24,21,28,34,0,0          // vbroadcastss  0x221c(%rip),%ymm10        # 49c0 <_sk_callback_hsw+0x350>
   .byte  196,194,53,168,202                  // vfmadd213ps   %ymm10,%ymm9,%ymm1
   .byte  196,194,53,168,210                  // vfmadd213ps   %ymm10,%ymm9,%ymm2
-  .byte  196,98,125,24,13,13,34,0,0          // vbroadcastss  0x220d(%rip),%ymm9        # 49a8 <_sk_callback_hsw+0x354>
+  .byte  196,98,125,24,13,13,34,0,0          // vbroadcastss  0x220d(%rip),%ymm9        # 49c4 <_sk_callback_hsw+0x354>
   .byte  196,66,125,184,200                  // vfmadd231ps   %ymm8,%ymm0,%ymm9
-  .byte  196,226,125,24,5,3,34,0,0           // vbroadcastss  0x2203(%rip),%ymm0        # 49ac <_sk_callback_hsw+0x358>
+  .byte  196,226,125,24,5,3,34,0,0           // vbroadcastss  0x2203(%rip),%ymm0        # 49c8 <_sk_callback_hsw+0x358>
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
-  .byte  196,98,125,24,5,250,33,0,0          // vbroadcastss  0x21fa(%rip),%ymm8        # 49b0 <_sk_callback_hsw+0x35c>
+  .byte  196,98,125,24,5,250,33,0,0          // vbroadcastss  0x21fa(%rip),%ymm8        # 49cc <_sk_callback_hsw+0x35c>
   .byte  196,98,117,168,192                  // vfmadd213ps   %ymm0,%ymm1,%ymm8
-  .byte  196,98,125,24,13,240,33,0,0         // vbroadcastss  0x21f0(%rip),%ymm9        # 49b4 <_sk_callback_hsw+0x360>
+  .byte  196,98,125,24,13,240,33,0,0         // vbroadcastss  0x21f0(%rip),%ymm9        # 49d0 <_sk_callback_hsw+0x360>
   .byte  196,98,109,172,200                  // vfnmadd213ps  %ymm0,%ymm2,%ymm9
   .byte  196,193,60,89,200                   // vmulps        %ymm8,%ymm8,%ymm1
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
-  .byte  196,226,125,24,21,221,33,0,0        // vbroadcastss  0x21dd(%rip),%ymm2        # 49b8 <_sk_callback_hsw+0x364>
+  .byte  196,226,125,24,21,221,33,0,0        // vbroadcastss  0x21dd(%rip),%ymm2        # 49d4 <_sk_callback_hsw+0x364>
   .byte  197,108,194,209,1                   // vcmpltps      %ymm1,%ymm2,%ymm10
-  .byte  196,98,125,24,29,211,33,0,0         // vbroadcastss  0x21d3(%rip),%ymm11        # 49bc <_sk_callback_hsw+0x368>
+  .byte  196,98,125,24,29,211,33,0,0         // vbroadcastss  0x21d3(%rip),%ymm11        # 49d8 <_sk_callback_hsw+0x368>
   .byte  196,65,60,88,195                    // vaddps        %ymm11,%ymm8,%ymm8
-  .byte  196,98,125,24,37,201,33,0,0         // vbroadcastss  0x21c9(%rip),%ymm12        # 49c0 <_sk_callback_hsw+0x36c>
+  .byte  196,98,125,24,37,201,33,0,0         // vbroadcastss  0x21c9(%rip),%ymm12        # 49dc <_sk_callback_hsw+0x36c>
   .byte  196,65,60,89,196                    // vmulps        %ymm12,%ymm8,%ymm8
   .byte  196,99,61,74,193,160                // vblendvps     %ymm10,%ymm1,%ymm8,%ymm8
   .byte  197,252,89,200                      // vmulps        %ymm0,%ymm0,%ymm1
@@ -11146,9 +11170,9 @@ _sk_lab_to_xyz_hsw:
   .byte  196,65,52,88,203                    // vaddps        %ymm11,%ymm9,%ymm9
   .byte  196,65,52,89,204                    // vmulps        %ymm12,%ymm9,%ymm9
   .byte  196,227,53,74,208,32                // vblendvps     %ymm2,%ymm0,%ymm9,%ymm2
-  .byte  196,226,125,24,5,126,33,0,0         // vbroadcastss  0x217e(%rip),%ymm0        # 49c4 <_sk_callback_hsw+0x370>
+  .byte  196,226,125,24,5,126,33,0,0         // vbroadcastss  0x217e(%rip),%ymm0        # 49e0 <_sk_callback_hsw+0x370>
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
-  .byte  196,98,125,24,5,117,33,0,0          // vbroadcastss  0x2175(%rip),%ymm8        # 49c8 <_sk_callback_hsw+0x374>
+  .byte  196,98,125,24,5,117,33,0,0          // vbroadcastss  0x2175(%rip),%ymm8        # 49e4 <_sk_callback_hsw+0x374>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11162,11 +11186,11 @@ _sk_load_a8_hsw:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,45                              // jne           2899 <_sk_load_a8_hsw+0x3d>
+  .byte  117,45                              // jne           28b5 <_sk_load_a8_hsw+0x3d>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,74,33,0,0         // vbroadcastss  0x214a(%rip),%ymm1        # 49cc <_sk_callback_hsw+0x378>
+  .byte  196,226,125,24,13,74,33,0,0         // vbroadcastss  0x214a(%rip),%ymm1        # 49e8 <_sk_callback_hsw+0x378>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -11183,9 +11207,9 @@ _sk_load_a8_hsw:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           28a1 <_sk_load_a8_hsw+0x45>
+  .byte  117,234                             // jne           28bd <_sk_load_a8_hsw+0x45>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,178                             // jmp           2870 <_sk_load_a8_hsw+0x14>
+  .byte  235,178                             // jmp           288c <_sk_load_a8_hsw+0x14>
 
 HIDDEN _sk_gather_a8_hsw
 .globl _sk_gather_a8_hsw
@@ -11231,7 +11255,7 @@ _sk_gather_a8_hsw:
   .byte  196,227,121,32,192,7                // vpinsrb       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,85,32,0,0         // vbroadcastss  0x2055(%rip),%ymm1        # 49d0 <_sk_callback_hsw+0x37c>
+  .byte  196,226,125,24,13,85,32,0,0         // vbroadcastss  0x2055(%rip),%ymm1        # 49ec <_sk_callback_hsw+0x37c>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -11249,14 +11273,14 @@ FUNCTION(_sk_store_a8_hsw)
 _sk_store_a8_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,48,32,0,0           // vbroadcastss  0x2030(%rip),%ymm8        # 49d4 <_sk_callback_hsw+0x380>
+  .byte  196,98,125,24,5,48,32,0,0           // vbroadcastss  0x2030(%rip),%ymm8        # 49f0 <_sk_callback_hsw+0x380>
   .byte  196,65,100,89,192                   // vmulps        %ymm8,%ymm3,%ymm8
   .byte  196,65,125,91,192                   // vcvtps2dq     %ymm8,%ymm8
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  196,65,57,103,192                   // vpackuswb     %xmm8,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           29cd <_sk_store_a8_hsw+0x37>
+  .byte  117,10                              // jne           29e9 <_sk_store_a8_hsw+0x37>
   .byte  196,65,123,17,4,58                  // vmovsd        %xmm8,(%r10,%rdi,1)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11264,10 +11288,10 @@ _sk_store_a8_hsw:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            29c9 <_sk_store_a8_hsw+0x33>
+  .byte  119,236                             // ja            29e5 <_sk_store_a8_hsw+0x33>
   .byte  196,66,121,48,192                   // vpmovzxbw     %xmm8,%xmm8
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 2a30 <_sk_store_a8_hsw+0x9a>
+  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 2a4c <_sk_store_a8_hsw+0x9a>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11278,7 +11302,7 @@ _sk_store_a8_hsw:
   .byte  196,67,121,20,68,58,2,4             // vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   .byte  196,67,121,20,68,58,1,2             // vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   .byte  196,67,121,20,4,58,0                // vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  .byte  235,154                             // jmp           29c9 <_sk_store_a8_hsw+0x33>
+  .byte  235,154                             // jmp           29e5 <_sk_store_a8_hsw+0x33>
   .byte  144                                 // nop
   .byte  246,255                             // idiv          %bh
   .byte  255                                 // (bad)
@@ -11312,14 +11336,14 @@ _sk_load_g8_hsw:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,50                              // jne           2a8e <_sk_load_g8_hsw+0x42>
+  .byte  117,50                              // jne           2aaa <_sk_load_g8_hsw+0x42>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,102,31,0,0        // vbroadcastss  0x1f66(%rip),%ymm1        # 49d8 <_sk_callback_hsw+0x384>
+  .byte  196,226,125,24,13,102,31,0,0        // vbroadcastss  0x1f66(%rip),%ymm1        # 49f4 <_sk_callback_hsw+0x384>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,91,31,0,0         // vbroadcastss  0x1f5b(%rip),%ymm3        # 49dc <_sk_callback_hsw+0x388>
+  .byte  196,226,125,24,29,91,31,0,0         // vbroadcastss  0x1f5b(%rip),%ymm3        # 49f8 <_sk_callback_hsw+0x388>
   .byte  76,137,193                          // mov           %r8,%rcx
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
@@ -11333,9 +11357,9 @@ _sk_load_g8_hsw:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           2a96 <_sk_load_g8_hsw+0x4a>
+  .byte  117,234                             // jne           2ab2 <_sk_load_g8_hsw+0x4a>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,173                             // jmp           2a60 <_sk_load_g8_hsw+0x14>
+  .byte  235,173                             // jmp           2a7c <_sk_load_g8_hsw+0x14>
 
 HIDDEN _sk_gather_g8_hsw
 .globl _sk_gather_g8_hsw
@@ -11381,10 +11405,10 @@ _sk_gather_g8_hsw:
   .byte  196,227,121,32,192,7                // vpinsrb       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,49,192                  // vpmovzxbd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,112,30,0,0        // vbroadcastss  0x1e70(%rip),%ymm1        # 49e0 <_sk_callback_hsw+0x38c>
+  .byte  196,226,125,24,13,112,30,0,0        // vbroadcastss  0x1e70(%rip),%ymm1        # 49fc <_sk_callback_hsw+0x38c>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,101,30,0,0        // vbroadcastss  0x1e65(%rip),%ymm3        # 49e4 <_sk_callback_hsw+0x390>
+  .byte  196,226,125,24,29,101,30,0,0        // vbroadcastss  0x1e65(%rip),%ymm3        # 4a00 <_sk_callback_hsw+0x390>
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
   .byte  91                                  // pop           %rbx
@@ -11400,9 +11424,9 @@ _sk_gather_i8_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            2b9f <_sk_gather_i8_hsw+0xf>
+  .byte  116,5                               // je            2bbb <_sk_gather_i8_hsw+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           2ba1 <_sk_gather_i8_hsw+0x11>
+  .byte  235,2                               // jmp           2bbd <_sk_gather_i8_hsw+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,87                               // push          %r15
   .byte  65,86                               // push          %r14
@@ -11440,14 +11464,14 @@ _sk_gather_i8_hsw:
   .byte  73,139,64,8                         // mov           0x8(%r8),%rax
   .byte  197,245,118,201                     // vpcmpeqd      %ymm1,%ymm1,%ymm1
   .byte  196,226,117,144,28,128              // vpgatherdd    %ymm1,(%rax,%ymm0,4),%ymm3
-  .byte  197,229,219,5,113,31,0,0            // vpand         0x1f71(%rip),%ymm3,%ymm0        # 4bc0 <_sk_callback_hsw+0x56c>
+  .byte  197,229,219,5,117,31,0,0            // vpand         0x1f75(%rip),%ymm3,%ymm0        # 4be0 <_sk_callback_hsw+0x570>
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,140,29,0,0          // vbroadcastss  0x1d8c(%rip),%ymm8        # 49e8 <_sk_callback_hsw+0x394>
+  .byte  196,98,125,24,5,140,29,0,0          // vbroadcastss  0x1d8c(%rip),%ymm8        # 4a04 <_sk_callback_hsw+0x394>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,226,101,0,13,118,31,0,0         // vpshufb       0x1f76(%rip),%ymm3,%ymm1        # 4be0 <_sk_callback_hsw+0x58c>
+  .byte  196,226,101,0,13,122,31,0,0         // vpshufb       0x1f7a(%rip),%ymm3,%ymm1        # 4c00 <_sk_callback_hsw+0x590>
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,226,101,0,21,132,31,0,0         // vpshufb       0x1f84(%rip),%ymm3,%ymm2        # 4c00 <_sk_callback_hsw+0x5ac>
+  .byte  196,226,101,0,21,136,31,0,0         // vpshufb       0x1f88(%rip),%ymm3,%ymm2        # 4c20 <_sk_callback_hsw+0x5b0>
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
@@ -11468,35 +11492,35 @@ _sk_load_565_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,114                             // jne           2d1c <_sk_load_565_hsw+0x7c>
+  .byte  117,114                             // jne           2d38 <_sk_load_565_hsw+0x7c>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  196,226,125,51,208                  // vpmovzxwd     %xmm0,%ymm2
-  .byte  196,226,125,88,5,46,29,0,0          // vpbroadcastd  0x1d2e(%rip),%ymm0        # 49ec <_sk_callback_hsw+0x398>
+  .byte  196,226,125,88,5,46,29,0,0          // vpbroadcastd  0x1d2e(%rip),%ymm0        # 4a08 <_sk_callback_hsw+0x398>
   .byte  197,237,219,192                     // vpand         %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,33,29,0,0         // vbroadcastss  0x1d21(%rip),%ymm1        # 49f0 <_sk_callback_hsw+0x39c>
+  .byte  196,226,125,24,13,33,29,0,0         // vbroadcastss  0x1d21(%rip),%ymm1        # 4a0c <_sk_callback_hsw+0x39c>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,24,29,0,0         // vpbroadcastd  0x1d18(%rip),%ymm1        # 49f4 <_sk_callback_hsw+0x3a0>
+  .byte  196,226,125,88,13,24,29,0,0         // vpbroadcastd  0x1d18(%rip),%ymm1        # 4a10 <_sk_callback_hsw+0x3a0>
   .byte  197,237,219,201                     // vpand         %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,11,29,0,0         // vbroadcastss  0x1d0b(%rip),%ymm3        # 49f8 <_sk_callback_hsw+0x3a4>
+  .byte  196,226,125,24,29,11,29,0,0         // vbroadcastss  0x1d0b(%rip),%ymm3        # 4a14 <_sk_callback_hsw+0x3a4>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,88,29,2,29,0,0          // vpbroadcastd  0x1d02(%rip),%ymm3        # 49fc <_sk_callback_hsw+0x3a8>
+  .byte  196,226,125,88,29,2,29,0,0          // vpbroadcastd  0x1d02(%rip),%ymm3        # 4a18 <_sk_callback_hsw+0x3a8>
   .byte  197,237,219,211                     // vpand         %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,245,28,0,0        // vbroadcastss  0x1cf5(%rip),%ymm3        # 4a00 <_sk_callback_hsw+0x3ac>
+  .byte  196,226,125,24,29,245,28,0,0        // vbroadcastss  0x1cf5(%rip),%ymm3        # 4a1c <_sk_callback_hsw+0x3ac>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,234,28,0,0        // vbroadcastss  0x1cea(%rip),%ymm3        # 4a04 <_sk_callback_hsw+0x3b0>
+  .byte  196,226,125,24,29,234,28,0,0        // vbroadcastss  0x1cea(%rip),%ymm3        # 4a20 <_sk_callback_hsw+0x3b0>
   .byte  255,224                             // jmpq          *%rax
   .byte  65,137,200                          // mov           %ecx,%r8d
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,128                             // ja            2cb0 <_sk_load_565_hsw+0x10>
+  .byte  119,128                             // ja            2ccc <_sk_load_565_hsw+0x10>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 2d84 <_sk_load_565_hsw+0xe4>
+  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 2da0 <_sk_load_565_hsw+0xe4>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11508,7 +11532,7 @@ _sk_load_565_hsw:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,44,255,255,255                  // jmpq          2cb0 <_sk_load_565_hsw+0x10>
+  .byte  233,44,255,255,255                  // jmpq          2ccc <_sk_load_565_hsw+0x10>
   .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -11578,23 +11602,23 @@ _sk_gather_565_hsw:
   .byte  65,15,183,4,88                      // movzwl        (%r8,%rbx,2),%eax
   .byte  197,249,196,192,7                   // vpinsrw       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,51,208                  // vpmovzxwd     %xmm0,%ymm2
-  .byte  196,226,125,88,5,173,27,0,0         // vpbroadcastd  0x1bad(%rip),%ymm0        # 4a08 <_sk_callback_hsw+0x3b4>
+  .byte  196,226,125,88,5,173,27,0,0         // vpbroadcastd  0x1bad(%rip),%ymm0        # 4a24 <_sk_callback_hsw+0x3b4>
   .byte  197,237,219,192                     // vpand         %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,160,27,0,0        // vbroadcastss  0x1ba0(%rip),%ymm1        # 4a0c <_sk_callback_hsw+0x3b8>
+  .byte  196,226,125,24,13,160,27,0,0        // vbroadcastss  0x1ba0(%rip),%ymm1        # 4a28 <_sk_callback_hsw+0x3b8>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,151,27,0,0        // vpbroadcastd  0x1b97(%rip),%ymm1        # 4a10 <_sk_callback_hsw+0x3bc>
+  .byte  196,226,125,88,13,151,27,0,0        // vpbroadcastd  0x1b97(%rip),%ymm1        # 4a2c <_sk_callback_hsw+0x3bc>
   .byte  197,237,219,201                     // vpand         %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,138,27,0,0        // vbroadcastss  0x1b8a(%rip),%ymm3        # 4a14 <_sk_callback_hsw+0x3c0>
+  .byte  196,226,125,24,29,138,27,0,0        // vbroadcastss  0x1b8a(%rip),%ymm3        # 4a30 <_sk_callback_hsw+0x3c0>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,88,29,129,27,0,0        // vpbroadcastd  0x1b81(%rip),%ymm3        # 4a18 <_sk_callback_hsw+0x3c4>
+  .byte  196,226,125,88,29,129,27,0,0        // vpbroadcastd  0x1b81(%rip),%ymm3        # 4a34 <_sk_callback_hsw+0x3c4>
   .byte  197,237,219,211                     // vpand         %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,116,27,0,0        // vbroadcastss  0x1b74(%rip),%ymm3        # 4a1c <_sk_callback_hsw+0x3c8>
+  .byte  196,226,125,24,29,116,27,0,0        // vbroadcastss  0x1b74(%rip),%ymm3        # 4a38 <_sk_callback_hsw+0x3c8>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,105,27,0,0        // vbroadcastss  0x1b69(%rip),%ymm3        # 4a20 <_sk_callback_hsw+0x3cc>
+  .byte  196,226,125,24,29,105,27,0,0        // vbroadcastss  0x1b69(%rip),%ymm3        # 4a3c <_sk_callback_hsw+0x3cc>
   .byte  91                                  // pop           %rbx
   .byte  65,92                               // pop           %r12
   .byte  65,94                               // pop           %r14
@@ -11607,11 +11631,11 @@ FUNCTION(_sk_store_565_hsw)
 _sk_store_565_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,86,27,0,0           // vbroadcastss  0x1b56(%rip),%ymm8        # 4a24 <_sk_callback_hsw+0x3d0>
+  .byte  196,98,125,24,5,86,27,0,0           // vbroadcastss  0x1b56(%rip),%ymm8        # 4a40 <_sk_callback_hsw+0x3d0>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,53,114,241,11               // vpslld        $0xb,%ymm9,%ymm9
-  .byte  196,98,125,24,21,65,27,0,0          // vbroadcastss  0x1b41(%rip),%ymm10        # 4a28 <_sk_callback_hsw+0x3d4>
+  .byte  196,98,125,24,21,65,27,0,0          // vbroadcastss  0x1b41(%rip),%ymm10        # 4a44 <_sk_callback_hsw+0x3d4>
   .byte  196,65,116,89,210                   // vmulps        %ymm10,%ymm1,%ymm10
   .byte  196,65,125,91,210                   // vcvtps2dq     %ymm10,%ymm10
   .byte  196,193,45,114,242,5                // vpslld        $0x5,%ymm10,%ymm10
@@ -11622,7 +11646,7 @@ _sk_store_565_hsw:
   .byte  196,67,125,57,193,1                 // vextracti128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           2f25 <_sk_store_565_hsw+0x65>
+  .byte  117,10                              // jne           2f41 <_sk_store_565_hsw+0x65>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11630,9 +11654,9 @@ _sk_store_565_hsw:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            2f21 <_sk_store_565_hsw+0x61>
+  .byte  119,236                             // ja            2f3d <_sk_store_565_hsw+0x61>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 2f84 <_sk_store_565_hsw+0xc4>
+  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 2fa0 <_sk_store_565_hsw+0xc4>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11643,7 +11667,7 @@ _sk_store_565_hsw:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           2f21 <_sk_store_565_hsw+0x61>
+  .byte  235,159                             // jmp           2f3d <_sk_store_565_hsw+0x61>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -11676,28 +11700,28 @@ _sk_load_4444_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,138,0,0,0                    // jne           3038 <_sk_load_4444_hsw+0x98>
+  .byte  15,133,138,0,0,0                    // jne           3054 <_sk_load_4444_hsw+0x98>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  196,226,125,51,216                  // vpmovzxwd     %xmm0,%ymm3
-  .byte  196,226,125,88,5,106,26,0,0         // vpbroadcastd  0x1a6a(%rip),%ymm0        # 4a2c <_sk_callback_hsw+0x3d8>
+  .byte  196,226,125,88,5,106,26,0,0         // vpbroadcastd  0x1a6a(%rip),%ymm0        # 4a48 <_sk_callback_hsw+0x3d8>
   .byte  197,229,219,192                     // vpand         %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,93,26,0,0         // vbroadcastss  0x1a5d(%rip),%ymm1        # 4a30 <_sk_callback_hsw+0x3dc>
+  .byte  196,226,125,24,13,93,26,0,0         // vbroadcastss  0x1a5d(%rip),%ymm1        # 4a4c <_sk_callback_hsw+0x3dc>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,84,26,0,0         // vpbroadcastd  0x1a54(%rip),%ymm1        # 4a34 <_sk_callback_hsw+0x3e0>
+  .byte  196,226,125,88,13,84,26,0,0         // vpbroadcastd  0x1a54(%rip),%ymm1        # 4a50 <_sk_callback_hsw+0x3e0>
   .byte  197,229,219,201                     // vpand         %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,71,26,0,0         // vbroadcastss  0x1a47(%rip),%ymm2        # 4a38 <_sk_callback_hsw+0x3e4>
+  .byte  196,226,125,24,21,71,26,0,0         // vbroadcastss  0x1a47(%rip),%ymm2        # 4a54 <_sk_callback_hsw+0x3e4>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,88,21,62,26,0,0         // vpbroadcastd  0x1a3e(%rip),%ymm2        # 4a3c <_sk_callback_hsw+0x3e8>
+  .byte  196,226,125,88,21,62,26,0,0         // vpbroadcastd  0x1a3e(%rip),%ymm2        # 4a58 <_sk_callback_hsw+0x3e8>
   .byte  197,229,219,210                     // vpand         %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,49,26,0,0           // vbroadcastss  0x1a31(%rip),%ymm8        # 4a40 <_sk_callback_hsw+0x3ec>
+  .byte  196,98,125,24,5,49,26,0,0           // vbroadcastss  0x1a31(%rip),%ymm8        # 4a5c <_sk_callback_hsw+0x3ec>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,88,5,39,26,0,0           // vpbroadcastd  0x1a27(%rip),%ymm8        # 4a44 <_sk_callback_hsw+0x3f0>
+  .byte  196,98,125,88,5,39,26,0,0           // vpbroadcastd  0x1a27(%rip),%ymm8        # 4a60 <_sk_callback_hsw+0x3f0>
   .byte  196,193,101,219,216                 // vpand         %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,25,26,0,0           // vbroadcastss  0x1a19(%rip),%ymm8        # 4a48 <_sk_callback_hsw+0x3f4>
+  .byte  196,98,125,24,5,25,26,0,0           // vbroadcastss  0x1a19(%rip),%ymm8        # 4a64 <_sk_callback_hsw+0x3f4>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11706,9 +11730,9 @@ _sk_load_4444_hsw:
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,100,255,255,255              // ja            2fb4 <_sk_load_4444_hsw+0x14>
+  .byte  15,135,100,255,255,255              // ja            2fd0 <_sk_load_4444_hsw+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 30a4 <_sk_load_4444_hsw+0x104>
+  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 30c0 <_sk_load_4444_hsw+0x104>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11720,7 +11744,7 @@ _sk_load_4444_hsw:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,16,255,255,255                  // jmpq          2fb4 <_sk_load_4444_hsw+0x14>
+  .byte  233,16,255,255,255                  // jmpq          2fd0 <_sk_load_4444_hsw+0x14>
   .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -11790,25 +11814,25 @@ _sk_gather_4444_hsw:
   .byte  65,15,183,4,88                      // movzwl        (%r8,%rbx,2),%eax
   .byte  197,249,196,192,7                   // vpinsrw       $0x7,%eax,%xmm0,%xmm0
   .byte  196,226,125,51,216                  // vpmovzxwd     %xmm0,%ymm3
-  .byte  196,226,125,88,5,209,24,0,0         // vpbroadcastd  0x18d1(%rip),%ymm0        # 4a4c <_sk_callback_hsw+0x3f8>
+  .byte  196,226,125,88,5,209,24,0,0         // vpbroadcastd  0x18d1(%rip),%ymm0        # 4a68 <_sk_callback_hsw+0x3f8>
   .byte  197,229,219,192                     // vpand         %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,196,24,0,0        // vbroadcastss  0x18c4(%rip),%ymm1        # 4a50 <_sk_callback_hsw+0x3fc>
+  .byte  196,226,125,24,13,196,24,0,0        // vbroadcastss  0x18c4(%rip),%ymm1        # 4a6c <_sk_callback_hsw+0x3fc>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,88,13,187,24,0,0        // vpbroadcastd  0x18bb(%rip),%ymm1        # 4a54 <_sk_callback_hsw+0x400>
+  .byte  196,226,125,88,13,187,24,0,0        // vpbroadcastd  0x18bb(%rip),%ymm1        # 4a70 <_sk_callback_hsw+0x400>
   .byte  197,229,219,201                     // vpand         %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,174,24,0,0        // vbroadcastss  0x18ae(%rip),%ymm2        # 4a58 <_sk_callback_hsw+0x404>
+  .byte  196,226,125,24,21,174,24,0,0        // vbroadcastss  0x18ae(%rip),%ymm2        # 4a74 <_sk_callback_hsw+0x404>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,88,21,165,24,0,0        // vpbroadcastd  0x18a5(%rip),%ymm2        # 4a5c <_sk_callback_hsw+0x408>
+  .byte  196,226,125,88,21,165,24,0,0        // vpbroadcastd  0x18a5(%rip),%ymm2        # 4a78 <_sk_callback_hsw+0x408>
   .byte  197,229,219,210                     // vpand         %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,152,24,0,0          // vbroadcastss  0x1898(%rip),%ymm8        # 4a60 <_sk_callback_hsw+0x40c>
+  .byte  196,98,125,24,5,152,24,0,0          // vbroadcastss  0x1898(%rip),%ymm8        # 4a7c <_sk_callback_hsw+0x40c>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,88,5,142,24,0,0          // vpbroadcastd  0x188e(%rip),%ymm8        # 4a64 <_sk_callback_hsw+0x410>
+  .byte  196,98,125,88,5,142,24,0,0          // vpbroadcastd  0x188e(%rip),%ymm8        # 4a80 <_sk_callback_hsw+0x410>
   .byte  196,193,101,219,216                 // vpand         %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,128,24,0,0          // vbroadcastss  0x1880(%rip),%ymm8        # 4a68 <_sk_callback_hsw+0x414>
+  .byte  196,98,125,24,5,128,24,0,0          // vbroadcastss  0x1880(%rip),%ymm8        # 4a84 <_sk_callback_hsw+0x414>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -11823,7 +11847,7 @@ FUNCTION(_sk_store_4444_hsw)
 _sk_store_4444_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,102,24,0,0          // vbroadcastss  0x1866(%rip),%ymm8        # 4a6c <_sk_callback_hsw+0x418>
+  .byte  196,98,125,24,5,102,24,0,0          // vbroadcastss  0x1866(%rip),%ymm8        # 4a88 <_sk_callback_hsw+0x418>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,53,114,241,12               // vpslld        $0xc,%ymm9,%ymm9
@@ -11841,7 +11865,7 @@ _sk_store_4444_hsw:
   .byte  196,67,125,57,193,1                 // vextracti128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3269 <_sk_store_4444_hsw+0x71>
+  .byte  117,10                              // jne           3285 <_sk_store_4444_hsw+0x71>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11849,9 +11873,9 @@ _sk_store_4444_hsw:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3265 <_sk_store_4444_hsw+0x6d>
+  .byte  119,236                             // ja            3281 <_sk_store_4444_hsw+0x6d>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 32c8 <_sk_store_4444_hsw+0xd0>
+  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 32e4 <_sk_store_4444_hsw+0xd0>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -11862,7 +11886,7 @@ _sk_store_4444_hsw:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           3265 <_sk_store_4444_hsw+0x6d>
+  .byte  235,159                             // jmp           3281 <_sk_store_4444_hsw+0x6d>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -11897,16 +11921,16 @@ _sk_load_8888_hsw:
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  76,3,8                              // add           (%rax),%r9
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,88                              // jne           3351 <_sk_load_8888_hsw+0x6d>
+  .byte  117,88                              // jne           336d <_sk_load_8888_hsw+0x6d>
   .byte  196,193,126,111,25                  // vmovdqu       (%r9),%ymm3
-  .byte  197,229,219,5,26,25,0,0             // vpand         0x191a(%rip),%ymm3,%ymm0        # 4c20 <_sk_callback_hsw+0x5cc>
+  .byte  197,229,219,5,30,25,0,0             // vpand         0x191e(%rip),%ymm3,%ymm0        # 4c40 <_sk_callback_hsw+0x5d0>
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,93,23,0,0           // vbroadcastss  0x175d(%rip),%ymm8        # 4a70 <_sk_callback_hsw+0x41c>
+  .byte  196,98,125,24,5,93,23,0,0           // vbroadcastss  0x175d(%rip),%ymm8        # 4a8c <_sk_callback_hsw+0x41c>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,226,101,0,13,31,25,0,0          // vpshufb       0x191f(%rip),%ymm3,%ymm1        # 4c40 <_sk_callback_hsw+0x5ec>
+  .byte  196,226,101,0,13,35,25,0,0          // vpshufb       0x1923(%rip),%ymm3,%ymm1        # 4c60 <_sk_callback_hsw+0x5f0>
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,226,101,0,21,45,25,0,0          // vpshufb       0x192d(%rip),%ymm3,%ymm2        # 4c60 <_sk_callback_hsw+0x60c>
+  .byte  196,226,101,0,21,49,25,0,0          // vpshufb       0x1931(%rip),%ymm3,%ymm2        # 4c80 <_sk_callback_hsw+0x610>
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
@@ -11923,7 +11947,7 @@ _sk_load_8888_hsw:
   .byte  196,225,249,110,192                 // vmovq         %rax,%xmm0
   .byte  196,226,125,33,192                  // vpmovsxbd     %xmm0,%ymm0
   .byte  196,194,125,140,25                  // vpmaskmovd    (%r9),%ymm0,%ymm3
-  .byte  235,135                             // jmp           32fe <_sk_load_8888_hsw+0x1a>
+  .byte  235,135                             // jmp           331a <_sk_load_8888_hsw+0x1a>
 
 HIDDEN _sk_gather_8888_hsw
 .globl _sk_gather_8888_hsw
@@ -11938,14 +11962,14 @@ _sk_gather_8888_hsw:
   .byte  197,245,254,192                     // vpaddd        %ymm0,%ymm1,%ymm0
   .byte  197,245,118,201                     // vpcmpeqd      %ymm1,%ymm1,%ymm1
   .byte  196,194,117,144,28,128              // vpgatherdd    %ymm1,(%r8,%ymm0,4),%ymm3
-  .byte  197,229,219,5,219,24,0,0            // vpand         0x18db(%rip),%ymm3,%ymm0        # 4c80 <_sk_callback_hsw+0x62c>
+  .byte  197,229,219,5,223,24,0,0            // vpand         0x18df(%rip),%ymm3,%ymm0        # 4ca0 <_sk_callback_hsw+0x630>
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,194,22,0,0          // vbroadcastss  0x16c2(%rip),%ymm8        # 4a74 <_sk_callback_hsw+0x420>
+  .byte  196,98,125,24,5,194,22,0,0          // vbroadcastss  0x16c2(%rip),%ymm8        # 4a90 <_sk_callback_hsw+0x420>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,226,101,0,13,224,24,0,0         // vpshufb       0x18e0(%rip),%ymm3,%ymm1        # 4ca0 <_sk_callback_hsw+0x64c>
+  .byte  196,226,101,0,13,228,24,0,0         // vpshufb       0x18e4(%rip),%ymm3,%ymm1        # 4cc0 <_sk_callback_hsw+0x650>
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,226,101,0,21,238,24,0,0         // vpshufb       0x18ee(%rip),%ymm3,%ymm2        # 4cc0 <_sk_callback_hsw+0x66c>
+  .byte  196,226,101,0,21,242,24,0,0         // vpshufb       0x18f2(%rip),%ymm3,%ymm2        # 4ce0 <_sk_callback_hsw+0x670>
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,229,114,211,24                  // vpsrld        $0x18,%ymm3,%ymm3
@@ -11962,7 +11986,7 @@ _sk_store_8888_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  76,3,8                              // add           (%rax),%r9
-  .byte  196,98,125,24,5,114,22,0,0          // vbroadcastss  0x1672(%rip),%ymm8        # 4a78 <_sk_callback_hsw+0x424>
+  .byte  196,98,125,24,5,114,22,0,0          // vbroadcastss  0x1672(%rip),%ymm8        # 4a94 <_sk_callback_hsw+0x424>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,65,116,89,208                   // vmulps        %ymm8,%ymm1,%ymm10
@@ -11978,7 +12002,7 @@ _sk_store_8888_hsw:
   .byte  196,65,45,235,192                   // vpor          %ymm8,%ymm10,%ymm8
   .byte  196,65,53,235,192                   // vpor          %ymm8,%ymm9,%ymm8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,12                              // jne           3460 <_sk_store_8888_hsw+0x73>
+  .byte  117,12                              // jne           347c <_sk_store_8888_hsw+0x73>
   .byte  196,65,126,127,1                    // vmovdqu       %ymm8,(%r9)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,137,193                          // mov           %r8,%rcx
@@ -11991,7 +12015,7 @@ _sk_store_8888_hsw:
   .byte  196,97,249,110,200                  // vmovq         %rax,%xmm9
   .byte  196,66,125,33,201                   // vpmovsxbd     %xmm9,%ymm9
   .byte  196,66,53,142,1                     // vpmaskmovd    %ymm8,%ymm9,(%r9)
-  .byte  235,211                             // jmp           3459 <_sk_store_8888_hsw+0x6c>
+  .byte  235,211                             // jmp           3475 <_sk_store_8888_hsw+0x6c>
 
 HIDDEN _sk_load_f16_hsw
 .globl _sk_load_f16_hsw
@@ -12000,7 +12024,7 @@ _sk_load_f16_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,97                              // jne           34f1 <_sk_load_f16_hsw+0x6b>
+  .byte  117,97                              // jne           350d <_sk_load_f16_hsw+0x6b>
   .byte  197,121,16,4,248                    // vmovupd       (%rax,%rdi,8),%xmm8
   .byte  197,249,16,84,248,16                // vmovupd       0x10(%rax,%rdi,8),%xmm2
   .byte  197,249,16,92,248,32                // vmovupd       0x20(%rax,%rdi,8),%xmm3
@@ -12026,29 +12050,29 @@ _sk_load_f16_hsw:
   .byte  197,123,16,4,248                    // vmovsd        (%rax,%rdi,8),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,79                              // je            3550 <_sk_load_f16_hsw+0xca>
+  .byte  116,79                              // je            356c <_sk_load_f16_hsw+0xca>
   .byte  197,57,22,68,248,8                  // vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,67                              // jb            3550 <_sk_load_f16_hsw+0xca>
+  .byte  114,67                              // jb            356c <_sk_load_f16_hsw+0xca>
   .byte  197,251,16,84,248,16                // vmovsd        0x10(%rax,%rdi,8),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,68                              // je            355d <_sk_load_f16_hsw+0xd7>
+  .byte  116,68                              // je            3579 <_sk_load_f16_hsw+0xd7>
   .byte  197,233,22,84,248,24                // vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,56                              // jb            355d <_sk_load_f16_hsw+0xd7>
+  .byte  114,56                              // jb            3579 <_sk_load_f16_hsw+0xd7>
   .byte  197,251,16,92,248,32                // vmovsd        0x20(%rax,%rdi,8),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,114,255,255,255              // je            34a7 <_sk_load_f16_hsw+0x21>
+  .byte  15,132,114,255,255,255              // je            34c3 <_sk_load_f16_hsw+0x21>
   .byte  197,225,22,92,248,40                // vmovhpd       0x28(%rax,%rdi,8),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,98,255,255,255               // jb            34a7 <_sk_load_f16_hsw+0x21>
+  .byte  15,130,98,255,255,255               // jb            34c3 <_sk_load_f16_hsw+0x21>
   .byte  197,122,126,76,248,48               // vmovq         0x30(%rax,%rdi,8),%xmm9
-  .byte  233,87,255,255,255                  // jmpq          34a7 <_sk_load_f16_hsw+0x21>
+  .byte  233,87,255,255,255                  // jmpq          34c3 <_sk_load_f16_hsw+0x21>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,74,255,255,255                  // jmpq          34a7 <_sk_load_f16_hsw+0x21>
+  .byte  233,74,255,255,255                  // jmpq          34c3 <_sk_load_f16_hsw+0x21>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,65,255,255,255                  // jmpq          34a7 <_sk_load_f16_hsw+0x21>
+  .byte  233,65,255,255,255                  // jmpq          34c3 <_sk_load_f16_hsw+0x21>
 
 HIDDEN _sk_gather_f16_hsw
 .globl _sk_gather_f16_hsw
@@ -12106,7 +12130,7 @@ _sk_store_f16_hsw:
   .byte  196,65,57,98,205                    // vpunpckldq    %xmm13,%xmm8,%xmm9
   .byte  196,65,57,106,197                   // vpunpckhdq    %xmm13,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,27                              // jne           3655 <_sk_store_f16_hsw+0x65>
+  .byte  117,27                              // jne           3671 <_sk_store_f16_hsw+0x65>
   .byte  197,120,17,28,248                   // vmovups       %xmm11,(%rax,%rdi,8)
   .byte  197,120,17,84,248,16                // vmovups       %xmm10,0x10(%rax,%rdi,8)
   .byte  197,120,17,76,248,32                // vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -12115,22 +12139,22 @@ _sk_store_f16_hsw:
   .byte  255,224                             // jmpq          *%rax
   .byte  197,121,214,28,248                  // vmovq         %xmm11,(%rax,%rdi,8)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,241                             // je            3651 <_sk_store_f16_hsw+0x61>
+  .byte  116,241                             // je            366d <_sk_store_f16_hsw+0x61>
   .byte  197,121,23,92,248,8                 // vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,229                             // jb            3651 <_sk_store_f16_hsw+0x61>
+  .byte  114,229                             // jb            366d <_sk_store_f16_hsw+0x61>
   .byte  197,121,214,84,248,16               // vmovq         %xmm10,0x10(%rax,%rdi,8)
-  .byte  116,221                             // je            3651 <_sk_store_f16_hsw+0x61>
+  .byte  116,221                             // je            366d <_sk_store_f16_hsw+0x61>
   .byte  197,121,23,84,248,24                // vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,209                             // jb            3651 <_sk_store_f16_hsw+0x61>
+  .byte  114,209                             // jb            366d <_sk_store_f16_hsw+0x61>
   .byte  197,121,214,76,248,32               // vmovq         %xmm9,0x20(%rax,%rdi,8)
-  .byte  116,201                             // je            3651 <_sk_store_f16_hsw+0x61>
+  .byte  116,201                             // je            366d <_sk_store_f16_hsw+0x61>
   .byte  197,121,23,76,248,40                // vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,189                             // jb            3651 <_sk_store_f16_hsw+0x61>
+  .byte  114,189                             // jb            366d <_sk_store_f16_hsw+0x61>
   .byte  197,121,214,68,248,48               // vmovq         %xmm8,0x30(%rax,%rdi,8)
-  .byte  235,181                             // jmp           3651 <_sk_store_f16_hsw+0x61>
+  .byte  235,181                             // jmp           366d <_sk_store_f16_hsw+0x61>
 
 HIDDEN _sk_load_u16_be_hsw
 .globl _sk_load_u16_be_hsw
@@ -12140,7 +12164,7 @@ _sk_load_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,204,0,0,0                    // jne           377e <_sk_load_u16_be_hsw+0xe2>
+  .byte  15,133,204,0,0,0                    // jne           379a <_sk_load_u16_be_hsw+0xe2>
   .byte  196,65,121,16,4,64                  // vmovupd       (%r8,%rax,2),%xmm8
   .byte  196,193,121,16,84,64,16             // vmovupd       0x10(%r8,%rax,2),%xmm2
   .byte  196,193,121,16,92,64,32             // vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -12159,7 +12183,7 @@ _sk_load_u16_be_hsw:
   .byte  197,241,235,192                     // vpor          %xmm0,%xmm1,%xmm0
   .byte  196,226,125,51,192                  // vpmovzxwd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,21,105,19,0,0         // vbroadcastss  0x1369(%rip),%ymm10        # 4a7c <_sk_callback_hsw+0x428>
+  .byte  196,98,125,24,21,105,19,0,0         // vbroadcastss  0x1369(%rip),%ymm10        # 4a98 <_sk_callback_hsw+0x428>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,185,109,202                     // vpunpckhqdq   %xmm2,%xmm8,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -12187,29 +12211,29 @@ _sk_load_u16_be_hsw:
   .byte  196,65,123,16,4,64                  // vmovsd        (%r8,%rax,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            37e4 <_sk_load_u16_be_hsw+0x148>
+  .byte  116,85                              // je            3800 <_sk_load_u16_be_hsw+0x148>
   .byte  196,65,57,22,68,64,8                // vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            37e4 <_sk_load_u16_be_hsw+0x148>
+  .byte  114,72                              // jb            3800 <_sk_load_u16_be_hsw+0x148>
   .byte  196,193,123,16,84,64,16             // vmovsd        0x10(%r8,%rax,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            37f1 <_sk_load_u16_be_hsw+0x155>
+  .byte  116,72                              // je            380d <_sk_load_u16_be_hsw+0x155>
   .byte  196,193,105,22,84,64,24             // vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            37f1 <_sk_load_u16_be_hsw+0x155>
+  .byte  114,59                              // jb            380d <_sk_load_u16_be_hsw+0x155>
   .byte  196,193,123,16,92,64,32             // vmovsd        0x20(%r8,%rax,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,6,255,255,255                // je            36cd <_sk_load_u16_be_hsw+0x31>
+  .byte  15,132,6,255,255,255                // je            36e9 <_sk_load_u16_be_hsw+0x31>
   .byte  196,193,97,22,92,64,40              // vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,245,254,255,255              // jb            36cd <_sk_load_u16_be_hsw+0x31>
+  .byte  15,130,245,254,255,255              // jb            36e9 <_sk_load_u16_be_hsw+0x31>
   .byte  196,65,122,126,76,64,48             // vmovq         0x30(%r8,%rax,2),%xmm9
-  .byte  233,233,254,255,255                 // jmpq          36cd <_sk_load_u16_be_hsw+0x31>
+  .byte  233,233,254,255,255                 // jmpq          36e9 <_sk_load_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,220,254,255,255                 // jmpq          36cd <_sk_load_u16_be_hsw+0x31>
+  .byte  233,220,254,255,255                 // jmpq          36e9 <_sk_load_u16_be_hsw+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,211,254,255,255                 // jmpq          36cd <_sk_load_u16_be_hsw+0x31>
+  .byte  233,211,254,255,255                 // jmpq          36e9 <_sk_load_u16_be_hsw+0x31>
 
 HIDDEN _sk_load_rgb_u16_be_hsw
 .globl _sk_load_rgb_u16_be_hsw
@@ -12219,7 +12243,7 @@ _sk_load_rgb_u16_be_hsw:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,127                        // lea           (%rdi,%rdi,2),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,204,0,0,0                    // jne           38d8 <_sk_load_rgb_u16_be_hsw+0xde>
+  .byte  15,133,204,0,0,0                    // jne           38f4 <_sk_load_rgb_u16_be_hsw+0xde>
   .byte  196,193,122,111,4,64                // vmovdqu       (%r8,%rax,2),%xmm0
   .byte  196,193,122,111,84,64,12            // vmovdqu       0xc(%r8,%rax,2),%xmm2
   .byte  196,193,122,111,76,64,24            // vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -12243,7 +12267,7 @@ _sk_load_rgb_u16_be_hsw:
   .byte  197,241,235,192                     // vpor          %xmm0,%xmm1,%xmm0
   .byte  196,226,125,51,192                  // vpmovzxwd     %xmm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,21,250,17,0,0         // vbroadcastss  0x11fa(%rip),%ymm10        # 4a80 <_sk_callback_hsw+0x42c>
+  .byte  196,98,125,24,21,250,17,0,0         // vbroadcastss  0x11fa(%rip),%ymm10        # 4a9c <_sk_callback_hsw+0x42c>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,185,109,202                     // vpunpckhqdq   %xmm2,%xmm8,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -12260,41 +12284,41 @@ _sk_load_rgb_u16_be_hsw:
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,174,17,0,0        // vbroadcastss  0x11ae(%rip),%ymm3        # 4a84 <_sk_callback_hsw+0x430>
+  .byte  196,226,125,24,29,174,17,0,0        // vbroadcastss  0x11ae(%rip),%ymm3        # 4aa0 <_sk_callback_hsw+0x430>
   .byte  255,224                             // jmpq          *%rax
   .byte  196,193,121,110,4,64                // vmovd         (%r8,%rax,2),%xmm0
   .byte  196,193,121,196,68,64,4,2           // vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           38f1 <_sk_load_rgb_u16_be_hsw+0xf7>
-  .byte  233,79,255,255,255                  // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  117,5                               // jne           390d <_sk_load_rgb_u16_be_hsw+0xf7>
+  .byte  233,79,255,255,255                  // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
   .byte  196,193,121,110,76,64,6             // vmovd         0x6(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,68,64,10,2           // vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            3920 <_sk_load_rgb_u16_be_hsw+0x126>
+  .byte  114,26                              // jb            393c <_sk_load_rgb_u16_be_hsw+0x126>
   .byte  196,193,121,110,76,64,12            // vmovd         0xc(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,84,64,16,2          // vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           3925 <_sk_load_rgb_u16_be_hsw+0x12b>
-  .byte  233,32,255,255,255                  // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
-  .byte  233,27,255,255,255                  // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           3941 <_sk_load_rgb_u16_be_hsw+0x12b>
+  .byte  233,32,255,255,255                  // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,27,255,255,255                  // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
   .byte  196,193,121,110,76,64,18            // vmovd         0x12(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,76,64,22,2           // vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            3954 <_sk_load_rgb_u16_be_hsw+0x15a>
+  .byte  114,26                              // jb            3970 <_sk_load_rgb_u16_be_hsw+0x15a>
   .byte  196,193,121,110,76,64,24            // vmovd         0x18(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,76,64,28,2          // vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           3959 <_sk_load_rgb_u16_be_hsw+0x15f>
-  .byte  233,236,254,255,255                 // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
-  .byte  233,231,254,255,255                 // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  117,10                              // jne           3975 <_sk_load_rgb_u16_be_hsw+0x15f>
+  .byte  233,236,254,255,255                 // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,231,254,255,255                 // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
   .byte  196,193,121,110,92,64,30            // vmovd         0x1e(%r8,%rax,2),%xmm3
   .byte  196,65,97,196,92,64,34,2            // vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            3982 <_sk_load_rgb_u16_be_hsw+0x188>
+  .byte  114,20                              // jb            399e <_sk_load_rgb_u16_be_hsw+0x188>
   .byte  196,193,121,110,92,64,36            // vmovd         0x24(%r8,%rax,2),%xmm3
   .byte  196,193,97,196,92,64,40,2           // vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  .byte  233,190,254,255,255                 // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
-  .byte  233,185,254,255,255                 // jmpq          3840 <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,190,254,255,255                 // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
+  .byte  233,185,254,255,255                 // jmpq          385c <_sk_load_rgb_u16_be_hsw+0x46>
 
 HIDDEN _sk_store_u16_be_hsw
 .globl _sk_store_u16_be_hsw
@@ -12303,7 +12327,7 @@ _sk_store_u16_be_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
-  .byte  196,98,125,24,5,235,16,0,0          // vbroadcastss  0x10eb(%rip),%ymm8        # 4a88 <_sk_callback_hsw+0x434>
+  .byte  196,98,125,24,5,235,16,0,0          // vbroadcastss  0x10eb(%rip),%ymm8        # 4aa4 <_sk_callback_hsw+0x434>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,67,125,25,202,1                 // vextractf128  $0x1,%ymm9,%xmm10
@@ -12341,7 +12365,7 @@ _sk_store_u16_be_hsw:
   .byte  196,65,17,98,200                    // vpunpckldq    %xmm8,%xmm13,%xmm9
   .byte  196,65,17,106,192                   // vpunpckhdq    %xmm8,%xmm13,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,31                              // jne           3a81 <_sk_store_u16_be_hsw+0xfa>
+  .byte  117,31                              // jne           3a9d <_sk_store_u16_be_hsw+0xfa>
   .byte  196,65,120,17,28,64                 // vmovups       %xmm11,(%r8,%rax,2)
   .byte  196,65,120,17,84,64,16              // vmovups       %xmm10,0x10(%r8,%rax,2)
   .byte  196,65,120,17,76,64,32              // vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -12350,22 +12374,22 @@ _sk_store_u16_be_hsw:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,214,28,64                // vmovq         %xmm11,(%r8,%rax,2)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  116,240                             // je            3a99 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,23,92,64,8               // vmovhpd       %xmm11,0x8(%r8,%rax,2)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  114,227                             // jb            3a99 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,214,84,64,16             // vmovq         %xmm10,0x10(%r8,%rax,2)
-  .byte  116,218                             // je            3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  116,218                             // je            3a99 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,23,84,64,24              // vmovhpd       %xmm10,0x18(%r8,%rax,2)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  114,205                             // jb            3a99 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,214,76,64,32             // vmovq         %xmm9,0x20(%r8,%rax,2)
-  .byte  116,196                             // je            3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  116,196                             // je            3a99 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,23,76,64,40              // vmovhpd       %xmm9,0x28(%r8,%rax,2)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,183                             // jb            3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  114,183                             // jb            3a99 <_sk_store_u16_be_hsw+0xf6>
   .byte  196,65,121,214,68,64,48             // vmovq         %xmm8,0x30(%r8,%rax,2)
-  .byte  235,174                             // jmp           3a7d <_sk_store_u16_be_hsw+0xf6>
+  .byte  235,174                             // jmp           3a99 <_sk_store_u16_be_hsw+0xf6>
 
 HIDDEN _sk_load_f32_hsw
 .globl _sk_load_f32_hsw
@@ -12373,10 +12397,10 @@ FUNCTION(_sk_load_f32_hsw)
 _sk_load_f32_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  119,110                             // ja            3b45 <_sk_load_f32_hsw+0x76>
+  .byte  119,110                             // ja            3b61 <_sk_load_f32_hsw+0x76>
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
-  .byte  76,141,21,135,0,0,0                 // lea           0x87(%rip),%r10        # 3b70 <_sk_load_f32_hsw+0xa1>
+  .byte  76,141,21,135,0,0,0                 // lea           0x87(%rip),%r10        # 3b8c <_sk_load_f32_hsw+0xa1>
   .byte  73,99,4,138                         // movslq        (%r10,%rcx,4),%rax
   .byte  76,1,208                            // add           %r10,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -12437,7 +12461,7 @@ _sk_store_f32_hsw:
   .byte  196,65,37,20,196                    // vunpcklpd     %ymm12,%ymm11,%ymm8
   .byte  196,65,37,21,220                    // vunpckhpd     %ymm12,%ymm11,%ymm11
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,55                              // jne           3bfd <_sk_store_f32_hsw+0x6d>
+  .byte  117,55                              // jne           3c19 <_sk_store_f32_hsw+0x6d>
   .byte  196,67,45,24,225,1                  // vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   .byte  196,67,61,24,235,1                  // vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   .byte  196,67,45,6,201,49                  // vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -12450,22 +12474,22 @@ _sk_store_f32_hsw:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,17,20,128                // vmovupd       %xmm10,(%r8,%rax,4)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  116,240                             // je            3c15 <_sk_store_f32_hsw+0x69>
   .byte  196,65,121,17,76,128,16             // vmovupd       %xmm9,0x10(%r8,%rax,4)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  114,227                             // jb            3c15 <_sk_store_f32_hsw+0x69>
   .byte  196,65,121,17,68,128,32             // vmovupd       %xmm8,0x20(%r8,%rax,4)
-  .byte  116,218                             // je            3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  116,218                             // je            3c15 <_sk_store_f32_hsw+0x69>
   .byte  196,65,121,17,92,128,48             // vmovupd       %xmm11,0x30(%r8,%rax,4)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  114,205                             // jb            3c15 <_sk_store_f32_hsw+0x69>
   .byte  196,67,125,25,84,128,64,1           // vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  .byte  116,195                             // je            3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  116,195                             // je            3c15 <_sk_store_f32_hsw+0x69>
   .byte  196,67,125,25,76,128,80,1           // vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,181                             // jb            3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  114,181                             // jb            3c15 <_sk_store_f32_hsw+0x69>
   .byte  196,67,125,25,68,128,96,1           // vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  .byte  235,171                             // jmp           3bf9 <_sk_store_f32_hsw+0x69>
+  .byte  235,171                             // jmp           3c15 <_sk_store_f32_hsw+0x69>
 
 HIDDEN _sk_clamp_x_hsw
 .globl _sk_clamp_x_hsw
@@ -12563,11 +12587,11 @@ HIDDEN _sk_luminance_to_alpha_hsw
 .globl _sk_luminance_to_alpha_hsw
 FUNCTION(_sk_luminance_to_alpha_hsw)
 _sk_luminance_to_alpha_hsw:
-  .byte  196,226,125,24,29,59,13,0,0         // vbroadcastss  0xd3b(%rip),%ymm3        # 4a8c <_sk_callback_hsw+0x438>
-  .byte  196,98,125,24,5,54,13,0,0           // vbroadcastss  0xd36(%rip),%ymm8        # 4a90 <_sk_callback_hsw+0x43c>
+  .byte  196,226,125,24,29,59,13,0,0         // vbroadcastss  0xd3b(%rip),%ymm3        # 4aa8 <_sk_callback_hsw+0x438>
+  .byte  196,98,125,24,5,54,13,0,0           // vbroadcastss  0xd36(%rip),%ymm8        # 4aac <_sk_callback_hsw+0x43c>
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  196,226,125,184,203                 // vfmadd231ps   %ymm3,%ymm0,%ymm1
-  .byte  196,226,125,24,29,39,13,0,0         // vbroadcastss  0xd27(%rip),%ymm3        # 4a94 <_sk_callback_hsw+0x440>
+  .byte  196,226,125,24,29,39,13,0,0         // vbroadcastss  0xd27(%rip),%ymm3        # 4ab0 <_sk_callback_hsw+0x440>
   .byte  196,226,109,168,217                 // vfmadd213ps   %ymm1,%ymm2,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -12710,9 +12734,9 @@ _sk_evenly_spaced_gradient_hsw:
   .byte  76,139,64,8                         // mov           0x8(%rax),%r8
   .byte  77,137,202                          // mov           %r9,%r10
   .byte  73,255,202                          // dec           %r10
-  .byte  120,7                               // js            3fa8 <_sk_evenly_spaced_gradient_hsw+0x18>
+  .byte  120,7                               // js            3fc4 <_sk_evenly_spaced_gradient_hsw+0x18>
   .byte  196,193,242,42,202                  // vcvtsi2ss     %r10,%xmm1,%xmm1
-  .byte  235,22                              // jmp           3fbe <_sk_evenly_spaced_gradient_hsw+0x2e>
+  .byte  235,22                              // jmp           3fda <_sk_evenly_spaced_gradient_hsw+0x2e>
   .byte  77,137,211                          // mov           %r10,%r11
   .byte  73,209,235                          // shr           %r11
   .byte  65,131,226,1                        // and           $0x1,%r10d
@@ -12723,7 +12747,7 @@ _sk_evenly_spaced_gradient_hsw:
   .byte  197,244,89,200                      // vmulps        %ymm0,%ymm1,%ymm1
   .byte  197,126,91,217                      // vcvttps2dq    %ymm1,%ymm11
   .byte  73,131,249,8                        // cmp           $0x8,%r9
-  .byte  119,70                              // ja            4017 <_sk_evenly_spaced_gradient_hsw+0x87>
+  .byte  119,70                              // ja            4033 <_sk_evenly_spaced_gradient_hsw+0x87>
   .byte  196,66,37,22,0                      // vpermps       (%r8),%ymm11,%ymm8
   .byte  76,139,64,40                        // mov           0x28(%rax),%r8
   .byte  196,66,37,22,8                      // vpermps       (%r8),%ymm11,%ymm9
@@ -12739,7 +12763,7 @@ _sk_evenly_spaced_gradient_hsw:
   .byte  196,194,37,22,24                    // vpermps       (%r8),%ymm11,%ymm3
   .byte  72,139,64,64                        // mov           0x40(%rax),%rax
   .byte  196,98,37,22,40                     // vpermps       (%rax),%ymm11,%ymm13
-  .byte  235,110                             // jmp           4085 <_sk_evenly_spaced_gradient_hsw+0xf5>
+  .byte  235,110                             // jmp           40a1 <_sk_evenly_spaced_gradient_hsw+0xf5>
   .byte  196,65,13,118,246                   // vpcmpeqd      %ymm14,%ymm14,%ymm14
   .byte  197,245,118,201                     // vpcmpeqd      %ymm1,%ymm1,%ymm1
   .byte  196,2,117,146,4,152                 // vgatherdps    %ymm1,(%r8,%ymm11,4),%ymm8
@@ -12778,11 +12802,11 @@ _sk_gradient_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  73,131,248,1                        // cmp           $0x1,%r8
-  .byte  15,134,180,0,0,0                    // jbe           4164 <_sk_gradient_hsw+0xc3>
+  .byte  15,134,180,0,0,0                    // jbe           4180 <_sk_gradient_hsw+0xc3>
   .byte  76,139,72,72                        // mov           0x48(%rax),%r9
   .byte  197,244,87,201                      // vxorps        %ymm1,%ymm1,%ymm1
   .byte  65,186,1,0,0,0                      // mov           $0x1,%r10d
-  .byte  196,226,125,24,21,209,9,0,0         // vbroadcastss  0x9d1(%rip),%ymm2        # 4a98 <_sk_callback_hsw+0x444>
+  .byte  196,226,125,24,21,209,9,0,0         // vbroadcastss  0x9d1(%rip),%ymm2        # 4ab4 <_sk_callback_hsw+0x444>
   .byte  196,65,53,239,201                   // vpxor         %ymm9,%ymm9,%ymm9
   .byte  196,130,125,24,28,145               // vbroadcastss  (%r9,%r10,4),%ymm3
   .byte  197,228,194,216,2                   // vcmpleps      %ymm0,%ymm3,%ymm3
@@ -12790,10 +12814,10 @@ _sk_gradient_hsw:
   .byte  196,65,101,254,201                  // vpaddd        %ymm9,%ymm3,%ymm9
   .byte  73,255,194                          // inc           %r10
   .byte  77,57,208                           // cmp           %r10,%r8
-  .byte  117,226                             // jne           40cc <_sk_gradient_hsw+0x2b>
+  .byte  117,226                             // jne           40e8 <_sk_gradient_hsw+0x2b>
   .byte  76,139,72,8                         // mov           0x8(%rax),%r9
   .byte  73,131,248,8                        // cmp           $0x8,%r8
-  .byte  118,121                             // jbe           416d <_sk_gradient_hsw+0xcc>
+  .byte  118,121                             // jbe           4189 <_sk_gradient_hsw+0xcc>
   .byte  196,65,13,118,246                   // vpcmpeqd      %ymm14,%ymm14,%ymm14
   .byte  197,245,118,201                     // vpcmpeqd      %ymm1,%ymm1,%ymm1
   .byte  196,2,117,146,4,137                 // vgatherdps    %ymm1,(%r9,%ymm9,4),%ymm8
@@ -12817,7 +12841,7 @@ _sk_gradient_hsw:
   .byte  196,130,21,146,28,136               // vgatherdps    %ymm13,(%r8,%ymm9,4),%ymm3
   .byte  72,139,64,64                        // mov           0x40(%rax),%rax
   .byte  196,34,13,146,44,136                // vgatherdps    %ymm14,(%rax,%ymm9,4),%ymm13
-  .byte  235,77                              // jmp           41b1 <_sk_gradient_hsw+0x110>
+  .byte  235,77                              // jmp           41cd <_sk_gradient_hsw+0x110>
   .byte  76,139,72,8                         // mov           0x8(%rax),%r9
   .byte  196,65,52,87,201                    // vxorps        %ymm9,%ymm9,%ymm9
   .byte  196,66,53,22,1                      // vpermps       (%r9),%ymm9,%ymm8
@@ -12877,24 +12901,24 @@ _sk_xy_to_unit_angle_hsw:
   .byte  196,65,52,95,226                    // vmaxps        %ymm10,%ymm9,%ymm12
   .byte  196,65,36,94,220                    // vdivps        %ymm12,%ymm11,%ymm11
   .byte  196,65,36,89,227                    // vmulps        %ymm11,%ymm11,%ymm12
-  .byte  196,98,125,24,45,80,8,0,0           // vbroadcastss  0x850(%rip),%ymm13        # 4a9c <_sk_callback_hsw+0x448>
-  .byte  196,98,125,24,53,75,8,0,0           // vbroadcastss  0x84b(%rip),%ymm14        # 4aa0 <_sk_callback_hsw+0x44c>
+  .byte  196,98,125,24,45,80,8,0,0           // vbroadcastss  0x850(%rip),%ymm13        # 4ab8 <_sk_callback_hsw+0x448>
+  .byte  196,98,125,24,53,75,8,0,0           // vbroadcastss  0x84b(%rip),%ymm14        # 4abc <_sk_callback_hsw+0x44c>
   .byte  196,66,29,184,245                   // vfmadd231ps   %ymm13,%ymm12,%ymm14
-  .byte  196,98,125,24,45,65,8,0,0           // vbroadcastss  0x841(%rip),%ymm13        # 4aa4 <_sk_callback_hsw+0x450>
+  .byte  196,98,125,24,45,65,8,0,0           // vbroadcastss  0x841(%rip),%ymm13        # 4ac0 <_sk_callback_hsw+0x450>
   .byte  196,66,29,184,238                   // vfmadd231ps   %ymm14,%ymm12,%ymm13
-  .byte  196,98,125,24,53,55,8,0,0           // vbroadcastss  0x837(%rip),%ymm14        # 4aa8 <_sk_callback_hsw+0x454>
+  .byte  196,98,125,24,53,55,8,0,0           // vbroadcastss  0x837(%rip),%ymm14        # 4ac4 <_sk_callback_hsw+0x454>
   .byte  196,66,29,184,245                   // vfmadd231ps   %ymm13,%ymm12,%ymm14
   .byte  196,65,36,89,222                    // vmulps        %ymm14,%ymm11,%ymm11
   .byte  196,65,52,194,202,1                 // vcmpltps      %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,21,34,8,0,0           // vbroadcastss  0x822(%rip),%ymm10        # 4aac <_sk_callback_hsw+0x458>
+  .byte  196,98,125,24,21,34,8,0,0           // vbroadcastss  0x822(%rip),%ymm10        # 4ac8 <_sk_callback_hsw+0x458>
   .byte  196,65,44,92,211                    // vsubps        %ymm11,%ymm10,%ymm10
   .byte  196,67,37,74,202,144                // vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   .byte  196,193,124,194,192,1               // vcmpltps      %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,21,12,8,0,0           // vbroadcastss  0x80c(%rip),%ymm10        # 4ab0 <_sk_callback_hsw+0x45c>
+  .byte  196,98,125,24,21,12,8,0,0           // vbroadcastss  0x80c(%rip),%ymm10        # 4acc <_sk_callback_hsw+0x45c>
   .byte  196,65,44,92,209                    // vsubps        %ymm9,%ymm10,%ymm10
   .byte  196,195,53,74,194,0                 // vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   .byte  196,65,116,194,200,1                // vcmpltps      %ymm8,%ymm1,%ymm9
-  .byte  196,98,125,24,21,246,7,0,0          // vbroadcastss  0x7f6(%rip),%ymm10        # 4ab4 <_sk_callback_hsw+0x460>
+  .byte  196,98,125,24,21,246,7,0,0          // vbroadcastss  0x7f6(%rip),%ymm10        # 4ad0 <_sk_callback_hsw+0x460>
   .byte  197,44,92,208                       // vsubps        %ymm0,%ymm10,%ymm10
   .byte  196,195,125,74,194,144              // vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   .byte  196,65,124,194,200,3                // vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -12917,7 +12941,7 @@ HIDDEN _sk_save_xy_hsw
 FUNCTION(_sk_save_xy_hsw)
 _sk_save_xy_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,195,7,0,0           // vbroadcastss  0x7c3(%rip),%ymm8        # 4ab8 <_sk_callback_hsw+0x464>
+  .byte  196,98,125,24,5,195,7,0,0           // vbroadcastss  0x7c3(%rip),%ymm8        # 4ad4 <_sk_callback_hsw+0x464>
   .byte  196,65,124,88,200                   // vaddps        %ymm8,%ymm0,%ymm9
   .byte  196,67,125,8,209,1                  // vroundps      $0x1,%ymm9,%ymm10
   .byte  196,65,52,92,202                    // vsubps        %ymm10,%ymm9,%ymm9
@@ -12951,9 +12975,9 @@ HIDDEN _sk_bilinear_nx_hsw
 FUNCTION(_sk_bilinear_nx_hsw)
 _sk_bilinear_nx_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,87,7,0,0           // vbroadcastss  0x757(%rip),%ymm0        # 4abc <_sk_callback_hsw+0x468>
+  .byte  196,226,125,24,5,87,7,0,0           // vbroadcastss  0x757(%rip),%ymm0        # 4ad8 <_sk_callback_hsw+0x468>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,78,7,0,0            // vbroadcastss  0x74e(%rip),%ymm8        # 4ac0 <_sk_callback_hsw+0x46c>
+  .byte  196,98,125,24,5,78,7,0,0            // vbroadcastss  0x74e(%rip),%ymm8        # 4adc <_sk_callback_hsw+0x46c>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12964,7 +12988,7 @@ HIDDEN _sk_bilinear_px_hsw
 FUNCTION(_sk_bilinear_px_hsw)
 _sk_bilinear_px_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,54,7,0,0           // vbroadcastss  0x736(%rip),%ymm0        # 4ac4 <_sk_callback_hsw+0x470>
+  .byte  196,226,125,24,5,54,7,0,0           // vbroadcastss  0x736(%rip),%ymm0        # 4ae0 <_sk_callback_hsw+0x470>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -12976,9 +13000,9 @@ HIDDEN _sk_bilinear_ny_hsw
 FUNCTION(_sk_bilinear_ny_hsw)
 _sk_bilinear_ny_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,26,7,0,0          // vbroadcastss  0x71a(%rip),%ymm1        # 4ac8 <_sk_callback_hsw+0x474>
+  .byte  196,226,125,24,13,26,7,0,0          // vbroadcastss  0x71a(%rip),%ymm1        # 4ae4 <_sk_callback_hsw+0x474>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,16,7,0,0            // vbroadcastss  0x710(%rip),%ymm8        # 4acc <_sk_callback_hsw+0x478>
+  .byte  196,98,125,24,5,16,7,0,0            // vbroadcastss  0x710(%rip),%ymm8        # 4ae8 <_sk_callback_hsw+0x478>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -12989,7 +13013,7 @@ HIDDEN _sk_bilinear_py_hsw
 FUNCTION(_sk_bilinear_py_hsw)
 _sk_bilinear_py_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,248,6,0,0         // vbroadcastss  0x6f8(%rip),%ymm1        # 4ad0 <_sk_callback_hsw+0x47c>
+  .byte  196,226,125,24,13,248,6,0,0         // vbroadcastss  0x6f8(%rip),%ymm1        # 4aec <_sk_callback_hsw+0x47c>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -13001,13 +13025,13 @@ HIDDEN _sk_bicubic_n3x_hsw
 FUNCTION(_sk_bicubic_n3x_hsw)
 _sk_bicubic_n3x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,219,6,0,0          // vbroadcastss  0x6db(%rip),%ymm0        # 4ad4 <_sk_callback_hsw+0x480>
+  .byte  196,226,125,24,5,219,6,0,0          // vbroadcastss  0x6db(%rip),%ymm0        # 4af0 <_sk_callback_hsw+0x480>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,210,6,0,0           // vbroadcastss  0x6d2(%rip),%ymm8        # 4ad8 <_sk_callback_hsw+0x484>
+  .byte  196,98,125,24,5,210,6,0,0           // vbroadcastss  0x6d2(%rip),%ymm8        # 4af4 <_sk_callback_hsw+0x484>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,195,6,0,0          // vbroadcastss  0x6c3(%rip),%ymm10        # 4adc <_sk_callback_hsw+0x488>
-  .byte  196,98,125,24,29,190,6,0,0          // vbroadcastss  0x6be(%rip),%ymm11        # 4ae0 <_sk_callback_hsw+0x48c>
+  .byte  196,98,125,24,21,195,6,0,0          // vbroadcastss  0x6c3(%rip),%ymm10        # 4af8 <_sk_callback_hsw+0x488>
+  .byte  196,98,125,24,29,190,6,0,0          // vbroadcastss  0x6be(%rip),%ymm11        # 4afc <_sk_callback_hsw+0x48c>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,36,89,193                    // vmulps        %ymm9,%ymm11,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -13019,16 +13043,16 @@ HIDDEN _sk_bicubic_n1x_hsw
 FUNCTION(_sk_bicubic_n1x_hsw)
 _sk_bicubic_n1x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,161,6,0,0          // vbroadcastss  0x6a1(%rip),%ymm0        # 4ae4 <_sk_callback_hsw+0x490>
+  .byte  196,226,125,24,5,161,6,0,0          // vbroadcastss  0x6a1(%rip),%ymm0        # 4b00 <_sk_callback_hsw+0x490>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,152,6,0,0           // vbroadcastss  0x698(%rip),%ymm8        # 4ae8 <_sk_callback_hsw+0x494>
+  .byte  196,98,125,24,5,152,6,0,0           // vbroadcastss  0x698(%rip),%ymm8        # 4b04 <_sk_callback_hsw+0x494>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,142,6,0,0          // vbroadcastss  0x68e(%rip),%ymm9        # 4aec <_sk_callback_hsw+0x498>
-  .byte  196,98,125,24,21,137,6,0,0          // vbroadcastss  0x689(%rip),%ymm10        # 4af0 <_sk_callback_hsw+0x49c>
+  .byte  196,98,125,24,13,142,6,0,0          // vbroadcastss  0x68e(%rip),%ymm9        # 4b08 <_sk_callback_hsw+0x498>
+  .byte  196,98,125,24,21,137,6,0,0          // vbroadcastss  0x689(%rip),%ymm10        # 4b0c <_sk_callback_hsw+0x49c>
   .byte  196,66,61,168,209                   // vfmadd213ps   %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,13,127,6,0,0          // vbroadcastss  0x67f(%rip),%ymm9        # 4af4 <_sk_callback_hsw+0x4a0>
+  .byte  196,98,125,24,13,127,6,0,0          // vbroadcastss  0x67f(%rip),%ymm9        # 4b10 <_sk_callback_hsw+0x4a0>
   .byte  196,66,61,184,202                   // vfmadd231ps   %ymm10,%ymm8,%ymm9
-  .byte  196,98,125,24,21,117,6,0,0          // vbroadcastss  0x675(%rip),%ymm10        # 4af8 <_sk_callback_hsw+0x4a4>
+  .byte  196,98,125,24,21,117,6,0,0          // vbroadcastss  0x675(%rip),%ymm10        # 4b14 <_sk_callback_hsw+0x4a4>
   .byte  196,66,61,184,209                   // vfmadd231ps   %ymm9,%ymm8,%ymm10
   .byte  197,124,17,144,128,0,0,0            // vmovups       %ymm10,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -13039,14 +13063,14 @@ HIDDEN _sk_bicubic_p1x_hsw
 FUNCTION(_sk_bicubic_p1x_hsw)
 _sk_bicubic_p1x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,93,6,0,0            // vbroadcastss  0x65d(%rip),%ymm8        # 4afc <_sk_callback_hsw+0x4a8>
+  .byte  196,98,125,24,5,93,6,0,0            // vbroadcastss  0x65d(%rip),%ymm8        # 4b18 <_sk_callback_hsw+0x4a8>
   .byte  197,188,88,0                        // vaddps        (%rax),%ymm8,%ymm0
   .byte  197,124,16,72,64                    // vmovups       0x40(%rax),%ymm9
-  .byte  196,98,125,24,21,79,6,0,0           // vbroadcastss  0x64f(%rip),%ymm10        # 4b00 <_sk_callback_hsw+0x4ac>
-  .byte  196,98,125,24,29,74,6,0,0           // vbroadcastss  0x64a(%rip),%ymm11        # 4b04 <_sk_callback_hsw+0x4b0>
+  .byte  196,98,125,24,21,79,6,0,0           // vbroadcastss  0x64f(%rip),%ymm10        # 4b1c <_sk_callback_hsw+0x4ac>
+  .byte  196,98,125,24,29,74,6,0,0           // vbroadcastss  0x64a(%rip),%ymm11        # 4b20 <_sk_callback_hsw+0x4b0>
   .byte  196,66,53,168,218                   // vfmadd213ps   %ymm10,%ymm9,%ymm11
   .byte  196,66,53,168,216                   // vfmadd213ps   %ymm8,%ymm9,%ymm11
-  .byte  196,98,125,24,5,59,6,0,0            // vbroadcastss  0x63b(%rip),%ymm8        # 4b08 <_sk_callback_hsw+0x4b4>
+  .byte  196,98,125,24,5,59,6,0,0            // vbroadcastss  0x63b(%rip),%ymm8        # 4b24 <_sk_callback_hsw+0x4b4>
   .byte  196,66,53,184,195                   // vfmadd231ps   %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -13057,12 +13081,12 @@ HIDDEN _sk_bicubic_p3x_hsw
 FUNCTION(_sk_bicubic_p3x_hsw)
 _sk_bicubic_p3x_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,35,6,0,0           // vbroadcastss  0x623(%rip),%ymm0        # 4b0c <_sk_callback_hsw+0x4b8>
+  .byte  196,226,125,24,5,35,6,0,0           // vbroadcastss  0x623(%rip),%ymm0        # 4b28 <_sk_callback_hsw+0x4b8>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,16,6,0,0           // vbroadcastss  0x610(%rip),%ymm10        # 4b10 <_sk_callback_hsw+0x4bc>
-  .byte  196,98,125,24,29,11,6,0,0           // vbroadcastss  0x60b(%rip),%ymm11        # 4b14 <_sk_callback_hsw+0x4c0>
+  .byte  196,98,125,24,21,16,6,0,0           // vbroadcastss  0x610(%rip),%ymm10        # 4b2c <_sk_callback_hsw+0x4bc>
+  .byte  196,98,125,24,29,11,6,0,0           // vbroadcastss  0x60b(%rip),%ymm11        # 4b30 <_sk_callback_hsw+0x4c0>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,52,89,195                    // vmulps        %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -13074,13 +13098,13 @@ HIDDEN _sk_bicubic_n3y_hsw
 FUNCTION(_sk_bicubic_n3y_hsw)
 _sk_bicubic_n3y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,238,5,0,0         // vbroadcastss  0x5ee(%rip),%ymm1        # 4b18 <_sk_callback_hsw+0x4c4>
+  .byte  196,226,125,24,13,238,5,0,0         // vbroadcastss  0x5ee(%rip),%ymm1        # 4b34 <_sk_callback_hsw+0x4c4>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,228,5,0,0           // vbroadcastss  0x5e4(%rip),%ymm8        # 4b1c <_sk_callback_hsw+0x4c8>
+  .byte  196,98,125,24,5,228,5,0,0           // vbroadcastss  0x5e4(%rip),%ymm8        # 4b38 <_sk_callback_hsw+0x4c8>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,213,5,0,0          // vbroadcastss  0x5d5(%rip),%ymm10        # 4b20 <_sk_callback_hsw+0x4cc>
-  .byte  196,98,125,24,29,208,5,0,0          // vbroadcastss  0x5d0(%rip),%ymm11        # 4b24 <_sk_callback_hsw+0x4d0>
+  .byte  196,98,125,24,21,213,5,0,0          // vbroadcastss  0x5d5(%rip),%ymm10        # 4b3c <_sk_callback_hsw+0x4cc>
+  .byte  196,98,125,24,29,208,5,0,0          // vbroadcastss  0x5d0(%rip),%ymm11        # 4b40 <_sk_callback_hsw+0x4d0>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,36,89,193                    // vmulps        %ymm9,%ymm11,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -13092,16 +13116,16 @@ HIDDEN _sk_bicubic_n1y_hsw
 FUNCTION(_sk_bicubic_n1y_hsw)
 _sk_bicubic_n1y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,179,5,0,0         // vbroadcastss  0x5b3(%rip),%ymm1        # 4b28 <_sk_callback_hsw+0x4d4>
+  .byte  196,226,125,24,13,179,5,0,0         // vbroadcastss  0x5b3(%rip),%ymm1        # 4b44 <_sk_callback_hsw+0x4d4>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,169,5,0,0           // vbroadcastss  0x5a9(%rip),%ymm8        # 4b2c <_sk_callback_hsw+0x4d8>
+  .byte  196,98,125,24,5,169,5,0,0           // vbroadcastss  0x5a9(%rip),%ymm8        # 4b48 <_sk_callback_hsw+0x4d8>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,159,5,0,0          // vbroadcastss  0x59f(%rip),%ymm9        # 4b30 <_sk_callback_hsw+0x4dc>
-  .byte  196,98,125,24,21,154,5,0,0          // vbroadcastss  0x59a(%rip),%ymm10        # 4b34 <_sk_callback_hsw+0x4e0>
+  .byte  196,98,125,24,13,159,5,0,0          // vbroadcastss  0x59f(%rip),%ymm9        # 4b4c <_sk_callback_hsw+0x4dc>
+  .byte  196,98,125,24,21,154,5,0,0          // vbroadcastss  0x59a(%rip),%ymm10        # 4b50 <_sk_callback_hsw+0x4e0>
   .byte  196,66,61,168,209                   // vfmadd213ps   %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,13,144,5,0,0          // vbroadcastss  0x590(%rip),%ymm9        # 4b38 <_sk_callback_hsw+0x4e4>
+  .byte  196,98,125,24,13,144,5,0,0          // vbroadcastss  0x590(%rip),%ymm9        # 4b54 <_sk_callback_hsw+0x4e4>
   .byte  196,66,61,184,202                   // vfmadd231ps   %ymm10,%ymm8,%ymm9
-  .byte  196,98,125,24,21,134,5,0,0          // vbroadcastss  0x586(%rip),%ymm10        # 4b3c <_sk_callback_hsw+0x4e8>
+  .byte  196,98,125,24,21,134,5,0,0          // vbroadcastss  0x586(%rip),%ymm10        # 4b58 <_sk_callback_hsw+0x4e8>
   .byte  196,66,61,184,209                   // vfmadd231ps   %ymm9,%ymm8,%ymm10
   .byte  197,124,17,144,160,0,0,0            // vmovups       %ymm10,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -13112,14 +13136,14 @@ HIDDEN _sk_bicubic_p1y_hsw
 FUNCTION(_sk_bicubic_p1y_hsw)
 _sk_bicubic_p1y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,110,5,0,0           // vbroadcastss  0x56e(%rip),%ymm8        # 4b40 <_sk_callback_hsw+0x4ec>
+  .byte  196,98,125,24,5,110,5,0,0           // vbroadcastss  0x56e(%rip),%ymm8        # 4b5c <_sk_callback_hsw+0x4ec>
   .byte  197,188,88,72,32                    // vaddps        0x20(%rax),%ymm8,%ymm1
   .byte  197,124,16,72,96                    // vmovups       0x60(%rax),%ymm9
-  .byte  196,98,125,24,21,95,5,0,0           // vbroadcastss  0x55f(%rip),%ymm10        # 4b44 <_sk_callback_hsw+0x4f0>
-  .byte  196,98,125,24,29,90,5,0,0           // vbroadcastss  0x55a(%rip),%ymm11        # 4b48 <_sk_callback_hsw+0x4f4>
+  .byte  196,98,125,24,21,95,5,0,0           // vbroadcastss  0x55f(%rip),%ymm10        # 4b60 <_sk_callback_hsw+0x4f0>
+  .byte  196,98,125,24,29,90,5,0,0           // vbroadcastss  0x55a(%rip),%ymm11        # 4b64 <_sk_callback_hsw+0x4f4>
   .byte  196,66,53,168,218                   // vfmadd213ps   %ymm10,%ymm9,%ymm11
   .byte  196,66,53,168,216                   // vfmadd213ps   %ymm8,%ymm9,%ymm11
-  .byte  196,98,125,24,5,75,5,0,0            // vbroadcastss  0x54b(%rip),%ymm8        # 4b4c <_sk_callback_hsw+0x4f8>
+  .byte  196,98,125,24,5,75,5,0,0            // vbroadcastss  0x54b(%rip),%ymm8        # 4b68 <_sk_callback_hsw+0x4f8>
   .byte  196,66,53,184,195                   // vfmadd231ps   %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -13130,12 +13154,12 @@ HIDDEN _sk_bicubic_p3y_hsw
 FUNCTION(_sk_bicubic_p3y_hsw)
 _sk_bicubic_p3y_hsw:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,51,5,0,0          // vbroadcastss  0x533(%rip),%ymm1        # 4b50 <_sk_callback_hsw+0x4fc>
+  .byte  196,226,125,24,13,51,5,0,0          // vbroadcastss  0x533(%rip),%ymm1        # 4b6c <_sk_callback_hsw+0x4fc>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,31,5,0,0           // vbroadcastss  0x51f(%rip),%ymm10        # 4b54 <_sk_callback_hsw+0x500>
-  .byte  196,98,125,24,29,26,5,0,0           // vbroadcastss  0x51a(%rip),%ymm11        # 4b58 <_sk_callback_hsw+0x504>
+  .byte  196,98,125,24,21,31,5,0,0           // vbroadcastss  0x51f(%rip),%ymm10        # 4b70 <_sk_callback_hsw+0x500>
+  .byte  196,98,125,24,29,26,5,0,0           // vbroadcastss  0x51a(%rip),%ymm11        # 4b74 <_sk_callback_hsw+0x504>
   .byte  196,66,61,168,218                   // vfmadd213ps   %ymm10,%ymm8,%ymm11
   .byte  196,65,52,89,195                    // vmulps        %ymm11,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -13259,25 +13283,25 @@ BALIGN4
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 482d <.literal4+0xb1>
+  .byte  71,225,61                           // rex.RXB       loope 4849 <.literal4+0xb1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 483d <.literal4+0xc1>
+  .byte  71,225,61                           // rex.RXB       loope 4859 <.literal4+0xc1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 484d <.literal4+0xd1>
+  .byte  71,225,61                           // rex.RXB       loope 4869 <.literal4+0xd1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 485d <.literal4+0xe1>
+  .byte  71,225,61                           // rex.RXB       loope 4879 <.literal4+0xe1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -13327,7 +13351,7 @@ BALIGN4
   .byte  190,129,128,128,59                  // mov           $0x3b808081,%esi
   .byte  129,128,128,59,0,248,0,0,8,33       // addl          $0x21080000,-0x7ffc480(%rax)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        48ad <.literal4+0x131>
+  .byte  224,7                               // loopne        48c9 <.literal4+0x131>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -13343,10 +13367,10 @@ BALIGN4
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
   .byte  0,52,255                            // add           %dh,(%rdi,%rdi,8)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            48d4 <.literal4+0x158>
+  .byte  127,0                               // jg            48f0 <.literal4+0x158>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            494d <.literal4+0x1d1>
+  .byte  119,115                             // ja            4969 <.literal4+0x1d1>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -13360,10 +13384,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4908 <.literal4+0x18c>
+  .byte  127,0                               // jg            4924 <.literal4+0x18c>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4981 <.literal4+0x205>
+  .byte  119,115                             // ja            499d <.literal4+0x205>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -13377,10 +13401,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            493c <.literal4+0x1c0>
+  .byte  127,0                               // jg            4958 <.literal4+0x1c0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            49b5 <.literal4+0x239>
+  .byte  119,115                             // ja            49d1 <.literal4+0x239>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -13394,10 +13418,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4970 <.literal4+0x1f4>
+  .byte  127,0                               // jg            498c <.literal4+0x1f4>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            49e9 <.literal4+0x26d>
+  .byte  119,115                             // ja            4a05 <.literal4+0x26d>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -13410,7 +13434,7 @@ BALIGN4
   .byte  0,75,0                              // add           %cl,0x0(%rbx)
   .byte  0,128,63,0,0,200                    // add           %al,-0x37ffffc1(%rax)
   .byte  66,0,0                              // rex.X         add %al,(%rax)
-  .byte  127,67                              // jg            49e7 <.literal4+0x26b>
+  .byte  127,67                              // jg            4a03 <.literal4+0x26b>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -13422,10 +13446,10 @@ BALIGN4
   .byte  190,80,128,3,62                     // mov           $0x3e038050,%esi
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           4a07 <.literal4+0x28b>
+  .byte  118,63                              // jbe           4a23 <.literal4+0x28b>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            4a1b <.literal4+0x29f>
+  .byte  127,67                              // jg            4a37 <.literal4+0x29f>
   .byte  129,128,128,59,0,0,128,63,129,128   // addl          $0x80813f80,0x3b80(%rax)
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,128,63,129,128,128                // add           %al,-0x7f7f7ec1(%rax)
@@ -13434,7 +13458,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        49fd <.literal4+0x281>
+  .byte  224,7                               // loopne        4a19 <.literal4+0x281>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -13446,7 +13470,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4a19 <.literal4+0x29d>
+  .byte  224,7                               // loopne        4a35 <.literal4+0x29d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -13457,7 +13481,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            4a6e <.literal4+0x2f2>
+  .byte  124,66                              // jl            4a8a <.literal4+0x2f2>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,55,0,15                 // mov           %ecx,0xf003788(%rax)
@@ -13475,9 +13499,9 @@ BALIGN4
   .byte  137,136,136,59,15,0                 // mov           %ecx,0xf3b88(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,61,0,0                  // mov           %ecx,0x3d88(%rax)
-  .byte  112,65                              // jo            4ab1 <.literal4+0x335>
+  .byte  112,65                              // jo            4acd <.literal4+0x335>
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            4abf <.literal4+0x343>
+  .byte  127,67                              // jg            4adb <.literal4+0x343>
   .byte  128,0,128                           // addb          $0x80,(%rax)
   .byte  55                                  // (bad)
   .byte  128,0,128                           // addb          $0x80,(%rax)
@@ -13485,7 +13509,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  255                                 // (bad)
-  .byte  127,71                              // jg            4ad3 <.literal4+0x357>
+  .byte  127,71                              // jg            4aef <.literal4+0x357>
   .byte  208                                 // (bad)
   .byte  179,89                              // mov           $0x59,%bl
   .byte  62,89                               // ds            pop %rcx
@@ -13585,16 +13609,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004b88 <_sk_callback_hsw+0xa000534>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004ba8 <_sk_callback_hsw+0xa000538>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004b90 <_sk_callback_hsw+0x1200053c>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004bb0 <_sk_callback_hsw+0x12000540>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004b98 <_sk_callback_hsw+0x1a000544>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004bb8 <_sk_callback_hsw+0x1a000548>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004ba0 <_sk_callback_hsw+0x300054c>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004bc0 <_sk_callback_hsw+0x3000550>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13637,16 +13661,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004be8 <_sk_callback_hsw+0xa000594>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004c08 <_sk_callback_hsw+0xa000598>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004bf0 <_sk_callback_hsw+0x1200059c>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004c10 <_sk_callback_hsw+0x120005a0>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004bf8 <_sk_callback_hsw+0x1a0005a4>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004c18 <_sk_callback_hsw+0x1a0005a8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004c00 <_sk_callback_hsw+0x30005ac>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004c20 <_sk_callback_hsw+0x30005b0>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13689,16 +13713,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004c48 <_sk_callback_hsw+0xa0005f4>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004c68 <_sk_callback_hsw+0xa0005f8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004c50 <_sk_callback_hsw+0x120005fc>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004c70 <_sk_callback_hsw+0x12000600>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004c58 <_sk_callback_hsw+0x1a000604>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004c78 <_sk_callback_hsw+0x1a000608>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004c60 <_sk_callback_hsw+0x300060c>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004c80 <_sk_callback_hsw+0x3000610>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13741,16 +13765,16 @@ BALIGN32
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004ca8 <_sk_callback_hsw+0xa000654>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004cc8 <_sk_callback_hsw+0xa000658>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004cb0 <_sk_callback_hsw+0x1200065c>
+  .byte  255,13,255,255,255,17               // decl          0x11ffffff(%rip)        # 12004cd0 <_sk_callback_hsw+0x12000660>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004cb8 <_sk_callback_hsw+0x1a000664>
+  .byte  255,21,255,255,255,25               // callq         *0x19ffffff(%rip)        # 1a004cd8 <_sk_callback_hsw+0x1a000668>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004cc0 <_sk_callback_hsw+0x300066c>
+  .byte  255,29,255,255,255,2                // lcall         *0x2ffffff(%rip)        # 3004ce0 <_sk_callback_hsw+0x3000670>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -13871,14 +13895,14 @@ _sk_seed_shader_avx:
   .byte  197,249,112,192,0                   // vpshufd       $0x0,%xmm0,%xmm0
   .byte  196,227,125,24,192,1                // vinsertf128   $0x1,%xmm0,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,163,98,0,0        // vbroadcastss  0x62a3(%rip),%ymm1        # 636c <_sk_callback_avx+0x126>
+  .byte  196,226,125,24,13,191,98,0,0        // vbroadcastss  0x62bf(%rip),%ymm1        # 6388 <_sk_callback_avx+0x126>
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
   .byte  197,252,88,2                        // vaddps        (%rdx),%ymm0,%ymm0
   .byte  196,226,125,24,16                   // vbroadcastss  (%rax),%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  197,236,88,201                      // vaddps        %ymm1,%ymm2,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,21,135,98,0,0        // vbroadcastss  0x6287(%rip),%ymm2        # 6370 <_sk_callback_avx+0x12a>
+  .byte  196,226,125,24,21,163,98,0,0        // vbroadcastss  0x62a3(%rip),%ymm2        # 638c <_sk_callback_avx+0x12a>
   .byte  197,228,87,219                      // vxorps        %ymm3,%ymm3,%ymm3
   .byte  197,220,87,228                      // vxorps        %ymm4,%ymm4,%ymm4
   .byte  197,212,87,237                      // vxorps        %ymm5,%ymm5,%ymm5
@@ -13900,7 +13924,7 @@ _sk_dither_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  196,66,125,24,8                     // vbroadcastss  (%r8),%ymm9
   .byte  196,65,60,87,209                    // vxorps        %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,29,63,98,0,0          // vbroadcastss  0x623f(%rip),%ymm11        # 6374 <_sk_callback_avx+0x12e>
+  .byte  196,98,125,24,29,91,98,0,0          // vbroadcastss  0x625b(%rip),%ymm11        # 6390 <_sk_callback_avx+0x12e>
   .byte  196,65,44,84,203                    // vandps        %ymm11,%ymm10,%ymm9
   .byte  196,193,25,114,241,5                // vpslld        $0x5,%xmm9,%xmm12
   .byte  196,67,125,25,201,1                 // vextractf128  $0x1,%ymm9,%xmm9
@@ -13911,8 +13935,8 @@ _sk_dither_avx:
   .byte  196,67,125,25,219,1                 // vextractf128  $0x1,%ymm11,%xmm11
   .byte  196,193,33,114,243,4                // vpslld        $0x4,%xmm11,%xmm11
   .byte  196,67,29,24,219,1                  // vinsertf128   $0x1,%xmm11,%ymm12,%ymm11
-  .byte  196,98,125,24,37,0,98,0,0           // vbroadcastss  0x6200(%rip),%ymm12        # 6378 <_sk_callback_avx+0x132>
-  .byte  196,98,125,24,45,251,97,0,0         // vbroadcastss  0x61fb(%rip),%ymm13        # 637c <_sk_callback_avx+0x136>
+  .byte  196,98,125,24,37,28,98,0,0          // vbroadcastss  0x621c(%rip),%ymm12        # 6394 <_sk_callback_avx+0x132>
+  .byte  196,98,125,24,45,23,98,0,0          // vbroadcastss  0x6217(%rip),%ymm13        # 6398 <_sk_callback_avx+0x136>
   .byte  196,65,44,84,245                    // vandps        %ymm13,%ymm10,%ymm14
   .byte  196,193,1,114,246,2                 // vpslld        $0x2,%xmm14,%xmm15
   .byte  196,67,125,25,246,1                 // vextractf128  $0x1,%ymm14,%xmm14
@@ -13939,15 +13963,22 @@ _sk_dither_avx:
   .byte  196,65,60,86,193                    // vorps         %ymm9,%ymm8,%ymm8
   .byte  196,65,60,86,194                    // vorps         %ymm10,%ymm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,102,97,0,0         // vbroadcastss  0x6166(%rip),%ymm9        # 6380 <_sk_callback_avx+0x13a>
+  .byte  196,98,125,24,13,130,97,0,0         // vbroadcastss  0x6182(%rip),%ymm9        # 639c <_sk_callback_avx+0x13a>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,92,97,0,0          // vbroadcastss  0x615c(%rip),%ymm9        # 6384 <_sk_callback_avx+0x13e>
+  .byte  196,98,125,24,13,120,97,0,0         // vbroadcastss  0x6178(%rip),%ymm9        # 63a0 <_sk_callback_avx+0x13e>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  196,98,125,24,72,8                  // vbroadcastss  0x8(%rax),%ymm9
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,188,88,192                      // vaddps        %ymm0,%ymm8,%ymm0
   .byte  197,188,88,201                      // vaddps        %ymm1,%ymm8,%ymm1
   .byte  197,188,88,210                      // vaddps        %ymm2,%ymm8,%ymm2
+  .byte  197,252,93,195                      // vminps        %ymm3,%ymm0,%ymm0
+  .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
+  .byte  197,188,95,192                      // vmaxps        %ymm0,%ymm8,%ymm0
+  .byte  197,244,93,203                      // vminps        %ymm3,%ymm1,%ymm1
+  .byte  197,188,95,201                      // vmaxps        %ymm1,%ymm8,%ymm1
+  .byte  197,236,93,211                      // vminps        %ymm3,%ymm2,%ymm2
+  .byte  197,188,95,210                      // vmaxps        %ymm2,%ymm8,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -14003,7 +14034,7 @@ HIDDEN _sk_srcatop_avx
 FUNCTION(_sk_srcatop_avx)
 _sk_srcatop_avx:
   .byte  197,252,89,199                      // vmulps        %ymm7,%ymm0,%ymm0
-  .byte  196,98,125,24,5,208,96,0,0          // vbroadcastss  0x60d0(%rip),%ymm8        # 6388 <_sk_callback_avx+0x142>
+  .byte  196,98,125,24,5,207,96,0,0          // vbroadcastss  0x60cf(%rip),%ymm8        # 63a4 <_sk_callback_avx+0x142>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,204                       // vmulps        %ymm4,%ymm8,%ymm9
   .byte  197,180,88,192                      // vaddps        %ymm0,%ymm9,%ymm0
@@ -14024,7 +14055,7 @@ HIDDEN _sk_dstatop_avx
 FUNCTION(_sk_dstatop_avx)
 _sk_dstatop_avx:
   .byte  197,100,89,196                      // vmulps        %ymm4,%ymm3,%ymm8
-  .byte  196,98,125,24,13,146,96,0,0         // vbroadcastss  0x6092(%rip),%ymm9        # 638c <_sk_callback_avx+0x146>
+  .byte  196,98,125,24,13,145,96,0,0         // vbroadcastss  0x6091(%rip),%ymm9        # 63a8 <_sk_callback_avx+0x146>
   .byte  197,52,92,207                       // vsubps        %ymm7,%ymm9,%ymm9
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
   .byte  197,188,88,192                      // vaddps        %ymm0,%ymm8,%ymm0
@@ -14066,7 +14097,7 @@ HIDDEN _sk_srcout_avx
 .globl _sk_srcout_avx
 FUNCTION(_sk_srcout_avx)
 _sk_srcout_avx:
-  .byte  196,98,125,24,5,49,96,0,0           // vbroadcastss  0x6031(%rip),%ymm8        # 6390 <_sk_callback_avx+0x14a>
+  .byte  196,98,125,24,5,48,96,0,0           // vbroadcastss  0x6030(%rip),%ymm8        # 63ac <_sk_callback_avx+0x14a>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -14079,7 +14110,7 @@ HIDDEN _sk_dstout_avx
 .globl _sk_dstout_avx
 FUNCTION(_sk_dstout_avx)
 _sk_dstout_avx:
-  .byte  196,226,125,24,5,20,96,0,0          // vbroadcastss  0x6014(%rip),%ymm0        # 6394 <_sk_callback_avx+0x14e>
+  .byte  196,226,125,24,5,19,96,0,0          // vbroadcastss  0x6013(%rip),%ymm0        # 63b0 <_sk_callback_avx+0x14e>
   .byte  197,252,92,219                      // vsubps        %ymm3,%ymm0,%ymm3
   .byte  197,228,89,196                      // vmulps        %ymm4,%ymm3,%ymm0
   .byte  197,228,89,205                      // vmulps        %ymm5,%ymm3,%ymm1
@@ -14092,7 +14123,7 @@ HIDDEN _sk_srcover_avx
 .globl _sk_srcover_avx
 FUNCTION(_sk_srcover_avx)
 _sk_srcover_avx:
-  .byte  196,98,125,24,5,247,95,0,0          // vbroadcastss  0x5ff7(%rip),%ymm8        # 6398 <_sk_callback_avx+0x152>
+  .byte  196,98,125,24,5,246,95,0,0          // vbroadcastss  0x5ff6(%rip),%ymm8        # 63b4 <_sk_callback_avx+0x152>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,204                       // vmulps        %ymm4,%ymm8,%ymm9
   .byte  197,180,88,192                      // vaddps        %ymm0,%ymm9,%ymm0
@@ -14109,7 +14140,7 @@ HIDDEN _sk_dstover_avx
 .globl _sk_dstover_avx
 FUNCTION(_sk_dstover_avx)
 _sk_dstover_avx:
-  .byte  196,98,125,24,5,202,95,0,0          // vbroadcastss  0x5fca(%rip),%ymm8        # 639c <_sk_callback_avx+0x156>
+  .byte  196,98,125,24,5,201,95,0,0          // vbroadcastss  0x5fc9(%rip),%ymm8        # 63b8 <_sk_callback_avx+0x156>
   .byte  197,60,92,199                       // vsubps        %ymm7,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,252,88,196                      // vaddps        %ymm4,%ymm0,%ymm0
@@ -14137,7 +14168,7 @@ HIDDEN _sk_multiply_avx
 .globl _sk_multiply_avx
 FUNCTION(_sk_multiply_avx)
 _sk_multiply_avx:
-  .byte  196,98,125,24,5,137,95,0,0          // vbroadcastss  0x5f89(%rip),%ymm8        # 63a0 <_sk_callback_avx+0x15a>
+  .byte  196,98,125,24,5,136,95,0,0          // vbroadcastss  0x5f88(%rip),%ymm8        # 63bc <_sk_callback_avx+0x15a>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,208                       // vmulps        %ymm0,%ymm9,%ymm10
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14197,7 +14228,7 @@ HIDDEN _sk_xor__avx
 .globl _sk_xor__avx
 FUNCTION(_sk_xor__avx)
 _sk_xor__avx:
-  .byte  196,98,125,24,5,216,94,0,0          // vbroadcastss  0x5ed8(%rip),%ymm8        # 63a4 <_sk_callback_avx+0x15e>
+  .byte  196,98,125,24,5,215,94,0,0          // vbroadcastss  0x5ed7(%rip),%ymm8        # 63c0 <_sk_callback_avx+0x15e>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,180,89,192                      // vmulps        %ymm0,%ymm9,%ymm0
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14234,7 +14265,7 @@ _sk_darken_avx:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,95,209                  // vmaxps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,88,94,0,0           // vbroadcastss  0x5e58(%rip),%ymm8        # 63a8 <_sk_callback_avx+0x162>
+  .byte  196,98,125,24,5,87,94,0,0           // vbroadcastss  0x5e57(%rip),%ymm8        # 63c4 <_sk_callback_avx+0x162>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -14260,7 +14291,7 @@ _sk_lighten_avx:
   .byte  197,100,89,206                      // vmulps        %ymm6,%ymm3,%ymm9
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,4,94,0,0            // vbroadcastss  0x5e04(%rip),%ymm8        # 63ac <_sk_callback_avx+0x166>
+  .byte  196,98,125,24,5,3,94,0,0            // vbroadcastss  0x5e03(%rip),%ymm8        # 63c8 <_sk_callback_avx+0x166>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -14289,7 +14320,7 @@ _sk_difference_avx:
   .byte  196,193,108,93,209                  // vminps        %ymm9,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,164,93,0,0          // vbroadcastss  0x5da4(%rip),%ymm8        # 63b0 <_sk_callback_avx+0x16a>
+  .byte  196,98,125,24,5,163,93,0,0          // vbroadcastss  0x5da3(%rip),%ymm8        # 63cc <_sk_callback_avx+0x16a>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -14312,7 +14343,7 @@ _sk_exclusion_avx:
   .byte  197,236,89,214                      // vmulps        %ymm6,%ymm2,%ymm2
   .byte  197,236,88,210                      // vaddps        %ymm2,%ymm2,%ymm2
   .byte  197,188,92,210                      // vsubps        %ymm2,%ymm8,%ymm2
-  .byte  196,98,125,24,5,95,93,0,0           // vbroadcastss  0x5d5f(%rip),%ymm8        # 63b4 <_sk_callback_avx+0x16e>
+  .byte  196,98,125,24,5,94,93,0,0           // vbroadcastss  0x5d5e(%rip),%ymm8        # 63d0 <_sk_callback_avx+0x16e>
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
   .byte  197,60,89,199                       // vmulps        %ymm7,%ymm8,%ymm8
   .byte  197,188,88,219                      // vaddps        %ymm3,%ymm8,%ymm3
@@ -14323,7 +14354,7 @@ HIDDEN _sk_colorburn_avx
 .globl _sk_colorburn_avx
 FUNCTION(_sk_colorburn_avx)
 _sk_colorburn_avx:
-  .byte  196,98,125,24,5,74,93,0,0           // vbroadcastss  0x5d4a(%rip),%ymm8        # 63b8 <_sk_callback_avx+0x172>
+  .byte  196,98,125,24,5,73,93,0,0           // vbroadcastss  0x5d49(%rip),%ymm8        # 63d4 <_sk_callback_avx+0x172>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,52,89,216                       // vmulps        %ymm0,%ymm9,%ymm11
   .byte  196,65,44,87,210                    // vxorps        %ymm10,%ymm10,%ymm10
@@ -14385,7 +14416,7 @@ HIDDEN _sk_colordodge_avx
 FUNCTION(_sk_colordodge_avx)
 _sk_colordodge_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
-  .byte  196,98,125,24,13,70,92,0,0          // vbroadcastss  0x5c46(%rip),%ymm9        # 63bc <_sk_callback_avx+0x176>
+  .byte  196,98,125,24,13,69,92,0,0          // vbroadcastss  0x5c45(%rip),%ymm9        # 63d8 <_sk_callback_avx+0x176>
   .byte  197,52,92,215                       // vsubps        %ymm7,%ymm9,%ymm10
   .byte  197,44,89,216                       // vmulps        %ymm0,%ymm10,%ymm11
   .byte  197,52,92,203                       // vsubps        %ymm3,%ymm9,%ymm9
@@ -14442,7 +14473,7 @@ HIDDEN _sk_hardlight_avx
 .globl _sk_hardlight_avx
 FUNCTION(_sk_hardlight_avx)
 _sk_hardlight_avx:
-  .byte  196,98,125,24,5,88,91,0,0           // vbroadcastss  0x5b58(%rip),%ymm8        # 63c0 <_sk_callback_avx+0x17a>
+  .byte  196,98,125,24,5,87,91,0,0           // vbroadcastss  0x5b57(%rip),%ymm8        # 63dc <_sk_callback_avx+0x17a>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,200                       // vmulps        %ymm0,%ymm10,%ymm9
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14497,7 +14528,7 @@ HIDDEN _sk_overlay_avx
 .globl _sk_overlay_avx
 FUNCTION(_sk_overlay_avx)
 _sk_overlay_avx:
-  .byte  196,98,125,24,5,129,90,0,0          // vbroadcastss  0x5a81(%rip),%ymm8        # 63c4 <_sk_callback_avx+0x17e>
+  .byte  196,98,125,24,5,128,90,0,0          // vbroadcastss  0x5a80(%rip),%ymm8        # 63e0 <_sk_callback_avx+0x17e>
   .byte  197,60,92,215                       // vsubps        %ymm7,%ymm8,%ymm10
   .byte  197,44,89,200                       // vmulps        %ymm0,%ymm10,%ymm9
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14563,10 +14594,10 @@ _sk_softlight_avx:
   .byte  196,65,60,88,192                    // vaddps        %ymm8,%ymm8,%ymm8
   .byte  196,65,60,89,216                    // vmulps        %ymm8,%ymm8,%ymm11
   .byte  196,65,60,88,195                    // vaddps        %ymm11,%ymm8,%ymm8
-  .byte  196,98,125,24,29,120,89,0,0         // vbroadcastss  0x5978(%rip),%ymm11        # 63cc <_sk_callback_avx+0x186>
+  .byte  196,98,125,24,29,119,89,0,0         // vbroadcastss  0x5977(%rip),%ymm11        # 63e8 <_sk_callback_avx+0x186>
   .byte  196,65,28,88,235                    // vaddps        %ymm11,%ymm12,%ymm13
   .byte  196,65,20,89,192                    // vmulps        %ymm8,%ymm13,%ymm8
-  .byte  196,98,125,24,45,105,89,0,0         // vbroadcastss  0x5969(%rip),%ymm13        # 63d0 <_sk_callback_avx+0x18a>
+  .byte  196,98,125,24,45,104,89,0,0         // vbroadcastss  0x5968(%rip),%ymm13        # 63ec <_sk_callback_avx+0x18a>
   .byte  196,65,28,89,245                    // vmulps        %ymm13,%ymm12,%ymm14
   .byte  196,65,12,88,192                    // vaddps        %ymm8,%ymm14,%ymm8
   .byte  196,65,124,82,244                   // vrsqrtps      %ymm12,%ymm14
@@ -14577,7 +14608,7 @@ _sk_softlight_avx:
   .byte  197,4,194,255,2                     // vcmpleps      %ymm7,%ymm15,%ymm15
   .byte  196,67,13,74,240,240                // vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   .byte  197,116,88,249                      // vaddps        %ymm1,%ymm1,%ymm15
-  .byte  196,98,125,24,5,39,89,0,0           // vbroadcastss  0x5927(%rip),%ymm8        # 63c8 <_sk_callback_avx+0x182>
+  .byte  196,98,125,24,5,38,89,0,0           // vbroadcastss  0x5926(%rip),%ymm8        # 63e4 <_sk_callback_avx+0x182>
   .byte  196,65,60,92,228                    // vsubps        %ymm12,%ymm8,%ymm12
   .byte  197,132,92,195                      // vsubps        %ymm3,%ymm15,%ymm0
   .byte  196,65,124,89,228                   // vmulps        %ymm12,%ymm0,%ymm12
@@ -14704,12 +14735,12 @@ _sk_hue_avx:
   .byte  196,65,28,89,219                    // vmulps        %ymm11,%ymm12,%ymm11
   .byte  196,65,36,94,222                    // vdivps        %ymm14,%ymm11,%ymm11
   .byte  196,67,37,74,224,240                // vblendvps     %ymm15,%ymm8,%ymm11,%ymm12
-  .byte  196,98,125,24,53,246,86,0,0         // vbroadcastss  0x56f6(%rip),%ymm14        # 63d4 <_sk_callback_avx+0x18e>
+  .byte  196,98,125,24,53,245,86,0,0         // vbroadcastss  0x56f5(%rip),%ymm14        # 63f0 <_sk_callback_avx+0x18e>
   .byte  196,65,92,89,222                    // vmulps        %ymm14,%ymm4,%ymm11
-  .byte  196,98,125,24,61,236,86,0,0         // vbroadcastss  0x56ec(%rip),%ymm15        # 63d8 <_sk_callback_avx+0x192>
+  .byte  196,98,125,24,61,235,86,0,0         // vbroadcastss  0x56eb(%rip),%ymm15        # 63f4 <_sk_callback_avx+0x192>
   .byte  196,65,84,89,239                    // vmulps        %ymm15,%ymm5,%ymm13
   .byte  196,65,36,88,221                    // vaddps        %ymm13,%ymm11,%ymm11
-  .byte  196,226,125,24,5,221,86,0,0         // vbroadcastss  0x56dd(%rip),%ymm0        # 63dc <_sk_callback_avx+0x196>
+  .byte  196,226,125,24,5,220,86,0,0         // vbroadcastss  0x56dc(%rip),%ymm0        # 63f8 <_sk_callback_avx+0x196>
   .byte  197,76,89,232                       // vmulps        %ymm0,%ymm6,%ymm13
   .byte  196,65,36,88,221                    // vaddps        %ymm13,%ymm11,%ymm11
   .byte  196,65,52,89,238                    // vmulps        %ymm14,%ymm9,%ymm13
@@ -14770,7 +14801,7 @@ _sk_hue_avx:
   .byte  196,65,36,95,208                    // vmaxps        %ymm8,%ymm11,%ymm10
   .byte  196,195,109,74,209,240              // vblendvps     %ymm15,%ymm9,%ymm2,%ymm2
   .byte  196,193,108,95,208                  // vmaxps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,182,85,0,0          // vbroadcastss  0x55b6(%rip),%ymm8        # 63e0 <_sk_callback_avx+0x19a>
+  .byte  196,98,125,24,5,181,85,0,0          // vbroadcastss  0x55b5(%rip),%ymm8        # 63fc <_sk_callback_avx+0x19a>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,180,89,201                      // vmulps        %ymm1,%ymm9,%ymm1
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14827,12 +14858,12 @@ _sk_saturation_avx:
   .byte  196,65,28,89,219                    // vmulps        %ymm11,%ymm12,%ymm11
   .byte  196,65,36,94,222                    // vdivps        %ymm14,%ymm11,%ymm11
   .byte  196,67,37,74,224,240                // vblendvps     %ymm15,%ymm8,%ymm11,%ymm12
-  .byte  196,98,125,24,53,196,84,0,0         // vbroadcastss  0x54c4(%rip),%ymm14        # 63e4 <_sk_callback_avx+0x19e>
+  .byte  196,98,125,24,53,195,84,0,0         // vbroadcastss  0x54c3(%rip),%ymm14        # 6400 <_sk_callback_avx+0x19e>
   .byte  196,65,92,89,222                    // vmulps        %ymm14,%ymm4,%ymm11
-  .byte  196,98,125,24,61,186,84,0,0         // vbroadcastss  0x54ba(%rip),%ymm15        # 63e8 <_sk_callback_avx+0x1a2>
+  .byte  196,98,125,24,61,185,84,0,0         // vbroadcastss  0x54b9(%rip),%ymm15        # 6404 <_sk_callback_avx+0x1a2>
   .byte  196,65,84,89,239                    // vmulps        %ymm15,%ymm5,%ymm13
   .byte  196,65,36,88,221                    // vaddps        %ymm13,%ymm11,%ymm11
-  .byte  196,226,125,24,5,171,84,0,0         // vbroadcastss  0x54ab(%rip),%ymm0        # 63ec <_sk_callback_avx+0x1a6>
+  .byte  196,226,125,24,5,170,84,0,0         // vbroadcastss  0x54aa(%rip),%ymm0        # 6408 <_sk_callback_avx+0x1a6>
   .byte  197,76,89,232                       // vmulps        %ymm0,%ymm6,%ymm13
   .byte  196,65,36,88,221                    // vaddps        %ymm13,%ymm11,%ymm11
   .byte  196,65,52,89,238                    // vmulps        %ymm14,%ymm9,%ymm13
@@ -14893,7 +14924,7 @@ _sk_saturation_avx:
   .byte  196,65,36,95,208                    // vmaxps        %ymm8,%ymm11,%ymm10
   .byte  196,195,109,74,209,240              // vblendvps     %ymm15,%ymm9,%ymm2,%ymm2
   .byte  196,193,108,95,208                  // vmaxps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,132,83,0,0          // vbroadcastss  0x5384(%rip),%ymm8        # 63f0 <_sk_callback_avx+0x1aa>
+  .byte  196,98,125,24,5,131,83,0,0          // vbroadcastss  0x5383(%rip),%ymm8        # 640c <_sk_callback_avx+0x1aa>
   .byte  197,60,92,207                       // vsubps        %ymm7,%ymm8,%ymm9
   .byte  197,180,89,201                      // vmulps        %ymm1,%ymm9,%ymm1
   .byte  197,60,92,195                       // vsubps        %ymm3,%ymm8,%ymm8
@@ -14922,12 +14953,12 @@ _sk_color_avx:
   .byte  197,252,17,68,36,168                // vmovups       %ymm0,-0x58(%rsp)
   .byte  197,124,89,199                      // vmulps        %ymm7,%ymm0,%ymm8
   .byte  197,116,89,207                      // vmulps        %ymm7,%ymm1,%ymm9
-  .byte  196,98,125,24,45,26,83,0,0          // vbroadcastss  0x531a(%rip),%ymm13        # 63f4 <_sk_callback_avx+0x1ae>
+  .byte  196,98,125,24,45,25,83,0,0          // vbroadcastss  0x5319(%rip),%ymm13        # 6410 <_sk_callback_avx+0x1ae>
   .byte  196,65,92,89,213                    // vmulps        %ymm13,%ymm4,%ymm10
-  .byte  196,98,125,24,53,16,83,0,0          // vbroadcastss  0x5310(%rip),%ymm14        # 63f8 <_sk_callback_avx+0x1b2>
+  .byte  196,98,125,24,53,15,83,0,0          // vbroadcastss  0x530f(%rip),%ymm14        # 6414 <_sk_callback_avx+0x1b2>
   .byte  196,65,84,89,222                    // vmulps        %ymm14,%ymm5,%ymm11
   .byte  196,65,44,88,211                    // vaddps        %ymm11,%ymm10,%ymm10
-  .byte  196,98,125,24,61,1,83,0,0           // vbroadcastss  0x5301(%rip),%ymm15        # 63fc <_sk_callback_avx+0x1b6>
+  .byte  196,98,125,24,61,0,83,0,0           // vbroadcastss  0x5300(%rip),%ymm15        # 6418 <_sk_callback_avx+0x1b6>
   .byte  196,65,76,89,223                    // vmulps        %ymm15,%ymm6,%ymm11
   .byte  196,193,44,88,195                   // vaddps        %ymm11,%ymm10,%ymm0
   .byte  196,65,60,89,221                    // vmulps        %ymm13,%ymm8,%ymm11
@@ -14990,7 +15021,7 @@ _sk_color_avx:
   .byte  196,65,44,95,207                    // vmaxps        %ymm15,%ymm10,%ymm9
   .byte  196,195,37,74,192,0                 // vblendvps     %ymm0,%ymm8,%ymm11,%ymm0
   .byte  196,65,124,95,199                   // vmaxps        %ymm15,%ymm0,%ymm8
-  .byte  196,226,125,24,5,200,81,0,0         // vbroadcastss  0x51c8(%rip),%ymm0        # 6400 <_sk_callback_avx+0x1ba>
+  .byte  196,226,125,24,5,199,81,0,0         // vbroadcastss  0x51c7(%rip),%ymm0        # 641c <_sk_callback_avx+0x1ba>
   .byte  197,124,92,215                      // vsubps        %ymm7,%ymm0,%ymm10
   .byte  197,172,89,84,36,168                // vmulps        -0x58(%rsp),%ymm10,%ymm2
   .byte  197,124,92,219                      // vsubps        %ymm3,%ymm0,%ymm11
@@ -15020,12 +15051,12 @@ _sk_luminosity_avx:
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
   .byte  197,100,89,196                      // vmulps        %ymm4,%ymm3,%ymm8
   .byte  197,100,89,205                      // vmulps        %ymm5,%ymm3,%ymm9
-  .byte  196,98,125,24,45,90,81,0,0          // vbroadcastss  0x515a(%rip),%ymm13        # 6404 <_sk_callback_avx+0x1be>
+  .byte  196,98,125,24,45,89,81,0,0          // vbroadcastss  0x5159(%rip),%ymm13        # 6420 <_sk_callback_avx+0x1be>
   .byte  196,65,108,89,213                   // vmulps        %ymm13,%ymm2,%ymm10
-  .byte  196,98,125,24,53,80,81,0,0          // vbroadcastss  0x5150(%rip),%ymm14        # 6408 <_sk_callback_avx+0x1c2>
+  .byte  196,98,125,24,53,79,81,0,0          // vbroadcastss  0x514f(%rip),%ymm14        # 6424 <_sk_callback_avx+0x1c2>
   .byte  196,65,116,89,222                   // vmulps        %ymm14,%ymm1,%ymm11
   .byte  196,65,44,88,211                    // vaddps        %ymm11,%ymm10,%ymm10
-  .byte  196,98,125,24,61,65,81,0,0          // vbroadcastss  0x5141(%rip),%ymm15        # 640c <_sk_callback_avx+0x1c6>
+  .byte  196,98,125,24,61,64,81,0,0          // vbroadcastss  0x5140(%rip),%ymm15        # 6428 <_sk_callback_avx+0x1c6>
   .byte  196,65,28,89,223                    // vmulps        %ymm15,%ymm12,%ymm11
   .byte  196,193,44,88,195                   // vaddps        %ymm11,%ymm10,%ymm0
   .byte  196,65,60,89,221                    // vmulps        %ymm13,%ymm8,%ymm11
@@ -15088,7 +15119,7 @@ _sk_luminosity_avx:
   .byte  196,65,44,95,207                    // vmaxps        %ymm15,%ymm10,%ymm9
   .byte  196,195,37,74,192,0                 // vblendvps     %ymm0,%ymm8,%ymm11,%ymm0
   .byte  196,65,124,95,199                   // vmaxps        %ymm15,%ymm0,%ymm8
-  .byte  196,226,125,24,5,8,80,0,0           // vbroadcastss  0x5008(%rip),%ymm0        # 6410 <_sk_callback_avx+0x1ca>
+  .byte  196,226,125,24,5,7,80,0,0           // vbroadcastss  0x5007(%rip),%ymm0        # 642c <_sk_callback_avx+0x1ca>
   .byte  197,124,92,215                      // vsubps        %ymm7,%ymm0,%ymm10
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  197,124,92,219                      // vsubps        %ymm3,%ymm0,%ymm11
@@ -15124,7 +15155,7 @@ HIDDEN _sk_clamp_1_avx
 .globl _sk_clamp_1_avx
 FUNCTION(_sk_clamp_1_avx)
 _sk_clamp_1_avx:
-  .byte  196,98,125,24,5,155,79,0,0          // vbroadcastss  0x4f9b(%rip),%ymm8        # 6414 <_sk_callback_avx+0x1ce>
+  .byte  196,98,125,24,5,154,79,0,0          // vbroadcastss  0x4f9a(%rip),%ymm8        # 6430 <_sk_callback_avx+0x1ce>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
@@ -15136,7 +15167,7 @@ HIDDEN _sk_clamp_a_avx
 .globl _sk_clamp_a_avx
 FUNCTION(_sk_clamp_a_avx)
 _sk_clamp_a_avx:
-  .byte  196,98,125,24,5,126,79,0,0          // vbroadcastss  0x4f7e(%rip),%ymm8        # 6418 <_sk_callback_avx+0x1d2>
+  .byte  196,98,125,24,5,125,79,0,0          // vbroadcastss  0x4f7d(%rip),%ymm8        # 6434 <_sk_callback_avx+0x1d2>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  197,252,93,195                      // vminps        %ymm3,%ymm0,%ymm0
   .byte  197,244,93,203                      // vminps        %ymm3,%ymm1,%ymm1
@@ -15222,7 +15253,7 @@ FUNCTION(_sk_unpremul_avx)
 _sk_unpremul_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,65,100,194,200,0                // vcmpeqps      %ymm8,%ymm3,%ymm9
-  .byte  196,98,125,24,21,198,78,0,0         // vbroadcastss  0x4ec6(%rip),%ymm10        # 641c <_sk_callback_avx+0x1d6>
+  .byte  196,98,125,24,21,197,78,0,0         // vbroadcastss  0x4ec5(%rip),%ymm10        # 6438 <_sk_callback_avx+0x1d6>
   .byte  197,44,94,211                       // vdivps        %ymm3,%ymm10,%ymm10
   .byte  196,67,45,74,192,144                // vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
@@ -15235,17 +15266,17 @@ HIDDEN _sk_from_srgb_avx
 .globl _sk_from_srgb_avx
 FUNCTION(_sk_from_srgb_avx)
 _sk_from_srgb_avx:
-  .byte  196,98,125,24,5,167,78,0,0          // vbroadcastss  0x4ea7(%rip),%ymm8        # 6420 <_sk_callback_avx+0x1da>
+  .byte  196,98,125,24,5,166,78,0,0          // vbroadcastss  0x4ea6(%rip),%ymm8        # 643c <_sk_callback_avx+0x1da>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  197,124,89,208                      // vmulps        %ymm0,%ymm0,%ymm10
-  .byte  196,98,125,24,29,153,78,0,0         // vbroadcastss  0x4e99(%rip),%ymm11        # 6424 <_sk_callback_avx+0x1de>
+  .byte  196,98,125,24,29,152,78,0,0         // vbroadcastss  0x4e98(%rip),%ymm11        # 6440 <_sk_callback_avx+0x1de>
   .byte  196,65,124,89,227                   // vmulps        %ymm11,%ymm0,%ymm12
-  .byte  196,98,125,24,45,143,78,0,0         // vbroadcastss  0x4e8f(%rip),%ymm13        # 6428 <_sk_callback_avx+0x1e2>
+  .byte  196,98,125,24,45,142,78,0,0         // vbroadcastss  0x4e8e(%rip),%ymm13        # 6444 <_sk_callback_avx+0x1e2>
   .byte  196,65,28,88,229                    // vaddps        %ymm13,%ymm12,%ymm12
   .byte  196,65,44,89,212                    // vmulps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,37,128,78,0,0         // vbroadcastss  0x4e80(%rip),%ymm12        # 642c <_sk_callback_avx+0x1e6>
+  .byte  196,98,125,24,37,127,78,0,0         // vbroadcastss  0x4e7f(%rip),%ymm12        # 6448 <_sk_callback_avx+0x1e6>
   .byte  196,65,44,88,212                    // vaddps        %ymm12,%ymm10,%ymm10
-  .byte  196,98,125,24,53,118,78,0,0         // vbroadcastss  0x4e76(%rip),%ymm14        # 6430 <_sk_callback_avx+0x1ea>
+  .byte  196,98,125,24,53,117,78,0,0         // vbroadcastss  0x4e75(%rip),%ymm14        # 644c <_sk_callback_avx+0x1ea>
   .byte  196,193,124,194,198,1               // vcmpltps      %ymm14,%ymm0,%ymm0
   .byte  196,195,45,74,193,0                 // vblendvps     %ymm0,%ymm9,%ymm10,%ymm0
   .byte  196,65,116,89,200                   // vmulps        %ymm8,%ymm1,%ymm9
@@ -15274,18 +15305,18 @@ _sk_to_srgb_avx:
   .byte  197,124,82,192                      // vrsqrtps      %ymm0,%ymm8
   .byte  196,65,124,83,200                   // vrcpps        %ymm8,%ymm9
   .byte  196,65,124,82,208                   // vrsqrtps      %ymm8,%ymm10
-  .byte  196,98,125,24,5,1,78,0,0            // vbroadcastss  0x4e01(%rip),%ymm8        # 6434 <_sk_callback_avx+0x1ee>
+  .byte  196,98,125,24,5,0,78,0,0            // vbroadcastss  0x4e00(%rip),%ymm8        # 6450 <_sk_callback_avx+0x1ee>
   .byte  196,65,124,89,216                   // vmulps        %ymm8,%ymm0,%ymm11
-  .byte  196,98,125,24,37,247,77,0,0         // vbroadcastss  0x4df7(%rip),%ymm12        # 6438 <_sk_callback_avx+0x1f2>
+  .byte  196,98,125,24,37,246,77,0,0         // vbroadcastss  0x4df6(%rip),%ymm12        # 6454 <_sk_callback_avx+0x1f2>
   .byte  196,65,52,89,204                    // vmulps        %ymm12,%ymm9,%ymm9
-  .byte  196,98,125,24,45,237,77,0,0         // vbroadcastss  0x4ded(%rip),%ymm13        # 643c <_sk_callback_avx+0x1f6>
+  .byte  196,98,125,24,45,236,77,0,0         // vbroadcastss  0x4dec(%rip),%ymm13        # 6458 <_sk_callback_avx+0x1f6>
   .byte  196,65,52,88,205                    // vaddps        %ymm13,%ymm9,%ymm9
-  .byte  196,98,125,24,53,227,77,0,0         // vbroadcastss  0x4de3(%rip),%ymm14        # 6440 <_sk_callback_avx+0x1fa>
+  .byte  196,98,125,24,53,226,77,0,0         // vbroadcastss  0x4de2(%rip),%ymm14        # 645c <_sk_callback_avx+0x1fa>
   .byte  196,65,44,89,214                    // vmulps        %ymm14,%ymm10,%ymm10
   .byte  196,65,44,88,201                    // vaddps        %ymm9,%ymm10,%ymm9
-  .byte  196,98,125,24,21,212,77,0,0         // vbroadcastss  0x4dd4(%rip),%ymm10        # 6444 <_sk_callback_avx+0x1fe>
+  .byte  196,98,125,24,21,211,77,0,0         // vbroadcastss  0x4dd3(%rip),%ymm10        # 6460 <_sk_callback_avx+0x1fe>
   .byte  196,65,44,93,201                    // vminps        %ymm9,%ymm10,%ymm9
-  .byte  196,98,125,24,61,202,77,0,0         // vbroadcastss  0x4dca(%rip),%ymm15        # 6448 <_sk_callback_avx+0x202>
+  .byte  196,98,125,24,61,201,77,0,0         // vbroadcastss  0x4dc9(%rip),%ymm15        # 6464 <_sk_callback_avx+0x202>
   .byte  196,193,124,194,199,1               // vcmpltps      %ymm15,%ymm0,%ymm0
   .byte  196,195,53,74,195,0                 // vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   .byte  197,124,82,201                      // vrsqrtps      %ymm1,%ymm9
@@ -15322,7 +15353,7 @@ _sk_rgb_to_hsl_avx:
   .byte  197,124,93,201                      // vminps        %ymm1,%ymm0,%ymm9
   .byte  197,52,93,202                       // vminps        %ymm2,%ymm9,%ymm9
   .byte  196,65,60,92,209                    // vsubps        %ymm9,%ymm8,%ymm10
-  .byte  196,98,125,24,29,48,77,0,0          // vbroadcastss  0x4d30(%rip),%ymm11        # 644c <_sk_callback_avx+0x206>
+  .byte  196,98,125,24,29,47,77,0,0          // vbroadcastss  0x4d2f(%rip),%ymm11        # 6468 <_sk_callback_avx+0x206>
   .byte  196,65,36,94,218                    // vdivps        %ymm10,%ymm11,%ymm11
   .byte  197,116,92,226                      // vsubps        %ymm2,%ymm1,%ymm12
   .byte  196,65,28,89,227                    // vmulps        %ymm11,%ymm12,%ymm12
@@ -15332,19 +15363,19 @@ _sk_rgb_to_hsl_avx:
   .byte  196,193,108,89,211                  // vmulps        %ymm11,%ymm2,%ymm2
   .byte  197,252,92,201                      // vsubps        %ymm1,%ymm0,%ymm1
   .byte  196,193,116,89,203                  // vmulps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,9,77,0,0           // vbroadcastss  0x4d09(%rip),%ymm11        # 6458 <_sk_callback_avx+0x212>
+  .byte  196,98,125,24,29,8,77,0,0           // vbroadcastss  0x4d08(%rip),%ymm11        # 6474 <_sk_callback_avx+0x212>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,247,76,0,0         // vbroadcastss  0x4cf7(%rip),%ymm11        # 6454 <_sk_callback_avx+0x20e>
+  .byte  196,98,125,24,29,246,76,0,0         // vbroadcastss  0x4cf6(%rip),%ymm11        # 6470 <_sk_callback_avx+0x20e>
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
   .byte  196,227,117,74,202,224              // vblendvps     %ymm14,%ymm2,%ymm1,%ymm1
-  .byte  196,226,125,24,21,223,76,0,0        // vbroadcastss  0x4cdf(%rip),%ymm2        # 6450 <_sk_callback_avx+0x20a>
+  .byte  196,226,125,24,21,222,76,0,0        // vbroadcastss  0x4cde(%rip),%ymm2        # 646c <_sk_callback_avx+0x20a>
   .byte  196,65,12,87,246                    // vxorps        %ymm14,%ymm14,%ymm14
   .byte  196,227,13,74,210,208               // vblendvps     %ymm13,%ymm2,%ymm14,%ymm2
   .byte  197,188,194,192,0                   // vcmpeqps      %ymm0,%ymm8,%ymm0
   .byte  196,193,108,88,212                  // vaddps        %ymm12,%ymm2,%ymm2
   .byte  196,227,117,74,194,0                // vblendvps     %ymm0,%ymm2,%ymm1,%ymm0
   .byte  196,193,60,88,201                   // vaddps        %ymm9,%ymm8,%ymm1
-  .byte  196,98,125,24,37,198,76,0,0         // vbroadcastss  0x4cc6(%rip),%ymm12        # 6460 <_sk_callback_avx+0x21a>
+  .byte  196,98,125,24,37,197,76,0,0         // vbroadcastss  0x4cc5(%rip),%ymm12        # 647c <_sk_callback_avx+0x21a>
   .byte  196,193,116,89,212                  // vmulps        %ymm12,%ymm1,%ymm2
   .byte  197,28,194,226,1                    // vcmpltps      %ymm2,%ymm12,%ymm12
   .byte  196,65,36,92,216                    // vsubps        %ymm8,%ymm11,%ymm11
@@ -15354,7 +15385,7 @@ _sk_rgb_to_hsl_avx:
   .byte  197,172,94,201                      // vdivps        %ymm1,%ymm10,%ymm1
   .byte  196,195,125,74,198,128              // vblendvps     %ymm8,%ymm14,%ymm0,%ymm0
   .byte  196,195,117,74,206,128              // vblendvps     %ymm8,%ymm14,%ymm1,%ymm1
-  .byte  196,98,125,24,5,137,76,0,0          // vbroadcastss  0x4c89(%rip),%ymm8        # 645c <_sk_callback_avx+0x216>
+  .byte  196,98,125,24,5,136,76,0,0          // vbroadcastss  0x4c88(%rip),%ymm8        # 6478 <_sk_callback_avx+0x216>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -15371,7 +15402,7 @@ _sk_hsl_to_rgb_avx:
   .byte  197,252,17,92,36,128                // vmovups       %ymm3,-0x80(%rsp)
   .byte  197,252,40,225                      // vmovaps       %ymm1,%ymm4
   .byte  197,252,40,216                      // vmovaps       %ymm0,%ymm3
-  .byte  196,98,125,24,5,86,76,0,0           // vbroadcastss  0x4c56(%rip),%ymm8        # 6464 <_sk_callback_avx+0x21e>
+  .byte  196,98,125,24,5,85,76,0,0           // vbroadcastss  0x4c55(%rip),%ymm8        # 6480 <_sk_callback_avx+0x21e>
   .byte  197,60,194,202,2                    // vcmpleps      %ymm2,%ymm8,%ymm9
   .byte  197,92,89,210                       // vmulps        %ymm2,%ymm4,%ymm10
   .byte  196,65,92,92,218                    // vsubps        %ymm10,%ymm4,%ymm11
@@ -15379,23 +15410,23 @@ _sk_hsl_to_rgb_avx:
   .byte  197,52,88,210                       // vaddps        %ymm2,%ymm9,%ymm10
   .byte  197,108,88,202                      // vaddps        %ymm2,%ymm2,%ymm9
   .byte  196,65,52,92,202                    // vsubps        %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,29,48,76,0,0          // vbroadcastss  0x4c30(%rip),%ymm11        # 6468 <_sk_callback_avx+0x222>
+  .byte  196,98,125,24,29,47,76,0,0          // vbroadcastss  0x4c2f(%rip),%ymm11        # 6484 <_sk_callback_avx+0x222>
   .byte  196,65,100,88,219                   // vaddps        %ymm11,%ymm3,%ymm11
   .byte  196,67,125,8,227,1                  // vroundps      $0x1,%ymm11,%ymm12
   .byte  196,65,36,92,252                    // vsubps        %ymm12,%ymm11,%ymm15
   .byte  196,65,44,92,217                    // vsubps        %ymm9,%ymm10,%ymm11
-  .byte  196,98,125,24,37,26,76,0,0          // vbroadcastss  0x4c1a(%rip),%ymm12        # 6470 <_sk_callback_avx+0x22a>
+  .byte  196,98,125,24,37,25,76,0,0          // vbroadcastss  0x4c19(%rip),%ymm12        # 648c <_sk_callback_avx+0x22a>
   .byte  196,193,4,89,196                    // vmulps        %ymm12,%ymm15,%ymm0
-  .byte  196,98,125,24,45,16,76,0,0          // vbroadcastss  0x4c10(%rip),%ymm13        # 6474 <_sk_callback_avx+0x22e>
+  .byte  196,98,125,24,45,15,76,0,0          // vbroadcastss  0x4c0f(%rip),%ymm13        # 6490 <_sk_callback_avx+0x22e>
   .byte  197,20,92,240                       // vsubps        %ymm0,%ymm13,%ymm14
   .byte  196,65,36,89,246                    // vmulps        %ymm14,%ymm11,%ymm14
   .byte  196,65,52,88,246                    // vaddps        %ymm14,%ymm9,%ymm14
-  .byte  196,226,125,24,13,241,75,0,0        // vbroadcastss  0x4bf1(%rip),%ymm1        # 646c <_sk_callback_avx+0x226>
+  .byte  196,226,125,24,13,240,75,0,0        // vbroadcastss  0x4bf0(%rip),%ymm1        # 6488 <_sk_callback_avx+0x226>
   .byte  196,193,116,194,255,2               // vcmpleps      %ymm15,%ymm1,%ymm7
   .byte  196,195,13,74,249,112               // vblendvps     %ymm7,%ymm9,%ymm14,%ymm7
   .byte  196,65,60,194,247,2                 // vcmpleps      %ymm15,%ymm8,%ymm14
   .byte  196,227,45,74,255,224               // vblendvps     %ymm14,%ymm7,%ymm10,%ymm7
-  .byte  196,98,125,24,53,220,75,0,0         // vbroadcastss  0x4bdc(%rip),%ymm14        # 6478 <_sk_callback_avx+0x232>
+  .byte  196,98,125,24,53,219,75,0,0         // vbroadcastss  0x4bdb(%rip),%ymm14        # 6494 <_sk_callback_avx+0x232>
   .byte  196,65,12,194,255,2                 // vcmpleps      %ymm15,%ymm14,%ymm15
   .byte  196,193,124,89,195                  // vmulps        %ymm11,%ymm0,%ymm0
   .byte  197,180,88,192                      // vaddps        %ymm0,%ymm9,%ymm0
@@ -15414,7 +15445,7 @@ _sk_hsl_to_rgb_avx:
   .byte  197,164,89,247                      // vmulps        %ymm7,%ymm11,%ymm6
   .byte  197,180,88,246                      // vaddps        %ymm6,%ymm9,%ymm6
   .byte  196,227,77,74,237,0                 // vblendvps     %ymm0,%ymm5,%ymm6,%ymm5
-  .byte  196,226,125,24,5,126,75,0,0         // vbroadcastss  0x4b7e(%rip),%ymm0        # 647c <_sk_callback_avx+0x236>
+  .byte  196,226,125,24,5,125,75,0,0         // vbroadcastss  0x4b7d(%rip),%ymm0        # 6498 <_sk_callback_avx+0x236>
   .byte  197,228,88,192                      // vaddps        %ymm0,%ymm3,%ymm0
   .byte  196,227,125,8,216,1                 // vroundps      $0x1,%ymm0,%ymm3
   .byte  197,252,92,195                      // vsubps        %ymm3,%ymm0,%ymm0
@@ -15466,14 +15497,14 @@ _sk_scale_u8_avx:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,68                              // jne           19f5 <_sk_scale_u8_avx+0x54>
+  .byte  117,68                              // jne           1a12 <_sk_scale_u8_avx+0x54>
   .byte  197,122,126,0                       // vmovq         (%rax),%xmm8
   .byte  196,66,121,49,200                   // vpmovzxbd     %xmm8,%xmm9
   .byte  196,67,121,4,192,229                // vpermilps     $0xe5,%xmm8,%xmm8
   .byte  196,66,121,49,192                   // vpmovzxbd     %xmm8,%xmm8
   .byte  196,67,53,24,192,1                  // vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,167,74,0,0         // vbroadcastss  0x4aa7(%rip),%ymm9        # 6480 <_sk_callback_avx+0x23a>
+  .byte  196,98,125,24,13,166,74,0,0         // vbroadcastss  0x4aa6(%rip),%ymm9        # 649c <_sk_callback_avx+0x23a>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
@@ -15491,9 +15522,9 @@ _sk_scale_u8_avx:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           19fd <_sk_scale_u8_avx+0x5c>
+  .byte  117,234                             // jne           1a1a <_sk_scale_u8_avx+0x5c>
   .byte  196,65,249,110,193                  // vmovq         %r9,%xmm8
-  .byte  235,155                             // jmp           19b5 <_sk_scale_u8_avx+0x14>
+  .byte  235,155                             // jmp           19d2 <_sk_scale_u8_avx+0x14>
 
 HIDDEN _sk_lerp_1_float_avx
 .globl _sk_lerp_1_float_avx
@@ -15525,14 +15556,14 @@ _sk_lerp_u8_avx:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,104                             // jne           1ad1 <_sk_lerp_u8_avx+0x78>
+  .byte  117,104                             // jne           1aee <_sk_lerp_u8_avx+0x78>
   .byte  197,122,126,0                       // vmovq         (%rax),%xmm8
   .byte  196,66,121,49,200                   // vpmovzxbd     %xmm8,%xmm9
   .byte  196,67,121,4,192,229                // vpermilps     $0xe5,%xmm8,%xmm8
   .byte  196,66,121,49,192                   // vpmovzxbd     %xmm8,%xmm8
   .byte  196,67,53,24,192,1                  // vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,13,243,73,0,0         // vbroadcastss  0x49f3(%rip),%ymm9        # 6484 <_sk_callback_avx+0x23e>
+  .byte  196,98,125,24,13,242,73,0,0         // vbroadcastss  0x49f2(%rip),%ymm9        # 64a0 <_sk_callback_avx+0x23e>
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
@@ -15558,9 +15589,9 @@ _sk_lerp_u8_avx:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           1ad9 <_sk_lerp_u8_avx+0x80>
+  .byte  117,234                             // jne           1af6 <_sk_lerp_u8_avx+0x80>
   .byte  196,65,249,110,193                  // vmovq         %r9,%xmm8
-  .byte  233,116,255,255,255                 // jmpq          1a6d <_sk_lerp_u8_avx+0x14>
+  .byte  233,116,255,255,255                 // jmpq          1a8a <_sk_lerp_u8_avx+0x14>
 
 HIDDEN _sk_lerp_565_avx
 .globl _sk_lerp_565_avx
@@ -15569,26 +15600,26 @@ _sk_lerp_565_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,208,0,0,0                    // jne           1bd7 <_sk_lerp_565_avx+0xde>
+  .byte  15,133,208,0,0,0                    // jne           1bf4 <_sk_lerp_565_avx+0xde>
   .byte  196,65,122,111,4,122                // vmovdqu       (%r10,%rdi,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  196,65,57,105,201                   // vpunpckhwd    %xmm9,%xmm8,%xmm9
   .byte  196,66,121,51,192                   // vpmovzxwd     %xmm8,%xmm8
   .byte  196,67,61,24,193,1                  // vinsertf128   $0x1,%xmm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,93,73,0,0          // vbroadcastss  0x495d(%rip),%ymm9        # 6488 <_sk_callback_avx+0x242>
+  .byte  196,98,125,24,13,92,73,0,0          // vbroadcastss  0x495c(%rip),%ymm9        # 64a4 <_sk_callback_avx+0x242>
   .byte  196,65,60,84,201                    // vandps        %ymm9,%ymm8,%ymm9
   .byte  196,65,124,91,201                   // vcvtdq2ps     %ymm9,%ymm9
-  .byte  196,98,125,24,21,78,73,0,0          // vbroadcastss  0x494e(%rip),%ymm10        # 648c <_sk_callback_avx+0x246>
+  .byte  196,98,125,24,21,77,73,0,0          // vbroadcastss  0x494d(%rip),%ymm10        # 64a8 <_sk_callback_avx+0x246>
   .byte  196,65,52,89,202                    // vmulps        %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,21,68,73,0,0          // vbroadcastss  0x4944(%rip),%ymm10        # 6490 <_sk_callback_avx+0x24a>
+  .byte  196,98,125,24,21,67,73,0,0          // vbroadcastss  0x4943(%rip),%ymm10        # 64ac <_sk_callback_avx+0x24a>
   .byte  196,65,60,84,210                    // vandps        %ymm10,%ymm8,%ymm10
   .byte  196,65,124,91,210                   // vcvtdq2ps     %ymm10,%ymm10
-  .byte  196,98,125,24,29,53,73,0,0          // vbroadcastss  0x4935(%rip),%ymm11        # 6494 <_sk_callback_avx+0x24e>
+  .byte  196,98,125,24,29,52,73,0,0          // vbroadcastss  0x4934(%rip),%ymm11        # 64b0 <_sk_callback_avx+0x24e>
   .byte  196,65,44,89,211                    // vmulps        %ymm11,%ymm10,%ymm10
-  .byte  196,98,125,24,29,43,73,0,0          // vbroadcastss  0x492b(%rip),%ymm11        # 6498 <_sk_callback_avx+0x252>
+  .byte  196,98,125,24,29,42,73,0,0          // vbroadcastss  0x492a(%rip),%ymm11        # 64b4 <_sk_callback_avx+0x252>
   .byte  196,65,60,84,195                    // vandps        %ymm11,%ymm8,%ymm8
   .byte  196,65,124,91,192                   // vcvtdq2ps     %ymm8,%ymm8
-  .byte  196,98,125,24,29,28,73,0,0          // vbroadcastss  0x491c(%rip),%ymm11        # 649c <_sk_callback_avx+0x256>
+  .byte  196,98,125,24,29,27,73,0,0          // vbroadcastss  0x491b(%rip),%ymm11        # 64b8 <_sk_callback_avx+0x256>
   .byte  196,65,60,89,195                    // vmulps        %ymm11,%ymm8,%ymm8
   .byte  197,252,92,196                      // vsubps        %ymm4,%ymm0,%ymm0
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
@@ -15615,9 +15646,9 @@ _sk_lerp_565_avx:
   .byte  196,65,57,239,192                   // vpxor         %xmm8,%xmm8,%xmm8
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,29,255,255,255               // ja            1b0d <_sk_lerp_565_avx+0x14>
+  .byte  15,135,29,255,255,255               // ja            1b2a <_sk_lerp_565_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,77,0,0,0                  // lea           0x4d(%rip),%r9        # 1c48 <_sk_lerp_565_avx+0x14f>
+  .byte  76,141,13,76,0,0,0                  // lea           0x4c(%rip),%r9        # 1c64 <_sk_lerp_565_avx+0x14e>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -15629,26 +15660,28 @@ _sk_lerp_565_avx:
   .byte  196,65,57,196,68,122,4,2            // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,68,122,2,1            // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
   .byte  196,65,57,196,4,122,0               // vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
-  .byte  233,200,254,255,255                 // jmpq          1b0d <_sk_lerp_565_avx+0x14>
-  .byte  15,31,0                             // nopl          (%rax)
-  .byte  241                                 // icebp
+  .byte  233,200,254,255,255                 // jmpq          1b2a <_sk_lerp_565_avx+0x14>
+  .byte  102,144                             // xchg          %ax,%ax
+  .byte  242,255                             // repnz         (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
+  .byte  234                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  233,255,255,255,225                 // jmpq          ffffffffe2001c50 <_sk_callback_avx+0xffffffffe1ffba0a>
+  .byte  255                                 // (bad)
+  .byte  255,226                             // jmpq          *%rdx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  217,255                             // fcos
+  .byte  218,255                             // (bad)
   .byte  255                                 // (bad)
-  .byte  255,209                             // callq         *%rcx
+  .byte  255,210                             // callq         *%rdx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,201                             // dec           %ecx
+  .byte  255,202                             // dec           %edx
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  188                                 // .byte         0xbc
+  .byte  189                                 // .byte         0xbd
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
@@ -15660,7 +15693,7 @@ _sk_load_tables_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,26,2,0,0                     // jne           1e8c <_sk_load_tables_avx+0x228>
+  .byte  15,133,26,2,0,0                     // jne           1ea8 <_sk_load_tables_avx+0x228>
   .byte  196,65,124,16,4,184                 // vmovups       (%r8,%rdi,4),%ymm8
   .byte  85                                  // push          %rbp
   .byte  65,87                               // push          %r15
@@ -15668,7 +15701,7 @@ _sk_load_tables_avx:
   .byte  65,85                               // push          %r13
   .byte  65,84                               // push          %r12
   .byte  83                                  // push          %rbx
-  .byte  197,124,40,13,246,74,0,0            // vmovaps       0x4af6(%rip),%ymm9        # 6780 <_sk_callback_avx+0x53a>
+  .byte  197,124,40,13,250,74,0,0            // vmovaps       0x4afa(%rip),%ymm9        # 67a0 <_sk_callback_avx+0x53e>
   .byte  196,193,60,84,193                   // vandps        %ymm9,%ymm8,%ymm0
   .byte  196,193,249,126,193                 // vmovq         %xmm0,%r9
   .byte  69,137,203                          // mov           %r9d,%r11d
@@ -15760,7 +15793,7 @@ _sk_load_tables_avx:
   .byte  196,193,97,114,210,24               // vpsrld        $0x18,%xmm10,%xmm3
   .byte  196,227,61,24,219,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,39,70,0,0           // vbroadcastss  0x4627(%rip),%ymm8        # 64a0 <_sk_callback_avx+0x25a>
+  .byte  196,98,125,24,5,39,70,0,0           // vbroadcastss  0x4627(%rip),%ymm8        # 64bc <_sk_callback_avx+0x25a>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -15775,9 +15808,9 @@ _sk_load_tables_avx:
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  65,254,201                          // dec           %r9b
   .byte  65,128,249,6                        // cmp           $0x6,%r9b
-  .byte  15,135,211,253,255,255              // ja            1c78 <_sk_load_tables_avx+0x14>
+  .byte  15,135,211,253,255,255              // ja            1c94 <_sk_load_tables_avx+0x14>
   .byte  69,15,182,201                       // movzbl        %r9b,%r9d
-  .byte  76,141,21,140,0,0,0                 // lea           0x8c(%rip),%r10        # 1f3c <_sk_load_tables_avx+0x2d8>
+  .byte  76,141,21,140,0,0,0                 // lea           0x8c(%rip),%r10        # 1f58 <_sk_load_tables_avx+0x2d8>
   .byte  79,99,12,138                        // movslq        (%r10,%r9,4),%r9
   .byte  77,1,209                            // add           %r10,%r9
   .byte  65,255,225                          // jmpq          *%r9
@@ -15800,7 +15833,7 @@ _sk_load_tables_avx:
   .byte  196,99,61,12,192,15                 // vblendps      $0xf,%ymm0,%ymm8,%ymm8
   .byte  196,195,57,34,4,184,0               // vpinsrd       $0x0,(%r8,%rdi,4),%xmm8,%xmm0
   .byte  196,99,61,12,192,15                 // vblendps      $0xf,%ymm0,%ymm8,%ymm8
-  .byte  233,62,253,255,255                  // jmpq          1c78 <_sk_load_tables_avx+0x14>
+  .byte  233,62,253,255,255                  // jmpq          1c94 <_sk_load_tables_avx+0x14>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  236                                 // in            (%dx),%al
   .byte  255                                 // (bad)
@@ -15818,7 +15851,7 @@ _sk_load_tables_avx:
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  126,255                             // jle           1f55 <_sk_load_tables_avx+0x2f1>
+  .byte  126,255                             // jle           1f71 <_sk_load_tables_avx+0x2f1>
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
 
@@ -15830,7 +15863,7 @@ _sk_load_tables_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,113,2,0,0                    // jne           21df <_sk_load_tables_u16_be_avx+0x287>
+  .byte  15,133,113,2,0,0                    // jne           21fb <_sk_load_tables_u16_be_avx+0x287>
   .byte  196,1,121,16,4,72                   // vmovupd       (%r8,%r9,2),%xmm8
   .byte  196,129,121,16,84,72,16             // vmovupd       0x10(%r8,%r9,2),%xmm2
   .byte  196,129,121,16,92,72,32             // vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -15852,7 +15885,7 @@ _sk_load_tables_u16_be_avx:
   .byte  197,177,108,208                     // vpunpcklqdq   %xmm0,%xmm9,%xmm2
   .byte  197,177,109,200                     // vpunpckhqdq   %xmm0,%xmm9,%xmm1
   .byte  196,65,57,108,212                   // vpunpcklqdq   %xmm12,%xmm8,%xmm10
-  .byte  197,121,111,29,54,72,0,0            // vmovdqa       0x4836(%rip),%xmm11        # 6800 <_sk_callback_avx+0x5ba>
+  .byte  197,121,111,29,58,72,0,0            // vmovdqa       0x483a(%rip),%xmm11        # 6820 <_sk_callback_avx+0x5be>
   .byte  196,193,105,219,195                 // vpand         %xmm11,%xmm2,%xmm0
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  196,193,121,105,209                 // vpunpckhwd    %xmm9,%xmm0,%xmm2
@@ -15951,7 +15984,7 @@ _sk_load_tables_u16_be_avx:
   .byte  196,226,121,51,219                  // vpmovzxwd     %xmm3,%xmm3
   .byte  196,195,101,24,216,1                // vinsertf128   $0x1,%xmm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,216,66,0,0          // vbroadcastss  0x42d8(%rip),%ymm8        # 64a4 <_sk_callback_avx+0x25e>
+  .byte  196,98,125,24,5,216,66,0,0          // vbroadcastss  0x42d8(%rip),%ymm8        # 64c0 <_sk_callback_avx+0x25e>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -15964,29 +15997,29 @@ _sk_load_tables_u16_be_avx:
   .byte  196,1,123,16,4,72                   // vmovsd        (%r8,%r9,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            2245 <_sk_load_tables_u16_be_avx+0x2ed>
+  .byte  116,85                              // je            2261 <_sk_load_tables_u16_be_avx+0x2ed>
   .byte  196,1,57,22,68,72,8                 // vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            2245 <_sk_load_tables_u16_be_avx+0x2ed>
+  .byte  114,72                              // jb            2261 <_sk_load_tables_u16_be_avx+0x2ed>
   .byte  196,129,123,16,84,72,16             // vmovsd        0x10(%r8,%r9,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            2252 <_sk_load_tables_u16_be_avx+0x2fa>
+  .byte  116,72                              // je            226e <_sk_load_tables_u16_be_avx+0x2fa>
   .byte  196,129,105,22,84,72,24             // vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            2252 <_sk_load_tables_u16_be_avx+0x2fa>
+  .byte  114,59                              // jb            226e <_sk_load_tables_u16_be_avx+0x2fa>
   .byte  196,129,123,16,92,72,32             // vmovsd        0x20(%r8,%r9,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,97,253,255,255               // je            1f89 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  15,132,97,253,255,255               // je            1fa5 <_sk_load_tables_u16_be_avx+0x31>
   .byte  196,129,97,22,92,72,40              // vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,80,253,255,255               // jb            1f89 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  15,130,80,253,255,255               // jb            1fa5 <_sk_load_tables_u16_be_avx+0x31>
   .byte  196,1,122,126,76,72,48              // vmovq         0x30(%r8,%r9,2),%xmm9
-  .byte  233,68,253,255,255                  // jmpq          1f89 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  233,68,253,255,255                  // jmpq          1fa5 <_sk_load_tables_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,55,253,255,255                  // jmpq          1f89 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  233,55,253,255,255                  // jmpq          1fa5 <_sk_load_tables_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,46,253,255,255                  // jmpq          1f89 <_sk_load_tables_u16_be_avx+0x31>
+  .byte  233,46,253,255,255                  // jmpq          1fa5 <_sk_load_tables_u16_be_avx+0x31>
 
 HIDDEN _sk_load_tables_rgb_u16_be_avx
 .globl _sk_load_tables_rgb_u16_be_avx
@@ -15996,7 +16029,7 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,127                       // lea           (%rdi,%rdi,2),%r9
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,93,2,0,0                     // jne           24ca <_sk_load_tables_rgb_u16_be_avx+0x26f>
+  .byte  15,133,93,2,0,0                     // jne           24e6 <_sk_load_tables_rgb_u16_be_avx+0x26f>
   .byte  196,129,122,111,4,72                // vmovdqu       (%r8,%r9,2),%xmm0
   .byte  196,129,122,111,84,72,12            // vmovdqu       0xc(%r8,%r9,2),%xmm2
   .byte  196,129,122,111,76,72,24            // vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -16023,7 +16056,7 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  197,185,108,202                     // vpunpcklqdq   %xmm2,%xmm8,%xmm1
   .byte  197,185,109,210                     // vpunpckhqdq   %xmm2,%xmm8,%xmm2
   .byte  197,121,108,195                     // vpunpcklqdq   %xmm3,%xmm0,%xmm8
-  .byte  197,121,111,13,47,69,0,0            // vmovdqa       0x452f(%rip),%xmm9        # 6810 <_sk_callback_avx+0x5ca>
+  .byte  197,121,111,13,51,69,0,0            // vmovdqa       0x4533(%rip),%xmm9        # 6830 <_sk_callback_avx+0x5ce>
   .byte  196,193,113,219,193                 // vpand         %xmm9,%xmm1,%xmm0
   .byte  196,65,41,239,210                   // vpxor         %xmm10,%xmm10,%xmm10
   .byte  196,193,121,105,202                 // vpunpckhwd    %xmm10,%xmm0,%xmm1
@@ -16115,7 +16148,7 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  196,227,105,33,211,48               // vinsertps     $0x30,%xmm3,%xmm2,%xmm2
   .byte  196,195,109,24,208,1                // vinsertf128   $0x1,%xmm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,234,63,0,0        // vbroadcastss  0x3fea(%rip),%ymm3        # 64a8 <_sk_callback_avx+0x262>
+  .byte  196,226,125,24,29,234,63,0,0        // vbroadcastss  0x3fea(%rip),%ymm3        # 64c4 <_sk_callback_avx+0x262>
   .byte  91                                  // pop           %rbx
   .byte  65,92                               // pop           %r12
   .byte  65,93                               // pop           %r13
@@ -16126,36 +16159,36 @@ _sk_load_tables_rgb_u16_be_avx:
   .byte  196,129,121,110,4,72                // vmovd         (%r8,%r9,2),%xmm0
   .byte  196,129,121,196,68,72,4,2           // vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           24e3 <_sk_load_tables_rgb_u16_be_avx+0x288>
-  .byte  233,190,253,255,255                 // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  117,5                               // jne           24ff <_sk_load_tables_rgb_u16_be_avx+0x288>
+  .byte  233,190,253,255,255                 // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
   .byte  196,129,121,110,76,72,6             // vmovd         0x6(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,68,72,10,2            // vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            2512 <_sk_load_tables_rgb_u16_be_avx+0x2b7>
+  .byte  114,26                              // jb            252e <_sk_load_tables_rgb_u16_be_avx+0x2b7>
   .byte  196,129,121,110,76,72,12            // vmovd         0xc(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,84,72,16,2          // vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           2517 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
-  .byte  233,143,253,255,255                 // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  .byte  233,138,253,255,255                 // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           2533 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
+  .byte  233,143,253,255,255                 // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,138,253,255,255                 // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
   .byte  196,129,121,110,76,72,18            // vmovd         0x12(%r8,%r9,2),%xmm1
   .byte  196,1,113,196,76,72,22,2            // vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            2546 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
+  .byte  114,26                              // jb            2562 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
   .byte  196,129,121,110,76,72,24            // vmovd         0x18(%r8,%r9,2),%xmm1
   .byte  196,129,113,196,76,72,28,2          // vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           254b <_sk_load_tables_rgb_u16_be_avx+0x2f0>
-  .byte  233,91,253,255,255                  // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  .byte  233,86,253,255,255                  // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           2567 <_sk_load_tables_rgb_u16_be_avx+0x2f0>
+  .byte  233,91,253,255,255                  // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,86,253,255,255                  // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
   .byte  196,129,121,110,92,72,30            // vmovd         0x1e(%r8,%r9,2),%xmm3
   .byte  196,1,97,196,92,72,34,2             // vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            2574 <_sk_load_tables_rgb_u16_be_avx+0x319>
+  .byte  114,20                              // jb            2590 <_sk_load_tables_rgb_u16_be_avx+0x319>
   .byte  196,129,121,110,92,72,36            // vmovd         0x24(%r8,%r9,2),%xmm3
   .byte  196,129,97,196,92,72,40,2           // vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  .byte  233,45,253,255,255                  // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  .byte  233,40,253,255,255                  // jmpq          22a1 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,45,253,255,255                  // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
+  .byte  233,40,253,255,255                  // jmpq          22bd <_sk_load_tables_rgb_u16_be_avx+0x46>
 
 HIDDEN _sk_byte_tables_avx
 .globl _sk_byte_tables_avx
@@ -16168,7 +16201,7 @@ _sk_byte_tables_avx:
   .byte  65,84                               // push          %r12
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,30,63,0,0           // vbroadcastss  0x3f1e(%rip),%ymm8        # 64ac <_sk_callback_avx+0x266>
+  .byte  196,98,125,24,5,30,63,0,0           // vbroadcastss  0x3f1e(%rip),%ymm8        # 64c8 <_sk_callback_avx+0x266>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,195,249,22,192,1                // vpextrq       $0x1,%xmm0,%r8
@@ -16205,7 +16238,7 @@ _sk_byte_tables_avx:
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,53,24,192,1                 // vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,108,62,0,0         // vbroadcastss  0x3e6c(%rip),%ymm9        # 64b0 <_sk_callback_avx+0x26a>
+  .byte  196,98,125,24,13,108,62,0,0         // vbroadcastss  0x3e6c(%rip),%ymm9        # 64cc <_sk_callback_avx+0x26a>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -16367,7 +16400,7 @@ _sk_byte_tables_rgb_avx:
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,53,24,192,1                 // vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,146,59,0,0         // vbroadcastss  0x3b92(%rip),%ymm9        # 64b4 <_sk_callback_avx+0x26e>
+  .byte  196,98,125,24,13,146,59,0,0         // vbroadcastss  0x3b92(%rip),%ymm9        # 64d0 <_sk_callback_avx+0x26e>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  197,188,89,201                      // vmulps        %ymm1,%ymm8,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
@@ -16664,36 +16697,36 @@ _sk_parametric_r_avx:
   .byte  196,193,124,88,195                  // vaddps        %ymm11,%ymm0,%ymm0
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,216                      // vcvtdq2ps     %ymm0,%ymm11
-  .byte  196,98,125,24,37,240,54,0,0         // vbroadcastss  0x36f0(%rip),%ymm12        # 64b8 <_sk_callback_avx+0x272>
+  .byte  196,98,125,24,37,240,54,0,0         // vbroadcastss  0x36f0(%rip),%ymm12        # 64d4 <_sk_callback_avx+0x272>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,230,54,0,0         // vbroadcastss  0x36e6(%rip),%ymm12        # 64bc <_sk_callback_avx+0x276>
+  .byte  196,98,125,24,37,230,54,0,0         // vbroadcastss  0x36e6(%rip),%ymm12        # 64d8 <_sk_callback_avx+0x276>
   .byte  196,193,124,84,196                  // vandps        %ymm12,%ymm0,%ymm0
-  .byte  196,98,125,24,37,220,54,0,0         // vbroadcastss  0x36dc(%rip),%ymm12        # 64c0 <_sk_callback_avx+0x27a>
+  .byte  196,98,125,24,37,220,54,0,0         // vbroadcastss  0x36dc(%rip),%ymm12        # 64dc <_sk_callback_avx+0x27a>
   .byte  196,193,124,86,196                  // vorps         %ymm12,%ymm0,%ymm0
-  .byte  196,98,125,24,37,210,54,0,0         // vbroadcastss  0x36d2(%rip),%ymm12        # 64c4 <_sk_callback_avx+0x27e>
+  .byte  196,98,125,24,37,210,54,0,0         // vbroadcastss  0x36d2(%rip),%ymm12        # 64e0 <_sk_callback_avx+0x27e>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,200,54,0,0         // vbroadcastss  0x36c8(%rip),%ymm12        # 64c8 <_sk_callback_avx+0x282>
+  .byte  196,98,125,24,37,200,54,0,0         // vbroadcastss  0x36c8(%rip),%ymm12        # 64e4 <_sk_callback_avx+0x282>
   .byte  196,65,124,89,228                   // vmulps        %ymm12,%ymm0,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,185,54,0,0         // vbroadcastss  0x36b9(%rip),%ymm12        # 64cc <_sk_callback_avx+0x286>
+  .byte  196,98,125,24,37,185,54,0,0         // vbroadcastss  0x36b9(%rip),%ymm12        # 64e8 <_sk_callback_avx+0x286>
   .byte  196,193,124,88,196                  // vaddps        %ymm12,%ymm0,%ymm0
-  .byte  196,98,125,24,37,175,54,0,0         // vbroadcastss  0x36af(%rip),%ymm12        # 64d0 <_sk_callback_avx+0x28a>
+  .byte  196,98,125,24,37,175,54,0,0         // vbroadcastss  0x36af(%rip),%ymm12        # 64ec <_sk_callback_avx+0x28a>
   .byte  197,156,94,192                      // vdivps        %ymm0,%ymm12,%ymm0
   .byte  197,164,92,192                      // vsubps        %ymm0,%ymm11,%ymm0
   .byte  197,172,89,192                      // vmulps        %ymm0,%ymm10,%ymm0
   .byte  196,99,125,8,208,1                  // vroundps      $0x1,%ymm0,%ymm10
   .byte  196,65,124,92,210                   // vsubps        %ymm10,%ymm0,%ymm10
-  .byte  196,98,125,24,29,147,54,0,0         // vbroadcastss  0x3693(%rip),%ymm11        # 64d4 <_sk_callback_avx+0x28e>
+  .byte  196,98,125,24,29,147,54,0,0         // vbroadcastss  0x3693(%rip),%ymm11        # 64f0 <_sk_callback_avx+0x28e>
   .byte  196,193,124,88,195                  // vaddps        %ymm11,%ymm0,%ymm0
-  .byte  196,98,125,24,29,137,54,0,0         // vbroadcastss  0x3689(%rip),%ymm11        # 64d8 <_sk_callback_avx+0x292>
+  .byte  196,98,125,24,29,137,54,0,0         // vbroadcastss  0x3689(%rip),%ymm11        # 64f4 <_sk_callback_avx+0x292>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,124,92,195                  // vsubps        %ymm11,%ymm0,%ymm0
-  .byte  196,98,125,24,29,122,54,0,0         // vbroadcastss  0x367a(%rip),%ymm11        # 64dc <_sk_callback_avx+0x296>
+  .byte  196,98,125,24,29,122,54,0,0         // vbroadcastss  0x367a(%rip),%ymm11        # 64f8 <_sk_callback_avx+0x296>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,112,54,0,0         // vbroadcastss  0x3670(%rip),%ymm11        # 64e0 <_sk_callback_avx+0x29a>
+  .byte  196,98,125,24,29,112,54,0,0         // vbroadcastss  0x3670(%rip),%ymm11        # 64fc <_sk_callback_avx+0x29a>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,124,88,194                  // vaddps        %ymm10,%ymm0,%ymm0
-  .byte  196,98,125,24,21,97,54,0,0          // vbroadcastss  0x3661(%rip),%ymm10        # 64e4 <_sk_callback_avx+0x29e>
+  .byte  196,98,125,24,21,97,54,0,0          // vbroadcastss  0x3661(%rip),%ymm10        # 6500 <_sk_callback_avx+0x29e>
   .byte  196,193,124,89,194                  // vmulps        %ymm10,%ymm0,%ymm0
   .byte  197,253,91,192                      // vcvtps2dq     %ymm0,%ymm0
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16701,7 +16734,7 @@ _sk_parametric_r_avx:
   .byte  196,195,125,74,193,128              // vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,124,95,192                  // vmaxps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,56,54,0,0           // vbroadcastss  0x3638(%rip),%ymm8        # 64e8 <_sk_callback_avx+0x2a2>
+  .byte  196,98,125,24,5,56,54,0,0           // vbroadcastss  0x3638(%rip),%ymm8        # 6504 <_sk_callback_avx+0x2a2>
   .byte  196,193,124,93,192                  // vminps        %ymm8,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16723,36 +16756,36 @@ _sk_parametric_g_avx:
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,217                      // vcvtdq2ps     %ymm1,%ymm11
-  .byte  196,98,125,24,37,233,53,0,0         // vbroadcastss  0x35e9(%rip),%ymm12        # 64ec <_sk_callback_avx+0x2a6>
+  .byte  196,98,125,24,37,233,53,0,0         // vbroadcastss  0x35e9(%rip),%ymm12        # 6508 <_sk_callback_avx+0x2a6>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,223,53,0,0         // vbroadcastss  0x35df(%rip),%ymm12        # 64f0 <_sk_callback_avx+0x2aa>
+  .byte  196,98,125,24,37,223,53,0,0         // vbroadcastss  0x35df(%rip),%ymm12        # 650c <_sk_callback_avx+0x2aa>
   .byte  196,193,116,84,204                  // vandps        %ymm12,%ymm1,%ymm1
-  .byte  196,98,125,24,37,213,53,0,0         // vbroadcastss  0x35d5(%rip),%ymm12        # 64f4 <_sk_callback_avx+0x2ae>
+  .byte  196,98,125,24,37,213,53,0,0         // vbroadcastss  0x35d5(%rip),%ymm12        # 6510 <_sk_callback_avx+0x2ae>
   .byte  196,193,116,86,204                  // vorps         %ymm12,%ymm1,%ymm1
-  .byte  196,98,125,24,37,203,53,0,0         // vbroadcastss  0x35cb(%rip),%ymm12        # 64f8 <_sk_callback_avx+0x2b2>
+  .byte  196,98,125,24,37,203,53,0,0         // vbroadcastss  0x35cb(%rip),%ymm12        # 6514 <_sk_callback_avx+0x2b2>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,193,53,0,0         // vbroadcastss  0x35c1(%rip),%ymm12        # 64fc <_sk_callback_avx+0x2b6>
+  .byte  196,98,125,24,37,193,53,0,0         // vbroadcastss  0x35c1(%rip),%ymm12        # 6518 <_sk_callback_avx+0x2b6>
   .byte  196,65,116,89,228                   // vmulps        %ymm12,%ymm1,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,178,53,0,0         // vbroadcastss  0x35b2(%rip),%ymm12        # 6500 <_sk_callback_avx+0x2ba>
+  .byte  196,98,125,24,37,178,53,0,0         // vbroadcastss  0x35b2(%rip),%ymm12        # 651c <_sk_callback_avx+0x2ba>
   .byte  196,193,116,88,204                  // vaddps        %ymm12,%ymm1,%ymm1
-  .byte  196,98,125,24,37,168,53,0,0         // vbroadcastss  0x35a8(%rip),%ymm12        # 6504 <_sk_callback_avx+0x2be>
+  .byte  196,98,125,24,37,168,53,0,0         // vbroadcastss  0x35a8(%rip),%ymm12        # 6520 <_sk_callback_avx+0x2be>
   .byte  197,156,94,201                      // vdivps        %ymm1,%ymm12,%ymm1
   .byte  197,164,92,201                      // vsubps        %ymm1,%ymm11,%ymm1
   .byte  197,172,89,201                      // vmulps        %ymm1,%ymm10,%ymm1
   .byte  196,99,125,8,209,1                  // vroundps      $0x1,%ymm1,%ymm10
   .byte  196,65,116,92,210                   // vsubps        %ymm10,%ymm1,%ymm10
-  .byte  196,98,125,24,29,140,53,0,0         // vbroadcastss  0x358c(%rip),%ymm11        # 6508 <_sk_callback_avx+0x2c2>
+  .byte  196,98,125,24,29,140,53,0,0         // vbroadcastss  0x358c(%rip),%ymm11        # 6524 <_sk_callback_avx+0x2c2>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,130,53,0,0         // vbroadcastss  0x3582(%rip),%ymm11        # 650c <_sk_callback_avx+0x2c6>
+  .byte  196,98,125,24,29,130,53,0,0         // vbroadcastss  0x3582(%rip),%ymm11        # 6528 <_sk_callback_avx+0x2c6>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,116,92,203                  // vsubps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,29,115,53,0,0         // vbroadcastss  0x3573(%rip),%ymm11        # 6510 <_sk_callback_avx+0x2ca>
+  .byte  196,98,125,24,29,115,53,0,0         // vbroadcastss  0x3573(%rip),%ymm11        # 652c <_sk_callback_avx+0x2ca>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,105,53,0,0         // vbroadcastss  0x3569(%rip),%ymm11        # 6514 <_sk_callback_avx+0x2ce>
+  .byte  196,98,125,24,29,105,53,0,0         // vbroadcastss  0x3569(%rip),%ymm11        # 6530 <_sk_callback_avx+0x2ce>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,116,88,202                  // vaddps        %ymm10,%ymm1,%ymm1
-  .byte  196,98,125,24,21,90,53,0,0          // vbroadcastss  0x355a(%rip),%ymm10        # 6518 <_sk_callback_avx+0x2d2>
+  .byte  196,98,125,24,21,90,53,0,0          // vbroadcastss  0x355a(%rip),%ymm10        # 6534 <_sk_callback_avx+0x2d2>
   .byte  196,193,116,89,202                  // vmulps        %ymm10,%ymm1,%ymm1
   .byte  197,253,91,201                      // vcvtps2dq     %ymm1,%ymm1
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16760,7 +16793,7 @@ _sk_parametric_g_avx:
   .byte  196,195,117,74,201,128              // vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,116,95,200                  // vmaxps        %ymm8,%ymm1,%ymm1
-  .byte  196,98,125,24,5,49,53,0,0           // vbroadcastss  0x3531(%rip),%ymm8        # 651c <_sk_callback_avx+0x2d6>
+  .byte  196,98,125,24,5,49,53,0,0           // vbroadcastss  0x3531(%rip),%ymm8        # 6538 <_sk_callback_avx+0x2d6>
   .byte  196,193,116,93,200                  // vminps        %ymm8,%ymm1,%ymm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16782,36 +16815,36 @@ _sk_parametric_b_avx:
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,218                      // vcvtdq2ps     %ymm2,%ymm11
-  .byte  196,98,125,24,37,226,52,0,0         // vbroadcastss  0x34e2(%rip),%ymm12        # 6520 <_sk_callback_avx+0x2da>
+  .byte  196,98,125,24,37,226,52,0,0         // vbroadcastss  0x34e2(%rip),%ymm12        # 653c <_sk_callback_avx+0x2da>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,216,52,0,0         // vbroadcastss  0x34d8(%rip),%ymm12        # 6524 <_sk_callback_avx+0x2de>
+  .byte  196,98,125,24,37,216,52,0,0         // vbroadcastss  0x34d8(%rip),%ymm12        # 6540 <_sk_callback_avx+0x2de>
   .byte  196,193,108,84,212                  // vandps        %ymm12,%ymm2,%ymm2
-  .byte  196,98,125,24,37,206,52,0,0         // vbroadcastss  0x34ce(%rip),%ymm12        # 6528 <_sk_callback_avx+0x2e2>
+  .byte  196,98,125,24,37,206,52,0,0         // vbroadcastss  0x34ce(%rip),%ymm12        # 6544 <_sk_callback_avx+0x2e2>
   .byte  196,193,108,86,212                  // vorps         %ymm12,%ymm2,%ymm2
-  .byte  196,98,125,24,37,196,52,0,0         // vbroadcastss  0x34c4(%rip),%ymm12        # 652c <_sk_callback_avx+0x2e6>
+  .byte  196,98,125,24,37,196,52,0,0         // vbroadcastss  0x34c4(%rip),%ymm12        # 6548 <_sk_callback_avx+0x2e6>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,186,52,0,0         // vbroadcastss  0x34ba(%rip),%ymm12        # 6530 <_sk_callback_avx+0x2ea>
+  .byte  196,98,125,24,37,186,52,0,0         // vbroadcastss  0x34ba(%rip),%ymm12        # 654c <_sk_callback_avx+0x2ea>
   .byte  196,65,108,89,228                   // vmulps        %ymm12,%ymm2,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,171,52,0,0         // vbroadcastss  0x34ab(%rip),%ymm12        # 6534 <_sk_callback_avx+0x2ee>
+  .byte  196,98,125,24,37,171,52,0,0         // vbroadcastss  0x34ab(%rip),%ymm12        # 6550 <_sk_callback_avx+0x2ee>
   .byte  196,193,108,88,212                  // vaddps        %ymm12,%ymm2,%ymm2
-  .byte  196,98,125,24,37,161,52,0,0         // vbroadcastss  0x34a1(%rip),%ymm12        # 6538 <_sk_callback_avx+0x2f2>
+  .byte  196,98,125,24,37,161,52,0,0         // vbroadcastss  0x34a1(%rip),%ymm12        # 6554 <_sk_callback_avx+0x2f2>
   .byte  197,156,94,210                      // vdivps        %ymm2,%ymm12,%ymm2
   .byte  197,164,92,210                      // vsubps        %ymm2,%ymm11,%ymm2
   .byte  197,172,89,210                      // vmulps        %ymm2,%ymm10,%ymm2
   .byte  196,99,125,8,210,1                  // vroundps      $0x1,%ymm2,%ymm10
   .byte  196,65,108,92,210                   // vsubps        %ymm10,%ymm2,%ymm10
-  .byte  196,98,125,24,29,133,52,0,0         // vbroadcastss  0x3485(%rip),%ymm11        # 653c <_sk_callback_avx+0x2f6>
+  .byte  196,98,125,24,29,133,52,0,0         // vbroadcastss  0x3485(%rip),%ymm11        # 6558 <_sk_callback_avx+0x2f6>
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
-  .byte  196,98,125,24,29,123,52,0,0         // vbroadcastss  0x347b(%rip),%ymm11        # 6540 <_sk_callback_avx+0x2fa>
+  .byte  196,98,125,24,29,123,52,0,0         // vbroadcastss  0x347b(%rip),%ymm11        # 655c <_sk_callback_avx+0x2fa>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,108,92,211                  // vsubps        %ymm11,%ymm2,%ymm2
-  .byte  196,98,125,24,29,108,52,0,0         // vbroadcastss  0x346c(%rip),%ymm11        # 6544 <_sk_callback_avx+0x2fe>
+  .byte  196,98,125,24,29,108,52,0,0         // vbroadcastss  0x346c(%rip),%ymm11        # 6560 <_sk_callback_avx+0x2fe>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,98,52,0,0          // vbroadcastss  0x3462(%rip),%ymm11        # 6548 <_sk_callback_avx+0x302>
+  .byte  196,98,125,24,29,98,52,0,0          // vbroadcastss  0x3462(%rip),%ymm11        # 6564 <_sk_callback_avx+0x302>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,108,88,210                  // vaddps        %ymm10,%ymm2,%ymm2
-  .byte  196,98,125,24,21,83,52,0,0          // vbroadcastss  0x3453(%rip),%ymm10        # 654c <_sk_callback_avx+0x306>
+  .byte  196,98,125,24,21,83,52,0,0          // vbroadcastss  0x3453(%rip),%ymm10        # 6568 <_sk_callback_avx+0x306>
   .byte  196,193,108,89,210                  // vmulps        %ymm10,%ymm2,%ymm2
   .byte  197,253,91,210                      // vcvtps2dq     %ymm2,%ymm2
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16819,7 +16852,7 @@ _sk_parametric_b_avx:
   .byte  196,195,109,74,209,128              // vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,108,95,208                  // vmaxps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,42,52,0,0           // vbroadcastss  0x342a(%rip),%ymm8        # 6550 <_sk_callback_avx+0x30a>
+  .byte  196,98,125,24,5,42,52,0,0           // vbroadcastss  0x342a(%rip),%ymm8        # 656c <_sk_callback_avx+0x30a>
   .byte  196,193,108,93,208                  // vminps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16841,36 +16874,36 @@ _sk_parametric_a_avx:
   .byte  196,193,100,88,219                  // vaddps        %ymm11,%ymm3,%ymm3
   .byte  196,98,125,24,16                    // vbroadcastss  (%rax),%ymm10
   .byte  197,124,91,219                      // vcvtdq2ps     %ymm3,%ymm11
-  .byte  196,98,125,24,37,219,51,0,0         // vbroadcastss  0x33db(%rip),%ymm12        # 6554 <_sk_callback_avx+0x30e>
+  .byte  196,98,125,24,37,219,51,0,0         // vbroadcastss  0x33db(%rip),%ymm12        # 6570 <_sk_callback_avx+0x30e>
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,209,51,0,0         // vbroadcastss  0x33d1(%rip),%ymm12        # 6558 <_sk_callback_avx+0x312>
+  .byte  196,98,125,24,37,209,51,0,0         // vbroadcastss  0x33d1(%rip),%ymm12        # 6574 <_sk_callback_avx+0x312>
   .byte  196,193,100,84,220                  // vandps        %ymm12,%ymm3,%ymm3
-  .byte  196,98,125,24,37,199,51,0,0         // vbroadcastss  0x33c7(%rip),%ymm12        # 655c <_sk_callback_avx+0x316>
+  .byte  196,98,125,24,37,199,51,0,0         // vbroadcastss  0x33c7(%rip),%ymm12        # 6578 <_sk_callback_avx+0x316>
   .byte  196,193,100,86,220                  // vorps         %ymm12,%ymm3,%ymm3
-  .byte  196,98,125,24,37,189,51,0,0         // vbroadcastss  0x33bd(%rip),%ymm12        # 6560 <_sk_callback_avx+0x31a>
+  .byte  196,98,125,24,37,189,51,0,0         // vbroadcastss  0x33bd(%rip),%ymm12        # 657c <_sk_callback_avx+0x31a>
   .byte  196,65,36,88,220                    // vaddps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,179,51,0,0         // vbroadcastss  0x33b3(%rip),%ymm12        # 6564 <_sk_callback_avx+0x31e>
+  .byte  196,98,125,24,37,179,51,0,0         // vbroadcastss  0x33b3(%rip),%ymm12        # 6580 <_sk_callback_avx+0x31e>
   .byte  196,65,100,89,228                   // vmulps        %ymm12,%ymm3,%ymm12
   .byte  196,65,36,92,220                    // vsubps        %ymm12,%ymm11,%ymm11
-  .byte  196,98,125,24,37,164,51,0,0         // vbroadcastss  0x33a4(%rip),%ymm12        # 6568 <_sk_callback_avx+0x322>
+  .byte  196,98,125,24,37,164,51,0,0         // vbroadcastss  0x33a4(%rip),%ymm12        # 6584 <_sk_callback_avx+0x322>
   .byte  196,193,100,88,220                  // vaddps        %ymm12,%ymm3,%ymm3
-  .byte  196,98,125,24,37,154,51,0,0         // vbroadcastss  0x339a(%rip),%ymm12        # 656c <_sk_callback_avx+0x326>
+  .byte  196,98,125,24,37,154,51,0,0         // vbroadcastss  0x339a(%rip),%ymm12        # 6588 <_sk_callback_avx+0x326>
   .byte  197,156,94,219                      // vdivps        %ymm3,%ymm12,%ymm3
   .byte  197,164,92,219                      // vsubps        %ymm3,%ymm11,%ymm3
   .byte  197,172,89,219                      // vmulps        %ymm3,%ymm10,%ymm3
   .byte  196,99,125,8,211,1                  // vroundps      $0x1,%ymm3,%ymm10
   .byte  196,65,100,92,210                   // vsubps        %ymm10,%ymm3,%ymm10
-  .byte  196,98,125,24,29,126,51,0,0         // vbroadcastss  0x337e(%rip),%ymm11        # 6570 <_sk_callback_avx+0x32a>
+  .byte  196,98,125,24,29,126,51,0,0         // vbroadcastss  0x337e(%rip),%ymm11        # 658c <_sk_callback_avx+0x32a>
   .byte  196,193,100,88,219                  // vaddps        %ymm11,%ymm3,%ymm3
-  .byte  196,98,125,24,29,116,51,0,0         // vbroadcastss  0x3374(%rip),%ymm11        # 6574 <_sk_callback_avx+0x32e>
+  .byte  196,98,125,24,29,116,51,0,0         // vbroadcastss  0x3374(%rip),%ymm11        # 6590 <_sk_callback_avx+0x32e>
   .byte  196,65,44,89,219                    // vmulps        %ymm11,%ymm10,%ymm11
   .byte  196,193,100,92,219                  // vsubps        %ymm11,%ymm3,%ymm3
-  .byte  196,98,125,24,29,101,51,0,0         // vbroadcastss  0x3365(%rip),%ymm11        # 6578 <_sk_callback_avx+0x332>
+  .byte  196,98,125,24,29,101,51,0,0         // vbroadcastss  0x3365(%rip),%ymm11        # 6594 <_sk_callback_avx+0x332>
   .byte  196,65,36,92,210                    // vsubps        %ymm10,%ymm11,%ymm10
-  .byte  196,98,125,24,29,91,51,0,0          // vbroadcastss  0x335b(%rip),%ymm11        # 657c <_sk_callback_avx+0x336>
+  .byte  196,98,125,24,29,91,51,0,0          // vbroadcastss  0x335b(%rip),%ymm11        # 6598 <_sk_callback_avx+0x336>
   .byte  196,65,36,94,210                    // vdivps        %ymm10,%ymm11,%ymm10
   .byte  196,193,100,88,218                  // vaddps        %ymm10,%ymm3,%ymm3
-  .byte  196,98,125,24,21,76,51,0,0          // vbroadcastss  0x334c(%rip),%ymm10        # 6580 <_sk_callback_avx+0x33a>
+  .byte  196,98,125,24,21,76,51,0,0          // vbroadcastss  0x334c(%rip),%ymm10        # 659c <_sk_callback_avx+0x33a>
   .byte  196,193,100,89,218                  // vmulps        %ymm10,%ymm3,%ymm3
   .byte  197,253,91,219                      // vcvtps2dq     %ymm3,%ymm3
   .byte  196,98,125,24,80,20                 // vbroadcastss  0x14(%rax),%ymm10
@@ -16878,7 +16911,7 @@ _sk_parametric_a_avx:
   .byte  196,195,101,74,217,128              // vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   .byte  196,65,60,87,192                    // vxorps        %ymm8,%ymm8,%ymm8
   .byte  196,193,100,95,216                  // vmaxps        %ymm8,%ymm3,%ymm3
-  .byte  196,98,125,24,5,35,51,0,0           // vbroadcastss  0x3323(%rip),%ymm8        # 6584 <_sk_callback_avx+0x33e>
+  .byte  196,98,125,24,5,35,51,0,0           // vbroadcastss  0x3323(%rip),%ymm8        # 65a0 <_sk_callback_avx+0x33e>
   .byte  196,193,100,93,216                  // vminps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16887,31 +16920,31 @@ HIDDEN _sk_lab_to_xyz_avx
 .globl _sk_lab_to_xyz_avx
 FUNCTION(_sk_lab_to_xyz_avx)
 _sk_lab_to_xyz_avx:
-  .byte  196,98,125,24,5,21,51,0,0           // vbroadcastss  0x3315(%rip),%ymm8        # 6588 <_sk_callback_avx+0x342>
+  .byte  196,98,125,24,5,21,51,0,0           // vbroadcastss  0x3315(%rip),%ymm8        # 65a4 <_sk_callback_avx+0x342>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,11,51,0,0           // vbroadcastss  0x330b(%rip),%ymm8        # 658c <_sk_callback_avx+0x346>
+  .byte  196,98,125,24,5,11,51,0,0           // vbroadcastss  0x330b(%rip),%ymm8        # 65a8 <_sk_callback_avx+0x346>
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
-  .byte  196,98,125,24,13,1,51,0,0           // vbroadcastss  0x3301(%rip),%ymm9        # 6590 <_sk_callback_avx+0x34a>
+  .byte  196,98,125,24,13,1,51,0,0           // vbroadcastss  0x3301(%rip),%ymm9        # 65ac <_sk_callback_avx+0x34a>
   .byte  196,193,116,88,201                  // vaddps        %ymm9,%ymm1,%ymm1
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  196,193,108,88,209                  // vaddps        %ymm9,%ymm2,%ymm2
-  .byte  196,98,125,24,5,237,50,0,0          // vbroadcastss  0x32ed(%rip),%ymm8        # 6594 <_sk_callback_avx+0x34e>
+  .byte  196,98,125,24,5,237,50,0,0          // vbroadcastss  0x32ed(%rip),%ymm8        # 65b0 <_sk_callback_avx+0x34e>
   .byte  196,193,124,88,192                  // vaddps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,227,50,0,0          // vbroadcastss  0x32e3(%rip),%ymm8        # 6598 <_sk_callback_avx+0x352>
+  .byte  196,98,125,24,5,227,50,0,0          // vbroadcastss  0x32e3(%rip),%ymm8        # 65b4 <_sk_callback_avx+0x352>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,5,217,50,0,0          // vbroadcastss  0x32d9(%rip),%ymm8        # 659c <_sk_callback_avx+0x356>
+  .byte  196,98,125,24,5,217,50,0,0          // vbroadcastss  0x32d9(%rip),%ymm8        # 65b8 <_sk_callback_avx+0x356>
   .byte  196,193,116,89,200                  // vmulps        %ymm8,%ymm1,%ymm1
   .byte  197,252,88,201                      // vaddps        %ymm1,%ymm0,%ymm1
-  .byte  196,98,125,24,5,203,50,0,0          // vbroadcastss  0x32cb(%rip),%ymm8        # 65a0 <_sk_callback_avx+0x35a>
+  .byte  196,98,125,24,5,203,50,0,0          // vbroadcastss  0x32cb(%rip),%ymm8        # 65bc <_sk_callback_avx+0x35a>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  197,252,92,210                      // vsubps        %ymm2,%ymm0,%ymm2
   .byte  197,116,89,193                      // vmulps        %ymm1,%ymm1,%ymm8
   .byte  196,65,116,89,192                   // vmulps        %ymm8,%ymm1,%ymm8
-  .byte  196,98,125,24,13,180,50,0,0         // vbroadcastss  0x32b4(%rip),%ymm9        # 65a4 <_sk_callback_avx+0x35e>
+  .byte  196,98,125,24,13,180,50,0,0         // vbroadcastss  0x32b4(%rip),%ymm9        # 65c0 <_sk_callback_avx+0x35e>
   .byte  196,65,52,194,208,1                 // vcmpltps      %ymm8,%ymm9,%ymm10
-  .byte  196,98,125,24,29,169,50,0,0         // vbroadcastss  0x32a9(%rip),%ymm11        # 65a8 <_sk_callback_avx+0x362>
+  .byte  196,98,125,24,29,169,50,0,0         // vbroadcastss  0x32a9(%rip),%ymm11        # 65c4 <_sk_callback_avx+0x362>
   .byte  196,193,116,88,203                  // vaddps        %ymm11,%ymm1,%ymm1
-  .byte  196,98,125,24,37,159,50,0,0         // vbroadcastss  0x329f(%rip),%ymm12        # 65ac <_sk_callback_avx+0x366>
+  .byte  196,98,125,24,37,159,50,0,0         // vbroadcastss  0x329f(%rip),%ymm12        # 65c8 <_sk_callback_avx+0x366>
   .byte  196,193,116,89,204                  // vmulps        %ymm12,%ymm1,%ymm1
   .byte  196,67,117,74,192,160               // vblendvps     %ymm10,%ymm8,%ymm1,%ymm8
   .byte  197,252,89,200                      // vmulps        %ymm0,%ymm0,%ymm1
@@ -16926,9 +16959,9 @@ _sk_lab_to_xyz_avx:
   .byte  196,193,108,88,211                  // vaddps        %ymm11,%ymm2,%ymm2
   .byte  196,193,108,89,212                  // vmulps        %ymm12,%ymm2,%ymm2
   .byte  196,227,109,74,208,144              // vblendvps     %ymm9,%ymm0,%ymm2,%ymm2
-  .byte  196,226,125,24,5,85,50,0,0          // vbroadcastss  0x3255(%rip),%ymm0        # 65b0 <_sk_callback_avx+0x36a>
+  .byte  196,226,125,24,5,85,50,0,0          // vbroadcastss  0x3255(%rip),%ymm0        # 65cc <_sk_callback_avx+0x36a>
   .byte  197,188,89,192                      // vmulps        %ymm0,%ymm8,%ymm0
-  .byte  196,98,125,24,5,76,50,0,0           // vbroadcastss  0x324c(%rip),%ymm8        # 65b4 <_sk_callback_avx+0x36e>
+  .byte  196,98,125,24,5,76,50,0,0           // vbroadcastss  0x324c(%rip),%ymm8        # 65d0 <_sk_callback_avx+0x36e>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -16942,14 +16975,14 @@ _sk_load_a8_avx:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,62                              // jne           33bf <_sk_load_a8_avx+0x4e>
+  .byte  117,62                              // jne           33db <_sk_load_a8_avx+0x4e>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,121,49,200                  // vpmovzxbd     %xmm0,%xmm1
   .byte  196,227,121,4,192,229               // vpermilps     $0xe5,%xmm0,%xmm0
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,117,24,192,1                // vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,16,50,0,0         // vbroadcastss  0x3210(%rip),%ymm1        # 65b8 <_sk_callback_avx+0x372>
+  .byte  196,226,125,24,13,16,50,0,0         // vbroadcastss  0x3210(%rip),%ymm1        # 65d4 <_sk_callback_avx+0x372>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -16966,9 +16999,9 @@ _sk_load_a8_avx:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           33c7 <_sk_load_a8_avx+0x56>
+  .byte  117,234                             // jne           33e3 <_sk_load_a8_avx+0x56>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,161                             // jmp           3385 <_sk_load_a8_avx+0x14>
+  .byte  235,161                             // jmp           33a1 <_sk_load_a8_avx+0x14>
 
 HIDDEN _sk_gather_a8_avx
 .globl _sk_gather_a8_avx
@@ -17018,7 +17051,7 @@ _sk_gather_a8_avx:
   .byte  196,226,121,49,201                  // vpmovzxbd     %xmm1,%xmm1
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,5,49,0,0          // vbroadcastss  0x3105(%rip),%ymm1        # 65bc <_sk_callback_avx+0x376>
+  .byte  196,226,125,24,13,5,49,0,0          // vbroadcastss  0x3105(%rip),%ymm1        # 65d8 <_sk_callback_avx+0x376>
   .byte  197,252,89,217                      // vmulps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  197,252,87,192                      // vxorps        %ymm0,%ymm0,%ymm0
@@ -17036,14 +17069,14 @@ FUNCTION(_sk_store_a8_avx)
 _sk_store_a8_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,224,48,0,0          // vbroadcastss  0x30e0(%rip),%ymm8        # 65c0 <_sk_callback_avx+0x37a>
+  .byte  196,98,125,24,5,224,48,0,0          // vbroadcastss  0x30e0(%rip),%ymm8        # 65dc <_sk_callback_avx+0x37a>
   .byte  196,65,100,89,192                   // vmulps        %ymm8,%ymm3,%ymm8
   .byte  196,65,125,91,192                   // vcvtps2dq     %ymm8,%ymm8
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  196,65,57,103,192                   // vpackuswb     %xmm8,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3509 <_sk_store_a8_avx+0x37>
+  .byte  117,10                              // jne           3525 <_sk_store_a8_avx+0x37>
   .byte  196,65,123,17,4,58                  // vmovsd        %xmm8,(%r10,%rdi,1)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17051,10 +17084,10 @@ _sk_store_a8_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3505 <_sk_store_a8_avx+0x33>
+  .byte  119,236                             // ja            3521 <_sk_store_a8_avx+0x33>
   .byte  196,66,121,48,192                   // vpmovzxbw     %xmm8,%xmm8
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 356c <_sk_store_a8_avx+0x9a>
+  .byte  76,141,13,67,0,0,0                  // lea           0x43(%rip),%r9        # 3588 <_sk_store_a8_avx+0x9a>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17065,7 +17098,7 @@ _sk_store_a8_avx:
   .byte  196,67,121,20,68,58,2,4             // vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   .byte  196,67,121,20,68,58,1,2             // vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   .byte  196,67,121,20,4,58,0                // vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  .byte  235,154                             // jmp           3505 <_sk_store_a8_avx+0x33>
+  .byte  235,154                             // jmp           3521 <_sk_store_a8_avx+0x33>
   .byte  144                                 // nop
   .byte  246,255                             // idiv          %bh
   .byte  255                                 // (bad)
@@ -17099,17 +17132,17 @@ _sk_load_g8_avx:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,1,248                            // add           %rdi,%rax
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  117,67                              // jne           35db <_sk_load_g8_avx+0x53>
+  .byte  117,67                              // jne           35f7 <_sk_load_g8_avx+0x53>
   .byte  197,250,126,0                       // vmovq         (%rax),%xmm0
   .byte  196,226,121,49,200                  // vpmovzxbd     %xmm0,%xmm1
   .byte  196,227,121,4,192,229               // vpermilps     $0xe5,%xmm0,%xmm0
   .byte  196,226,121,49,192                  // vpmovzxbd     %xmm0,%xmm0
   .byte  196,227,117,24,192,1                // vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,5,48,0,0          // vbroadcastss  0x3005(%rip),%ymm1        # 65c4 <_sk_callback_avx+0x37e>
+  .byte  196,226,125,24,13,5,48,0,0          // vbroadcastss  0x3005(%rip),%ymm1        # 65e0 <_sk_callback_avx+0x37e>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,250,47,0,0        // vbroadcastss  0x2ffa(%rip),%ymm3        # 65c8 <_sk_callback_avx+0x382>
+  .byte  196,226,125,24,29,250,47,0,0        // vbroadcastss  0x2ffa(%rip),%ymm3        # 65e4 <_sk_callback_avx+0x382>
   .byte  76,137,193                          // mov           %r8,%rcx
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
@@ -17123,9 +17156,9 @@ _sk_load_g8_avx:
   .byte  77,9,217                            // or            %r11,%r9
   .byte  72,131,193,8                        // add           $0x8,%rcx
   .byte  73,255,202                          // dec           %r10
-  .byte  117,234                             // jne           35e3 <_sk_load_g8_avx+0x5b>
+  .byte  117,234                             // jne           35ff <_sk_load_g8_avx+0x5b>
   .byte  196,193,249,110,193                 // vmovq         %r9,%xmm0
-  .byte  235,156                             // jmp           359c <_sk_load_g8_avx+0x14>
+  .byte  235,156                             // jmp           35b8 <_sk_load_g8_avx+0x14>
 
 HIDDEN _sk_gather_g8_avx
 .globl _sk_gather_g8_avx
@@ -17175,10 +17208,10 @@ _sk_gather_g8_avx:
   .byte  196,226,121,49,201                  // vpmovzxbd     %xmm1,%xmm1
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,249,46,0,0        // vbroadcastss  0x2ef9(%rip),%ymm1        # 65cc <_sk_callback_avx+0x386>
+  .byte  196,226,125,24,13,249,46,0,0        // vbroadcastss  0x2ef9(%rip),%ymm1        # 65e8 <_sk_callback_avx+0x386>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,238,46,0,0        // vbroadcastss  0x2eee(%rip),%ymm3        # 65d0 <_sk_callback_avx+0x38a>
+  .byte  196,226,125,24,29,238,46,0,0        // vbroadcastss  0x2eee(%rip),%ymm3        # 65ec <_sk_callback_avx+0x38a>
   .byte  197,252,40,200                      // vmovaps       %ymm0,%ymm1
   .byte  197,252,40,208                      // vmovaps       %ymm0,%ymm2
   .byte  91                                  // pop           %rbx
@@ -17194,9 +17227,9 @@ _sk_gather_i8_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            3702 <_sk_gather_i8_avx+0xf>
+  .byte  116,5                               // je            371e <_sk_gather_i8_avx+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           3704 <_sk_gather_i8_avx+0x11>
+  .byte  235,2                               // jmp           3720 <_sk_gather_i8_avx+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,87                               // push          %r15
   .byte  65,86                               // push          %r14
@@ -17258,10 +17291,10 @@ _sk_gather_i8_avx:
   .byte  196,163,121,34,4,163,2              // vpinsrd       $0x2,(%rbx,%r12,4),%xmm0,%xmm0
   .byte  196,163,121,34,28,19,3              // vpinsrd       $0x3,(%rbx,%r10,1),%xmm0,%xmm3
   .byte  196,227,61,24,195,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  .byte  197,124,40,21,114,47,0,0            // vmovaps       0x2f72(%rip),%ymm10        # 67a0 <_sk_callback_avx+0x55a>
+  .byte  197,124,40,21,118,47,0,0            // vmovaps       0x2f76(%rip),%ymm10        # 67c0 <_sk_callback_avx+0x55e>
   .byte  196,193,124,84,194                  // vandps        %ymm10,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,148,45,0,0         // vbroadcastss  0x2d94(%rip),%ymm9        # 65d4 <_sk_callback_avx+0x38e>
+  .byte  196,98,125,24,13,148,45,0,0         // vbroadcastss  0x2d94(%rip),%ymm9        # 65f0 <_sk_callback_avx+0x38e>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,113,114,208,8               // vpsrld        $0x8,%xmm8,%xmm1
   .byte  197,233,114,211,8                   // vpsrld        $0x8,%xmm3,%xmm2
@@ -17295,38 +17328,38 @@ _sk_load_565_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,128,0,0,0                    // jne           3938 <_sk_load_565_avx+0x8e>
+  .byte  15,133,128,0,0,0                    // jne           3954 <_sk_load_565_avx+0x8e>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  197,241,239,201                     // vpxor         %xmm1,%xmm1,%xmm1
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,209,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  .byte  196,226,125,24,5,254,44,0,0         // vbroadcastss  0x2cfe(%rip),%ymm0        # 65d8 <_sk_callback_avx+0x392>
+  .byte  196,226,125,24,5,254,44,0,0         // vbroadcastss  0x2cfe(%rip),%ymm0        # 65f4 <_sk_callback_avx+0x392>
   .byte  197,236,84,192                      // vandps        %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,241,44,0,0        // vbroadcastss  0x2cf1(%rip),%ymm1        # 65dc <_sk_callback_avx+0x396>
+  .byte  196,226,125,24,13,241,44,0,0        // vbroadcastss  0x2cf1(%rip),%ymm1        # 65f8 <_sk_callback_avx+0x396>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,232,44,0,0        // vbroadcastss  0x2ce8(%rip),%ymm1        # 65e0 <_sk_callback_avx+0x39a>
+  .byte  196,226,125,24,13,232,44,0,0        // vbroadcastss  0x2ce8(%rip),%ymm1        # 65fc <_sk_callback_avx+0x39a>
   .byte  197,236,84,201                      // vandps        %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,219,44,0,0        // vbroadcastss  0x2cdb(%rip),%ymm3        # 65e4 <_sk_callback_avx+0x39e>
+  .byte  196,226,125,24,29,219,44,0,0        // vbroadcastss  0x2cdb(%rip),%ymm3        # 6600 <_sk_callback_avx+0x39e>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,24,29,210,44,0,0        // vbroadcastss  0x2cd2(%rip),%ymm3        # 65e8 <_sk_callback_avx+0x3a2>
+  .byte  196,226,125,24,29,210,44,0,0        // vbroadcastss  0x2cd2(%rip),%ymm3        # 6604 <_sk_callback_avx+0x3a2>
   .byte  197,236,84,211                      // vandps        %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,197,44,0,0        // vbroadcastss  0x2cc5(%rip),%ymm3        # 65ec <_sk_callback_avx+0x3a6>
+  .byte  196,226,125,24,29,197,44,0,0        // vbroadcastss  0x2cc5(%rip),%ymm3        # 6608 <_sk_callback_avx+0x3a6>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,186,44,0,0        // vbroadcastss  0x2cba(%rip),%ymm3        # 65f0 <_sk_callback_avx+0x3aa>
+  .byte  196,226,125,24,29,186,44,0,0        // vbroadcastss  0x2cba(%rip),%ymm3        # 660c <_sk_callback_avx+0x3aa>
   .byte  255,224                             // jmpq          *%rax
   .byte  65,137,200                          // mov           %ecx,%r8d
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,110,255,255,255              // ja            38be <_sk_load_565_avx+0x14>
+  .byte  15,135,110,255,255,255              // ja            38da <_sk_load_565_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 39a4 <_sk_load_565_avx+0xfa>
+  .byte  76,141,13,73,0,0,0                  // lea           0x49(%rip),%r9        # 39c0 <_sk_load_565_avx+0xfa>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17338,7 +17371,7 @@ _sk_load_565_avx:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,26,255,255,255                  // jmpq          38be <_sk_load_565_avx+0x14>
+  .byte  233,26,255,255,255                  // jmpq          38da <_sk_load_565_avx+0x14>
   .byte  244                                 // hlt
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -17416,23 +17449,23 @@ _sk_gather_565_avx:
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,209,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  .byte  196,226,125,24,5,90,43,0,0          // vbroadcastss  0x2b5a(%rip),%ymm0        # 65f4 <_sk_callback_avx+0x3ae>
+  .byte  196,226,125,24,5,90,43,0,0          // vbroadcastss  0x2b5a(%rip),%ymm0        # 6610 <_sk_callback_avx+0x3ae>
   .byte  197,236,84,192                      // vandps        %ymm0,%ymm2,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,77,43,0,0         // vbroadcastss  0x2b4d(%rip),%ymm1        # 65f8 <_sk_callback_avx+0x3b2>
+  .byte  196,226,125,24,13,77,43,0,0         // vbroadcastss  0x2b4d(%rip),%ymm1        # 6614 <_sk_callback_avx+0x3b2>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,68,43,0,0         // vbroadcastss  0x2b44(%rip),%ymm1        # 65fc <_sk_callback_avx+0x3b6>
+  .byte  196,226,125,24,13,68,43,0,0         // vbroadcastss  0x2b44(%rip),%ymm1        # 6618 <_sk_callback_avx+0x3b6>
   .byte  197,236,84,201                      // vandps        %ymm1,%ymm2,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,29,55,43,0,0         // vbroadcastss  0x2b37(%rip),%ymm3        # 6600 <_sk_callback_avx+0x3ba>
+  .byte  196,226,125,24,29,55,43,0,0         // vbroadcastss  0x2b37(%rip),%ymm3        # 661c <_sk_callback_avx+0x3ba>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
-  .byte  196,226,125,24,29,46,43,0,0         // vbroadcastss  0x2b2e(%rip),%ymm3        # 6604 <_sk_callback_avx+0x3be>
+  .byte  196,226,125,24,29,46,43,0,0         // vbroadcastss  0x2b2e(%rip),%ymm3        # 6620 <_sk_callback_avx+0x3be>
   .byte  197,236,84,211                      // vandps        %ymm3,%ymm2,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,226,125,24,29,33,43,0,0         // vbroadcastss  0x2b21(%rip),%ymm3        # 6608 <_sk_callback_avx+0x3c2>
+  .byte  196,226,125,24,29,33,43,0,0         // vbroadcastss  0x2b21(%rip),%ymm3        # 6624 <_sk_callback_avx+0x3c2>
   .byte  197,236,89,211                      // vmulps        %ymm3,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,22,43,0,0         // vbroadcastss  0x2b16(%rip),%ymm3        # 660c <_sk_callback_avx+0x3c6>
+  .byte  196,226,125,24,29,22,43,0,0         // vbroadcastss  0x2b16(%rip),%ymm3        # 6628 <_sk_callback_avx+0x3c6>
   .byte  91                                  // pop           %rbx
   .byte  65,92                               // pop           %r12
   .byte  65,94                               // pop           %r14
@@ -17446,14 +17479,14 @@ FUNCTION(_sk_store_565_avx)
 _sk_store_565_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,2,43,0,0            // vbroadcastss  0x2b02(%rip),%ymm8        # 6610 <_sk_callback_avx+0x3ca>
+  .byte  196,98,125,24,5,2,43,0,0            // vbroadcastss  0x2b02(%rip),%ymm8        # 662c <_sk_callback_avx+0x3ca>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,41,114,241,11               // vpslld        $0xb,%xmm9,%xmm10
   .byte  196,67,125,25,201,1                 // vextractf128  $0x1,%ymm9,%xmm9
   .byte  196,193,49,114,241,11               // vpslld        $0xb,%xmm9,%xmm9
   .byte  196,67,45,24,201,1                  // vinsertf128   $0x1,%xmm9,%ymm10,%ymm9
-  .byte  196,98,125,24,21,219,42,0,0         // vbroadcastss  0x2adb(%rip),%ymm10        # 6614 <_sk_callback_avx+0x3ce>
+  .byte  196,98,125,24,21,219,42,0,0         // vbroadcastss  0x2adb(%rip),%ymm10        # 6630 <_sk_callback_avx+0x3ce>
   .byte  196,65,116,89,210                   // vmulps        %ymm10,%ymm1,%ymm10
   .byte  196,65,125,91,210                   // vcvtps2dq     %ymm10,%ymm10
   .byte  196,193,33,114,242,5                // vpslld        $0x5,%xmm10,%xmm11
@@ -17467,7 +17500,7 @@ _sk_store_565_avx:
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3b89 <_sk_store_565_avx+0x89>
+  .byte  117,10                              // jne           3ba5 <_sk_store_565_avx+0x89>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17475,9 +17508,9 @@ _sk_store_565_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3b85 <_sk_store_565_avx+0x85>
+  .byte  119,236                             // ja            3ba1 <_sk_store_565_avx+0x85>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 3be8 <_sk_store_565_avx+0xe8>
+  .byte  76,141,13,68,0,0,0                  // lea           0x44(%rip),%r9        # 3c04 <_sk_store_565_avx+0xe8>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17488,7 +17521,7 @@ _sk_store_565_avx:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           3b85 <_sk_store_565_avx+0x85>
+  .byte  235,159                             // jmp           3ba1 <_sk_store_565_avx+0x85>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -17521,31 +17554,31 @@ _sk_load_4444_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,152,0,0,0                    // jne           3caa <_sk_load_4444_avx+0xa6>
+  .byte  15,133,152,0,0,0                    // jne           3cc6 <_sk_load_4444_avx+0xa6>
   .byte  196,193,122,111,4,122               // vmovdqu       (%r10,%rdi,2),%xmm0
   .byte  197,241,239,201                     // vpxor         %xmm1,%xmm1,%xmm1
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,217,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  .byte  196,226,125,24,5,228,41,0,0         // vbroadcastss  0x29e4(%rip),%ymm0        # 6618 <_sk_callback_avx+0x3d2>
+  .byte  196,226,125,24,5,228,41,0,0         // vbroadcastss  0x29e4(%rip),%ymm0        # 6634 <_sk_callback_avx+0x3d2>
   .byte  197,228,84,192                      // vandps        %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,215,41,0,0        // vbroadcastss  0x29d7(%rip),%ymm1        # 661c <_sk_callback_avx+0x3d6>
+  .byte  196,226,125,24,13,215,41,0,0        // vbroadcastss  0x29d7(%rip),%ymm1        # 6638 <_sk_callback_avx+0x3d6>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,206,41,0,0        // vbroadcastss  0x29ce(%rip),%ymm1        # 6620 <_sk_callback_avx+0x3da>
+  .byte  196,226,125,24,13,206,41,0,0        // vbroadcastss  0x29ce(%rip),%ymm1        # 663c <_sk_callback_avx+0x3da>
   .byte  197,228,84,201                      // vandps        %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,193,41,0,0        // vbroadcastss  0x29c1(%rip),%ymm2        # 6624 <_sk_callback_avx+0x3de>
+  .byte  196,226,125,24,21,193,41,0,0        // vbroadcastss  0x29c1(%rip),%ymm2        # 6640 <_sk_callback_avx+0x3de>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,24,21,184,41,0,0        // vbroadcastss  0x29b8(%rip),%ymm2        # 6628 <_sk_callback_avx+0x3e2>
+  .byte  196,226,125,24,21,184,41,0,0        // vbroadcastss  0x29b8(%rip),%ymm2        # 6644 <_sk_callback_avx+0x3e2>
   .byte  197,228,84,210                      // vandps        %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,171,41,0,0          // vbroadcastss  0x29ab(%rip),%ymm8        # 662c <_sk_callback_avx+0x3e6>
+  .byte  196,98,125,24,5,171,41,0,0          // vbroadcastss  0x29ab(%rip),%ymm8        # 6648 <_sk_callback_avx+0x3e6>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,161,41,0,0          // vbroadcastss  0x29a1(%rip),%ymm8        # 6630 <_sk_callback_avx+0x3ea>
+  .byte  196,98,125,24,5,161,41,0,0          // vbroadcastss  0x29a1(%rip),%ymm8        # 664c <_sk_callback_avx+0x3ea>
   .byte  196,193,100,84,216                  // vandps        %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,147,41,0,0          // vbroadcastss  0x2993(%rip),%ymm8        # 6634 <_sk_callback_avx+0x3ee>
+  .byte  196,98,125,24,5,147,41,0,0          // vbroadcastss  0x2993(%rip),%ymm8        # 6650 <_sk_callback_avx+0x3ee>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17554,9 +17587,9 @@ _sk_load_4444_avx:
   .byte  197,249,239,192                     // vpxor         %xmm0,%xmm0,%xmm0
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,86,255,255,255               // ja            3c18 <_sk_load_4444_avx+0x14>
+  .byte  15,135,86,255,255,255               // ja            3c34 <_sk_load_4444_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,75,0,0,0                  // lea           0x4b(%rip),%r9        # 3d18 <_sk_load_4444_avx+0x114>
+  .byte  76,141,13,75,0,0,0                  // lea           0x4b(%rip),%r9        # 3d34 <_sk_load_4444_avx+0x114>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17568,7 +17601,7 @@ _sk_load_4444_avx:
   .byte  196,193,121,196,68,122,4,2          // vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,68,122,2,1          // vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   .byte  196,193,121,196,4,122,0             // vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  .byte  233,2,255,255,255                   // jmpq          3c18 <_sk_load_4444_avx+0x14>
+  .byte  233,2,255,255,255                   // jmpq          3c34 <_sk_load_4444_avx+0x14>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  242,255                             // repnz         (bad)
   .byte  255                                 // (bad)
@@ -17647,25 +17680,25 @@ _sk_gather_4444_avx:
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,217,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  .byte  196,226,125,24,5,42,40,0,0          // vbroadcastss  0x282a(%rip),%ymm0        # 6638 <_sk_callback_avx+0x3f2>
+  .byte  196,226,125,24,5,42,40,0,0          // vbroadcastss  0x282a(%rip),%ymm0        # 6654 <_sk_callback_avx+0x3f2>
   .byte  197,228,84,192                      // vandps        %ymm0,%ymm3,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,226,125,24,13,29,40,0,0         // vbroadcastss  0x281d(%rip),%ymm1        # 663c <_sk_callback_avx+0x3f6>
+  .byte  196,226,125,24,13,29,40,0,0         // vbroadcastss  0x281d(%rip),%ymm1        # 6658 <_sk_callback_avx+0x3f6>
   .byte  197,252,89,193                      // vmulps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,20,40,0,0         // vbroadcastss  0x2814(%rip),%ymm1        # 6640 <_sk_callback_avx+0x3fa>
+  .byte  196,226,125,24,13,20,40,0,0         // vbroadcastss  0x2814(%rip),%ymm1        # 665c <_sk_callback_avx+0x3fa>
   .byte  197,228,84,201                      // vandps        %ymm1,%ymm3,%ymm1
   .byte  197,252,91,201                      // vcvtdq2ps     %ymm1,%ymm1
-  .byte  196,226,125,24,21,7,40,0,0          // vbroadcastss  0x2807(%rip),%ymm2        # 6644 <_sk_callback_avx+0x3fe>
+  .byte  196,226,125,24,21,7,40,0,0          // vbroadcastss  0x2807(%rip),%ymm2        # 6660 <_sk_callback_avx+0x3fe>
   .byte  197,244,89,202                      // vmulps        %ymm2,%ymm1,%ymm1
-  .byte  196,226,125,24,21,254,39,0,0        // vbroadcastss  0x27fe(%rip),%ymm2        # 6648 <_sk_callback_avx+0x402>
+  .byte  196,226,125,24,21,254,39,0,0        // vbroadcastss  0x27fe(%rip),%ymm2        # 6664 <_sk_callback_avx+0x402>
   .byte  197,228,84,210                      // vandps        %ymm2,%ymm3,%ymm2
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
-  .byte  196,98,125,24,5,241,39,0,0          // vbroadcastss  0x27f1(%rip),%ymm8        # 664c <_sk_callback_avx+0x406>
+  .byte  196,98,125,24,5,241,39,0,0          // vbroadcastss  0x27f1(%rip),%ymm8        # 6668 <_sk_callback_avx+0x406>
   .byte  196,193,108,89,208                  // vmulps        %ymm8,%ymm2,%ymm2
-  .byte  196,98,125,24,5,231,39,0,0          // vbroadcastss  0x27e7(%rip),%ymm8        # 6650 <_sk_callback_avx+0x40a>
+  .byte  196,98,125,24,5,231,39,0,0          // vbroadcastss  0x27e7(%rip),%ymm8        # 666c <_sk_callback_avx+0x40a>
   .byte  196,193,100,84,216                  // vandps        %ymm8,%ymm3,%ymm3
   .byte  197,252,91,219                      // vcvtdq2ps     %ymm3,%ymm3
-  .byte  196,98,125,24,5,217,39,0,0          // vbroadcastss  0x27d9(%rip),%ymm8        # 6654 <_sk_callback_avx+0x40e>
+  .byte  196,98,125,24,5,217,39,0,0          // vbroadcastss  0x27d9(%rip),%ymm8        # 6670 <_sk_callback_avx+0x40e>
   .byte  196,193,100,89,216                  // vmulps        %ymm8,%ymm3,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  91                                  // pop           %rbx
@@ -17681,7 +17714,7 @@ FUNCTION(_sk_store_4444_avx)
 _sk_store_4444_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,190,39,0,0          // vbroadcastss  0x27be(%rip),%ymm8        # 6658 <_sk_callback_avx+0x412>
+  .byte  196,98,125,24,5,190,39,0,0          // vbroadcastss  0x27be(%rip),%ymm8        # 6674 <_sk_callback_avx+0x412>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,193,41,114,241,12               // vpslld        $0xc,%xmm9,%xmm10
@@ -17708,7 +17741,7 @@ _sk_store_4444_avx:
   .byte  196,67,125,25,193,1                 // vextractf128  $0x1,%ymm8,%xmm9
   .byte  196,66,57,43,193                    // vpackusdw     %xmm9,%xmm8,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           3f33 <_sk_store_4444_avx+0xa7>
+  .byte  117,10                              // jne           3f4f <_sk_store_4444_avx+0xa7>
   .byte  196,65,122,127,4,122                // vmovdqu       %xmm8,(%r10,%rdi,2)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17716,9 +17749,9 @@ _sk_store_4444_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            3f2f <_sk_store_4444_avx+0xa3>
+  .byte  119,236                             // ja            3f4b <_sk_store_4444_avx+0xa3>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,66,0,0,0                  // lea           0x42(%rip),%r9        # 3f90 <_sk_store_4444_avx+0x104>
+  .byte  76,141,13,66,0,0,0                  // lea           0x42(%rip),%r9        # 3fac <_sk_store_4444_avx+0x104>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17729,7 +17762,7 @@ _sk_store_4444_avx:
   .byte  196,67,121,21,68,122,4,2            // vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   .byte  196,67,121,21,68,122,2,1            // vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   .byte  196,67,121,21,4,122,0               // vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  .byte  235,159                             // jmp           3f2f <_sk_store_4444_avx+0xa3>
+  .byte  235,159                             // jmp           3f4b <_sk_store_4444_avx+0xa3>
   .byte  247,255                             // idiv          %edi
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
@@ -17760,12 +17793,12 @@ _sk_load_8888_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,135,0,0,0                    // jne           4041 <_sk_load_8888_avx+0x95>
+  .byte  15,133,135,0,0,0                    // jne           405d <_sk_load_8888_avx+0x95>
   .byte  196,65,124,16,12,186                // vmovups       (%r10,%rdi,4),%ymm9
-  .byte  197,124,40,21,248,39,0,0            // vmovaps       0x27f8(%rip),%ymm10        # 67c0 <_sk_callback_avx+0x57a>
+  .byte  197,124,40,21,252,39,0,0            // vmovaps       0x27fc(%rip),%ymm10        # 67e0 <_sk_callback_avx+0x57e>
   .byte  196,193,52,84,194                   // vandps        %ymm10,%ymm9,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,5,130,38,0,0          // vbroadcastss  0x2682(%rip),%ymm8        # 665c <_sk_callback_avx+0x416>
+  .byte  196,98,125,24,5,130,38,0,0          // vbroadcastss  0x2682(%rip),%ymm8        # 6678 <_sk_callback_avx+0x416>
   .byte  196,193,124,89,192                  // vmulps        %ymm8,%ymm0,%ymm0
   .byte  196,193,113,114,209,8               // vpsrld        $0x8,%xmm9,%xmm1
   .byte  196,99,125,25,203,1                 // vextractf128  $0x1,%ymm9,%xmm3
@@ -17792,9 +17825,9 @@ _sk_load_8888_avx:
   .byte  196,65,52,87,201                    // vxorps        %ymm9,%ymm9,%ymm9
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  15,135,102,255,255,255              // ja            3fc0 <_sk_load_8888_avx+0x14>
+  .byte  15,135,102,255,255,255              // ja            3fdc <_sk_load_8888_avx+0x14>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,139,0,0,0                 // lea           0x8b(%rip),%r9        # 40f0 <_sk_load_8888_avx+0x144>
+  .byte  76,141,13,139,0,0,0                 // lea           0x8b(%rip),%r9        # 410c <_sk_load_8888_avx+0x144>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17817,7 +17850,7 @@ _sk_load_8888_avx:
   .byte  196,99,53,12,200,15                 // vblendps      $0xf,%ymm0,%ymm9,%ymm9
   .byte  196,195,49,34,4,186,0               // vpinsrd       $0x0,(%r10,%rdi,4),%xmm9,%xmm0
   .byte  196,99,53,12,200,15                 // vblendps      $0xf,%ymm0,%ymm9,%ymm9
-  .byte  233,210,254,255,255                 // jmpq          3fc0 <_sk_load_8888_avx+0x14>
+  .byte  233,210,254,255,255                 // jmpq          3fdc <_sk_load_8888_avx+0x14>
   .byte  102,144                             // xchg          %ax,%ax
   .byte  236                                 // in            (%dx),%al
   .byte  255                                 // (bad)
@@ -17835,7 +17868,7 @@ _sk_load_8888_avx:
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  126,255                             // jle           4109 <_sk_load_8888_avx+0x15d>
+  .byte  126,255                             // jle           4125 <_sk_load_8888_avx+0x15d>
   .byte  255                                 // (bad)
   .byte  255                                 // .byte         0xff
 
@@ -17880,10 +17913,10 @@ _sk_gather_8888_avx:
   .byte  196,131,121,34,4,152,2              // vpinsrd       $0x2,(%r8,%r11,4),%xmm0,%xmm0
   .byte  196,131,121,34,28,144,3             // vpinsrd       $0x3,(%r8,%r10,4),%xmm0,%xmm3
   .byte  196,227,61,24,195,1                 // vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  .byte  197,124,40,21,34,38,0,0             // vmovaps       0x2622(%rip),%ymm10        # 67e0 <_sk_callback_avx+0x59a>
+  .byte  197,124,40,21,38,38,0,0             // vmovaps       0x2626(%rip),%ymm10        # 6800 <_sk_callback_avx+0x59e>
   .byte  196,193,124,84,194                  // vandps        %ymm10,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,13,144,36,0,0         // vbroadcastss  0x2490(%rip),%ymm9        # 6660 <_sk_callback_avx+0x41a>
+  .byte  196,98,125,24,13,144,36,0,0         // vbroadcastss  0x2490(%rip),%ymm9        # 667c <_sk_callback_avx+0x41a>
   .byte  196,193,124,89,193                  // vmulps        %ymm9,%ymm0,%ymm0
   .byte  196,193,113,114,208,8               // vpsrld        $0x8,%xmm8,%xmm1
   .byte  197,233,114,211,8                   // vpsrld        $0x8,%xmm3,%xmm2
@@ -17915,7 +17948,7 @@ FUNCTION(_sk_store_8888_avx)
 _sk_store_8888_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
-  .byte  196,98,125,24,5,30,36,0,0           // vbroadcastss  0x241e(%rip),%ymm8        # 6664 <_sk_callback_avx+0x41e>
+  .byte  196,98,125,24,5,30,36,0,0           // vbroadcastss  0x241e(%rip),%ymm8        # 6680 <_sk_callback_avx+0x41e>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,65,116,89,208                   // vmulps        %ymm8,%ymm1,%ymm10
@@ -17940,7 +17973,7 @@ _sk_store_8888_avx:
   .byte  196,65,45,86,192                    // vorpd         %ymm8,%ymm10,%ymm8
   .byte  196,65,53,86,192                    // vorpd         %ymm8,%ymm9,%ymm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,10                              // jne           42d4 <_sk_store_8888_avx+0x9c>
+  .byte  117,10                              // jne           42f0 <_sk_store_8888_avx+0x9c>
   .byte  196,65,124,17,4,186                 // vmovups       %ymm8,(%r10,%rdi,4)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17948,9 +17981,9 @@ _sk_store_8888_avx:
   .byte  65,128,224,7                        // and           $0x7,%r8b
   .byte  65,254,200                          // dec           %r8b
   .byte  65,128,248,6                        // cmp           $0x6,%r8b
-  .byte  119,236                             // ja            42d0 <_sk_store_8888_avx+0x98>
+  .byte  119,236                             // ja            42ec <_sk_store_8888_avx+0x98>
   .byte  69,15,182,192                       // movzbl        %r8b,%r8d
-  .byte  76,141,13,85,0,0,0                  // lea           0x55(%rip),%r9        # 4344 <_sk_store_8888_avx+0x10c>
+  .byte  76,141,13,85,0,0,0                  // lea           0x55(%rip),%r9        # 4360 <_sk_store_8888_avx+0x10c>
   .byte  75,99,4,129                         // movslq        (%r9,%r8,4),%rax
   .byte  76,1,200                            // add           %r9,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -17964,7 +17997,7 @@ _sk_store_8888_avx:
   .byte  196,67,121,22,68,186,8,2            // vpextrd       $0x2,%xmm8,0x8(%r10,%rdi,4)
   .byte  196,67,121,22,68,186,4,1            // vpextrd       $0x1,%xmm8,0x4(%r10,%rdi,4)
   .byte  196,65,121,126,4,186                // vmovd         %xmm8,(%r10,%rdi,4)
-  .byte  235,143                             // jmp           42d0 <_sk_store_8888_avx+0x98>
+  .byte  235,143                             // jmp           42ec <_sk_store_8888_avx+0x98>
   .byte  15,31,0                             // nopl          (%rax)
   .byte  245                                 // cmc
   .byte  255                                 // (bad)
@@ -18002,7 +18035,7 @@ _sk_load_f16_avx:
   .byte  197,252,17,116,36,192               // vmovups       %ymm6,-0x40(%rsp)
   .byte  197,252,17,108,36,160               // vmovups       %ymm5,-0x60(%rsp)
   .byte  197,254,127,100,36,128              // vmovdqu       %ymm4,-0x80(%rsp)
-  .byte  15,133,141,2,0,0                    // jne           4617 <_sk_load_f16_avx+0x2b7>
+  .byte  15,133,141,2,0,0                    // jne           4633 <_sk_load_f16_avx+0x2b7>
   .byte  197,121,16,4,248                    // vmovupd       (%rax,%rdi,8),%xmm8
   .byte  197,249,16,84,248,16                // vmovupd       0x10(%rax,%rdi,8),%xmm2
   .byte  197,249,16,76,248,32                // vmovupd       0x20(%rax,%rdi,8),%xmm1
@@ -18020,13 +18053,13 @@ _sk_load_f16_avx:
   .byte  197,249,105,201                     // vpunpckhwd    %xmm1,%xmm0,%xmm1
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
-  .byte  196,98,125,24,37,133,34,0,0         // vbroadcastss  0x2285(%rip),%ymm12        # 6668 <_sk_callback_avx+0x422>
+  .byte  196,98,125,24,37,133,34,0,0         // vbroadcastss  0x2285(%rip),%ymm12        # 6684 <_sk_callback_avx+0x422>
   .byte  196,193,124,84,204                  // vandps        %ymm12,%ymm0,%ymm1
   .byte  197,252,87,193                      // vxorps        %ymm1,%ymm0,%ymm0
   .byte  196,195,125,25,198,1                // vextractf128  $0x1,%ymm0,%xmm14
-  .byte  196,98,121,24,29,113,34,0,0         // vbroadcastss  0x2271(%rip),%xmm11        # 666c <_sk_callback_avx+0x426>
+  .byte  196,98,121,24,29,113,34,0,0         // vbroadcastss  0x2271(%rip),%xmm11        # 6688 <_sk_callback_avx+0x426>
   .byte  196,193,8,87,219                    // vxorps        %xmm11,%xmm14,%xmm3
-  .byte  196,98,121,24,45,103,34,0,0         // vbroadcastss  0x2267(%rip),%xmm13        # 6670 <_sk_callback_avx+0x42a>
+  .byte  196,98,121,24,45,103,34,0,0         // vbroadcastss  0x2267(%rip),%xmm13        # 668c <_sk_callback_avx+0x42a>
   .byte  197,145,102,219                     // vpcmpgtd      %xmm3,%xmm13,%xmm3
   .byte  196,65,120,87,211                   // vxorps        %xmm11,%xmm0,%xmm10
   .byte  196,65,17,102,210                   // vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -18040,7 +18073,7 @@ _sk_load_f16_avx:
   .byte  196,227,125,24,195,1                // vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   .byte  197,252,86,193                      // vorps         %ymm1,%ymm0,%ymm0
   .byte  196,227,125,25,193,1                // vextractf128  $0x1,%ymm0,%xmm1
-  .byte  196,226,121,24,29,29,34,0,0         // vbroadcastss  0x221d(%rip),%xmm3        # 6674 <_sk_callback_avx+0x42e>
+  .byte  196,226,121,24,29,29,34,0,0         // vbroadcastss  0x221d(%rip),%xmm3        # 6690 <_sk_callback_avx+0x42e>
   .byte  197,241,254,203                     // vpaddd        %xmm3,%xmm1,%xmm1
   .byte  197,249,254,195                     // vpaddd        %xmm3,%xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
@@ -18133,29 +18166,29 @@ _sk_load_f16_avx:
   .byte  197,123,16,4,248                    // vmovsd        (%rax,%rdi,8),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,79                              // je            4676 <_sk_load_f16_avx+0x316>
+  .byte  116,79                              // je            4692 <_sk_load_f16_avx+0x316>
   .byte  197,57,22,68,248,8                  // vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,67                              // jb            4676 <_sk_load_f16_avx+0x316>
+  .byte  114,67                              // jb            4692 <_sk_load_f16_avx+0x316>
   .byte  197,251,16,84,248,16                // vmovsd        0x10(%rax,%rdi,8),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,68                              // je            4683 <_sk_load_f16_avx+0x323>
+  .byte  116,68                              // je            469f <_sk_load_f16_avx+0x323>
   .byte  197,233,22,84,248,24                // vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,56                              // jb            4683 <_sk_load_f16_avx+0x323>
+  .byte  114,56                              // jb            469f <_sk_load_f16_avx+0x323>
   .byte  197,251,16,76,248,32                // vmovsd        0x20(%rax,%rdi,8),%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,70,253,255,255               // je            43a1 <_sk_load_f16_avx+0x41>
+  .byte  15,132,70,253,255,255               // je            43bd <_sk_load_f16_avx+0x41>
   .byte  197,241,22,76,248,40                // vmovhpd       0x28(%rax,%rdi,8),%xmm1,%xmm1
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,54,253,255,255               // jb            43a1 <_sk_load_f16_avx+0x41>
+  .byte  15,130,54,253,255,255               // jb            43bd <_sk_load_f16_avx+0x41>
   .byte  197,122,126,76,248,48               // vmovq         0x30(%rax,%rdi,8),%xmm9
-  .byte  233,43,253,255,255                  // jmpq          43a1 <_sk_load_f16_avx+0x41>
+  .byte  233,43,253,255,255                  // jmpq          43bd <_sk_load_f16_avx+0x41>
   .byte  197,241,87,201                      // vxorpd        %xmm1,%xmm1,%xmm1
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,30,253,255,255                  // jmpq          43a1 <_sk_load_f16_avx+0x41>
+  .byte  233,30,253,255,255                  // jmpq          43bd <_sk_load_f16_avx+0x41>
   .byte  197,241,87,201                      // vxorpd        %xmm1,%xmm1,%xmm1
-  .byte  233,21,253,255,255                  // jmpq          43a1 <_sk_load_f16_avx+0x41>
+  .byte  233,21,253,255,255                  // jmpq          43bd <_sk_load_f16_avx+0x41>
 
 HIDDEN _sk_gather_f16_avx
 .globl _sk_gather_f16_avx
@@ -18219,13 +18252,13 @@ _sk_gather_f16_avx:
   .byte  197,249,105,210                     // vpunpckhwd    %xmm2,%xmm0,%xmm2
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,194,1                // vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
-  .byte  196,98,125,24,37,225,30,0,0         // vbroadcastss  0x1ee1(%rip),%ymm12        # 6678 <_sk_callback_avx+0x432>
+  .byte  196,98,125,24,37,225,30,0,0         // vbroadcastss  0x1ee1(%rip),%ymm12        # 6694 <_sk_callback_avx+0x432>
   .byte  196,193,124,84,212                  // vandps        %ymm12,%ymm0,%ymm2
   .byte  197,252,87,194                      // vxorps        %ymm2,%ymm0,%ymm0
   .byte  196,195,125,25,198,1                // vextractf128  $0x1,%ymm0,%xmm14
-  .byte  196,98,121,24,29,205,30,0,0         // vbroadcastss  0x1ecd(%rip),%xmm11        # 667c <_sk_callback_avx+0x436>
+  .byte  196,98,121,24,29,205,30,0,0         // vbroadcastss  0x1ecd(%rip),%xmm11        # 6698 <_sk_callback_avx+0x436>
   .byte  196,193,8,87,219                    // vxorps        %xmm11,%xmm14,%xmm3
-  .byte  196,98,121,24,45,195,30,0,0         // vbroadcastss  0x1ec3(%rip),%xmm13        # 6680 <_sk_callback_avx+0x43a>
+  .byte  196,98,121,24,45,195,30,0,0         // vbroadcastss  0x1ec3(%rip),%xmm13        # 669c <_sk_callback_avx+0x43a>
   .byte  197,145,102,219                     // vpcmpgtd      %xmm3,%xmm13,%xmm3
   .byte  196,65,120,87,211                   // vxorps        %xmm11,%xmm0,%xmm10
   .byte  196,65,17,102,210                   // vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -18239,7 +18272,7 @@ _sk_gather_f16_avx:
   .byte  196,227,125,24,195,1                // vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   .byte  197,252,86,194                      // vorps         %ymm2,%ymm0,%ymm0
   .byte  196,227,125,25,194,1                // vextractf128  $0x1,%ymm0,%xmm2
-  .byte  196,226,121,24,29,121,30,0,0        // vbroadcastss  0x1e79(%rip),%xmm3        # 6684 <_sk_callback_avx+0x43e>
+  .byte  196,226,121,24,29,121,30,0,0        // vbroadcastss  0x1e79(%rip),%xmm3        # 66a0 <_sk_callback_avx+0x43e>
   .byte  197,233,254,211                     // vpaddd        %xmm3,%xmm2,%xmm2
   .byte  197,249,254,195                     // vpaddd        %xmm3,%xmm0,%xmm0
   .byte  196,227,125,24,194,1                // vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
@@ -18343,12 +18376,12 @@ _sk_store_f16_avx:
   .byte  197,252,17,52,36                    // vmovups       %ymm6,(%rsp)
   .byte  197,252,17,108,36,224               // vmovups       %ymm5,-0x20(%rsp)
   .byte  197,252,17,100,36,192               // vmovups       %ymm4,-0x40(%rsp)
-  .byte  196,98,125,24,13,146,28,0,0         // vbroadcastss  0x1c92(%rip),%ymm9        # 6688 <_sk_callback_avx+0x442>
+  .byte  196,98,125,24,13,146,28,0,0         // vbroadcastss  0x1c92(%rip),%ymm9        # 66a4 <_sk_callback_avx+0x442>
   .byte  196,65,124,84,209                   // vandps        %ymm9,%ymm0,%ymm10
   .byte  197,252,17,68,36,128                // vmovups       %ymm0,-0x80(%rsp)
   .byte  196,65,124,87,218                   // vxorps        %ymm10,%ymm0,%ymm11
   .byte  196,67,125,25,220,1                 // vextractf128  $0x1,%ymm11,%xmm12
-  .byte  196,98,121,24,5,119,28,0,0          // vbroadcastss  0x1c77(%rip),%xmm8        # 668c <_sk_callback_avx+0x446>
+  .byte  196,98,121,24,5,119,28,0,0          // vbroadcastss  0x1c77(%rip),%xmm8        # 66a8 <_sk_callback_avx+0x446>
   .byte  196,65,57,102,236                   // vpcmpgtd      %xmm12,%xmm8,%xmm13
   .byte  196,65,57,102,243                   // vpcmpgtd      %xmm11,%xmm8,%xmm14
   .byte  196,67,13,24,237,1                  // vinsertf128   $0x1,%xmm13,%ymm14,%ymm13
@@ -18358,7 +18391,7 @@ _sk_store_f16_avx:
   .byte  196,67,13,24,242,1                  // vinsertf128   $0x1,%xmm10,%ymm14,%ymm14
   .byte  196,193,33,114,211,13               // vpsrld        $0xd,%xmm11,%xmm11
   .byte  196,193,25,114,212,13               // vpsrld        $0xd,%xmm12,%xmm12
-  .byte  196,98,125,24,21,62,28,0,0          // vbroadcastss  0x1c3e(%rip),%ymm10        # 6690 <_sk_callback_avx+0x44a>
+  .byte  196,98,125,24,21,62,28,0,0          // vbroadcastss  0x1c3e(%rip),%ymm10        # 66ac <_sk_callback_avx+0x44a>
   .byte  196,65,12,86,242                    // vorps         %ymm10,%ymm14,%ymm14
   .byte  196,67,125,25,247,1                 // vextractf128  $0x1,%ymm14,%xmm15
   .byte  196,65,1,254,228                    // vpaddd        %xmm12,%xmm15,%xmm12
@@ -18440,7 +18473,7 @@ _sk_store_f16_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,66                              // jne           4c30 <_sk_store_f16_avx+0x25e>
+  .byte  117,66                              // jne           4c4c <_sk_store_f16_avx+0x25e>
   .byte  197,120,17,28,248                   // vmovups       %xmm11,(%rax,%rdi,8)
   .byte  197,120,17,84,248,16                // vmovups       %xmm10,0x10(%rax,%rdi,8)
   .byte  197,120,17,76,248,32                // vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -18456,22 +18489,22 @@ _sk_store_f16_avx:
   .byte  255,224                             // jmpq          *%rax
   .byte  197,121,214,28,248                  // vmovq         %xmm11,(%rax,%rdi,8)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,202                             // je            4c05 <_sk_store_f16_avx+0x233>
+  .byte  116,202                             // je            4c21 <_sk_store_f16_avx+0x233>
   .byte  197,121,23,92,248,8                 // vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,190                             // jb            4c05 <_sk_store_f16_avx+0x233>
+  .byte  114,190                             // jb            4c21 <_sk_store_f16_avx+0x233>
   .byte  197,121,214,84,248,16               // vmovq         %xmm10,0x10(%rax,%rdi,8)
-  .byte  116,182                             // je            4c05 <_sk_store_f16_avx+0x233>
+  .byte  116,182                             // je            4c21 <_sk_store_f16_avx+0x233>
   .byte  197,121,23,84,248,24                // vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,170                             // jb            4c05 <_sk_store_f16_avx+0x233>
+  .byte  114,170                             // jb            4c21 <_sk_store_f16_avx+0x233>
   .byte  197,121,214,76,248,32               // vmovq         %xmm9,0x20(%rax,%rdi,8)
-  .byte  116,162                             // je            4c05 <_sk_store_f16_avx+0x233>
+  .byte  116,162                             // je            4c21 <_sk_store_f16_avx+0x233>
   .byte  197,121,23,76,248,40                // vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,150                             // jb            4c05 <_sk_store_f16_avx+0x233>
+  .byte  114,150                             // jb            4c21 <_sk_store_f16_avx+0x233>
   .byte  197,121,214,68,248,48               // vmovq         %xmm8,0x30(%rax,%rdi,8)
-  .byte  235,142                             // jmp           4c05 <_sk_store_f16_avx+0x233>
+  .byte  235,142                             // jmp           4c21 <_sk_store_f16_avx+0x233>
 
 HIDDEN _sk_load_u16_be_avx
 .globl _sk_load_u16_be_avx
@@ -18481,7 +18514,7 @@ _sk_load_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,253,0,0,0                    // jne           4d8a <_sk_load_u16_be_avx+0x113>
+  .byte  15,133,253,0,0,0                    // jne           4da6 <_sk_load_u16_be_avx+0x113>
   .byte  196,65,121,16,4,64                  // vmovupd       (%r8,%rax,2),%xmm8
   .byte  196,193,121,16,84,64,16             // vmovupd       0x10(%r8,%rax,2),%xmm2
   .byte  196,193,121,16,92,64,32             // vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -18503,7 +18536,7 @@ _sk_load_u16_be_avx:
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,29,150,25,0,0         // vbroadcastss  0x1996(%rip),%ymm11        # 6694 <_sk_callback_avx+0x44e>
+  .byte  196,98,125,24,29,150,25,0,0         // vbroadcastss  0x1996(%rip),%ymm11        # 66b0 <_sk_callback_avx+0x44e>
   .byte  196,193,124,89,195                  // vmulps        %ymm11,%ymm0,%ymm0
   .byte  197,177,109,202                     // vpunpckhqdq   %xmm2,%xmm9,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -18537,29 +18570,29 @@ _sk_load_u16_be_avx:
   .byte  196,65,123,16,4,64                  // vmovsd        (%r8,%rax,2),%xmm8
   .byte  196,65,49,239,201                   // vpxor         %xmm9,%xmm9,%xmm9
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,85                              // je            4df0 <_sk_load_u16_be_avx+0x179>
+  .byte  116,85                              // je            4e0c <_sk_load_u16_be_avx+0x179>
   .byte  196,65,57,22,68,64,8                // vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,72                              // jb            4df0 <_sk_load_u16_be_avx+0x179>
+  .byte  114,72                              // jb            4e0c <_sk_load_u16_be_avx+0x179>
   .byte  196,193,123,16,84,64,16             // vmovsd        0x10(%r8,%rax,2),%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  116,72                              // je            4dfd <_sk_load_u16_be_avx+0x186>
+  .byte  116,72                              // je            4e19 <_sk_load_u16_be_avx+0x186>
   .byte  196,193,105,22,84,64,24             // vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,59                              // jb            4dfd <_sk_load_u16_be_avx+0x186>
+  .byte  114,59                              // jb            4e19 <_sk_load_u16_be_avx+0x186>
   .byte  196,193,123,16,92,64,32             // vmovsd        0x20(%r8,%rax,2),%xmm3
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  15,132,213,254,255,255              // je            4ca8 <_sk_load_u16_be_avx+0x31>
+  .byte  15,132,213,254,255,255              // je            4cc4 <_sk_load_u16_be_avx+0x31>
   .byte  196,193,97,22,92,64,40              // vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  15,130,196,254,255,255              // jb            4ca8 <_sk_load_u16_be_avx+0x31>
+  .byte  15,130,196,254,255,255              // jb            4cc4 <_sk_load_u16_be_avx+0x31>
   .byte  196,65,122,126,76,64,48             // vmovq         0x30(%r8,%rax,2),%xmm9
-  .byte  233,184,254,255,255                 // jmpq          4ca8 <_sk_load_u16_be_avx+0x31>
+  .byte  233,184,254,255,255                 // jmpq          4cc4 <_sk_load_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
   .byte  197,233,87,210                      // vxorpd        %xmm2,%xmm2,%xmm2
-  .byte  233,171,254,255,255                 // jmpq          4ca8 <_sk_load_u16_be_avx+0x31>
+  .byte  233,171,254,255,255                 // jmpq          4cc4 <_sk_load_u16_be_avx+0x31>
   .byte  197,225,87,219                      // vxorpd        %xmm3,%xmm3,%xmm3
-  .byte  233,162,254,255,255                 // jmpq          4ca8 <_sk_load_u16_be_avx+0x31>
+  .byte  233,162,254,255,255                 // jmpq          4cc4 <_sk_load_u16_be_avx+0x31>
 
 HIDDEN _sk_load_rgb_u16_be_avx
 .globl _sk_load_rgb_u16_be_avx
@@ -18569,7 +18602,7 @@ _sk_load_rgb_u16_be_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,127                        // lea           (%rdi,%rdi,2),%rax
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  15,133,243,0,0,0                    // jne           4f0b <_sk_load_rgb_u16_be_avx+0x105>
+  .byte  15,133,243,0,0,0                    // jne           4f27 <_sk_load_rgb_u16_be_avx+0x105>
   .byte  196,193,122,111,4,64                // vmovdqu       (%r8,%rax,2),%xmm0
   .byte  196,193,122,111,84,64,12            // vmovdqu       0xc(%r8,%rax,2),%xmm2
   .byte  196,193,122,111,76,64,24            // vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -18596,7 +18629,7 @@ _sk_load_rgb_u16_be_avx:
   .byte  196,226,121,51,192                  // vpmovzxwd     %xmm0,%xmm0
   .byte  196,227,125,24,193,1                // vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   .byte  197,252,91,192                      // vcvtdq2ps     %ymm0,%ymm0
-  .byte  196,98,125,24,29,246,23,0,0         // vbroadcastss  0x17f6(%rip),%ymm11        # 6698 <_sk_callback_avx+0x452>
+  .byte  196,98,125,24,29,246,23,0,0         // vbroadcastss  0x17f6(%rip),%ymm11        # 66b4 <_sk_callback_avx+0x452>
   .byte  196,193,124,89,195                  // vmulps        %ymm11,%ymm0,%ymm0
   .byte  197,185,109,202                     // vpunpckhqdq   %xmm2,%xmm8,%xmm1
   .byte  197,233,113,241,8                   // vpsllw        $0x8,%xmm1,%xmm2
@@ -18617,41 +18650,41 @@ _sk_load_rgb_u16_be_avx:
   .byte  197,252,91,210                      // vcvtdq2ps     %ymm2,%ymm2
   .byte  196,193,108,89,211                  // vmulps        %ymm11,%ymm2,%ymm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,29,147,23,0,0        // vbroadcastss  0x1793(%rip),%ymm3        # 669c <_sk_callback_avx+0x456>
+  .byte  196,226,125,24,29,147,23,0,0        // vbroadcastss  0x1793(%rip),%ymm3        # 66b8 <_sk_callback_avx+0x456>
   .byte  255,224                             // jmpq          *%rax
   .byte  196,193,121,110,4,64                // vmovd         (%r8,%rax,2),%xmm0
   .byte  196,193,121,196,68,64,4,2           // vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  117,5                               // jne           4f24 <_sk_load_rgb_u16_be_avx+0x11e>
-  .byte  233,40,255,255,255                  // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  117,5                               // jne           4f40 <_sk_load_rgb_u16_be_avx+0x11e>
+  .byte  233,40,255,255,255                  // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
   .byte  196,193,121,110,76,64,6             // vmovd         0x6(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,68,64,10,2           // vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,26                              // jb            4f53 <_sk_load_rgb_u16_be_avx+0x14d>
+  .byte  114,26                              // jb            4f6f <_sk_load_rgb_u16_be_avx+0x14d>
   .byte  196,193,121,110,76,64,12            // vmovd         0xc(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,84,64,16,2          // vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  117,10                              // jne           4f58 <_sk_load_rgb_u16_be_avx+0x152>
-  .byte  233,249,254,255,255                 // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
-  .byte  233,244,254,255,255                 // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           4f74 <_sk_load_rgb_u16_be_avx+0x152>
+  .byte  233,249,254,255,255                 // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,244,254,255,255                 // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
   .byte  196,193,121,110,76,64,18            // vmovd         0x12(%r8,%rax,2),%xmm1
   .byte  196,65,113,196,76,64,22,2           // vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,26                              // jb            4f87 <_sk_load_rgb_u16_be_avx+0x181>
+  .byte  114,26                              // jb            4fa3 <_sk_load_rgb_u16_be_avx+0x181>
   .byte  196,193,121,110,76,64,24            // vmovd         0x18(%r8,%rax,2),%xmm1
   .byte  196,193,113,196,76,64,28,2          // vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  117,10                              // jne           4f8c <_sk_load_rgb_u16_be_avx+0x186>
-  .byte  233,197,254,255,255                 // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
-  .byte  233,192,254,255,255                 // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  117,10                              // jne           4fa8 <_sk_load_rgb_u16_be_avx+0x186>
+  .byte  233,197,254,255,255                 // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,192,254,255,255                 // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
   .byte  196,193,121,110,92,64,30            // vmovd         0x1e(%r8,%rax,2),%xmm3
   .byte  196,65,97,196,92,64,34,2            // vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,20                              // jb            4fb5 <_sk_load_rgb_u16_be_avx+0x1af>
+  .byte  114,20                              // jb            4fd1 <_sk_load_rgb_u16_be_avx+0x1af>
   .byte  196,193,121,110,92,64,36            // vmovd         0x24(%r8,%rax,2),%xmm3
   .byte  196,193,97,196,92,64,40,2           // vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  .byte  233,151,254,255,255                 // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
-  .byte  233,146,254,255,255                 // jmpq          4e4c <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,151,254,255,255                 // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
+  .byte  233,146,254,255,255                 // jmpq          4e68 <_sk_load_rgb_u16_be_avx+0x46>
 
 HIDDEN _sk_store_u16_be_avx
 .globl _sk_store_u16_be_avx
@@ -18660,7 +18693,7 @@ _sk_store_u16_be_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  72,141,4,189,0,0,0,0                // lea           0x0(,%rdi,4),%rax
-  .byte  196,98,125,24,5,208,22,0,0          // vbroadcastss  0x16d0(%rip),%ymm8        # 66a0 <_sk_callback_avx+0x45a>
+  .byte  196,98,125,24,5,208,22,0,0          // vbroadcastss  0x16d0(%rip),%ymm8        # 66bc <_sk_callback_avx+0x45a>
   .byte  196,65,124,89,200                   // vmulps        %ymm8,%ymm0,%ymm9
   .byte  196,65,125,91,201                   // vcvtps2dq     %ymm9,%ymm9
   .byte  196,67,125,25,202,1                 // vextractf128  $0x1,%ymm9,%xmm10
@@ -18698,7 +18731,7 @@ _sk_store_u16_be_avx:
   .byte  196,65,17,98,200                    // vpunpckldq    %xmm8,%xmm13,%xmm9
   .byte  196,65,17,106,192                   // vpunpckhdq    %xmm8,%xmm13,%xmm8
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,31                              // jne           50b4 <_sk_store_u16_be_avx+0xfa>
+  .byte  117,31                              // jne           50d0 <_sk_store_u16_be_avx+0xfa>
   .byte  196,65,120,17,28,64                 // vmovups       %xmm11,(%r8,%rax,2)
   .byte  196,65,120,17,84,64,16              // vmovups       %xmm10,0x10(%r8,%rax,2)
   .byte  196,65,120,17,76,64,32              // vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -18707,22 +18740,22 @@ _sk_store_u16_be_avx:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,214,28,64                // vmovq         %xmm11,(%r8,%rax,2)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  116,240                             // je            50cc <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,23,92,64,8               // vmovhpd       %xmm11,0x8(%r8,%rax,2)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  114,227                             // jb            50cc <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,214,84,64,16             // vmovq         %xmm10,0x10(%r8,%rax,2)
-  .byte  116,218                             // je            50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  116,218                             // je            50cc <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,23,84,64,24              // vmovhpd       %xmm10,0x18(%r8,%rax,2)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  114,205                             // jb            50cc <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,214,76,64,32             // vmovq         %xmm9,0x20(%r8,%rax,2)
-  .byte  116,196                             // je            50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  116,196                             // je            50cc <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,23,76,64,40              // vmovhpd       %xmm9,0x28(%r8,%rax,2)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,183                             // jb            50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  114,183                             // jb            50cc <_sk_store_u16_be_avx+0xf6>
   .byte  196,65,121,214,68,64,48             // vmovq         %xmm8,0x30(%r8,%rax,2)
-  .byte  235,174                             // jmp           50b0 <_sk_store_u16_be_avx+0xf6>
+  .byte  235,174                             // jmp           50cc <_sk_store_u16_be_avx+0xf6>
 
 HIDDEN _sk_load_f32_avx
 .globl _sk_load_f32_avx
@@ -18730,10 +18763,10 @@ FUNCTION(_sk_load_f32_avx)
 _sk_load_f32_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  119,110                             // ja            5178 <_sk_load_f32_avx+0x76>
+  .byte  119,110                             // ja            5194 <_sk_load_f32_avx+0x76>
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,141,12,189,0,0,0,0               // lea           0x0(,%rdi,4),%r9
-  .byte  76,141,21,132,0,0,0                 // lea           0x84(%rip),%r10        # 51a0 <_sk_load_f32_avx+0x9e>
+  .byte  76,141,21,132,0,0,0                 // lea           0x84(%rip),%r10        # 51bc <_sk_load_f32_avx+0x9e>
   .byte  73,99,4,138                         // movslq        (%r10,%rcx,4),%rax
   .byte  76,1,208                            // add           %r10,%rax
   .byte  255,224                             // jmpq          *%rax
@@ -18792,7 +18825,7 @@ _sk_store_f32_avx:
   .byte  196,65,37,20,196                    // vunpcklpd     %ymm12,%ymm11,%ymm8
   .byte  196,65,37,21,220                    // vunpckhpd     %ymm12,%ymm11,%ymm11
   .byte  72,133,201                          // test          %rcx,%rcx
-  .byte  117,55                              // jne           522d <_sk_store_f32_avx+0x6d>
+  .byte  117,55                              // jne           5249 <_sk_store_f32_avx+0x6d>
   .byte  196,67,45,24,225,1                  // vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   .byte  196,67,61,24,235,1                  // vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   .byte  196,67,45,6,201,49                  // vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -18805,22 +18838,22 @@ _sk_store_f32_avx:
   .byte  255,224                             // jmpq          *%rax
   .byte  196,65,121,17,20,128                // vmovupd       %xmm10,(%r8,%rax,4)
   .byte  72,131,249,1                        // cmp           $0x1,%rcx
-  .byte  116,240                             // je            5229 <_sk_store_f32_avx+0x69>
+  .byte  116,240                             // je            5245 <_sk_store_f32_avx+0x69>
   .byte  196,65,121,17,76,128,16             // vmovupd       %xmm9,0x10(%r8,%rax,4)
   .byte  72,131,249,3                        // cmp           $0x3,%rcx
-  .byte  114,227                             // jb            5229 <_sk_store_f32_avx+0x69>
+  .byte  114,227                             // jb            5245 <_sk_store_f32_avx+0x69>
   .byte  196,65,121,17,68,128,32             // vmovupd       %xmm8,0x20(%r8,%rax,4)
-  .byte  116,218                             // je            5229 <_sk_store_f32_avx+0x69>
+  .byte  116,218                             // je            5245 <_sk_store_f32_avx+0x69>
   .byte  196,65,121,17,92,128,48             // vmovupd       %xmm11,0x30(%r8,%rax,4)
   .byte  72,131,249,5                        // cmp           $0x5,%rcx
-  .byte  114,205                             // jb            5229 <_sk_store_f32_avx+0x69>
+  .byte  114,205                             // jb            5245 <_sk_store_f32_avx+0x69>
   .byte  196,67,125,25,84,128,64,1           // vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  .byte  116,195                             // je            5229 <_sk_store_f32_avx+0x69>
+  .byte  116,195                             // je            5245 <_sk_store_f32_avx+0x69>
   .byte  196,67,125,25,76,128,80,1           // vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   .byte  72,131,249,7                        // cmp           $0x7,%rcx
-  .byte  114,181                             // jb            5229 <_sk_store_f32_avx+0x69>
+  .byte  114,181                             // jb            5245 <_sk_store_f32_avx+0x69>
   .byte  196,67,125,25,68,128,96,1           // vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  .byte  235,171                             // jmp           5229 <_sk_store_f32_avx+0x69>
+  .byte  235,171                             // jmp           5245 <_sk_store_f32_avx+0x69>
 
 HIDDEN _sk_clamp_x_avx
 .globl _sk_clamp_x_avx
@@ -18926,12 +18959,12 @@ HIDDEN _sk_luminance_to_alpha_avx
 .globl _sk_luminance_to_alpha_avx
 FUNCTION(_sk_luminance_to_alpha_avx)
 _sk_luminance_to_alpha_avx:
-  .byte  196,226,125,24,29,247,18,0,0        // vbroadcastss  0x12f7(%rip),%ymm3        # 66a4 <_sk_callback_avx+0x45e>
+  .byte  196,226,125,24,29,247,18,0,0        // vbroadcastss  0x12f7(%rip),%ymm3        # 66c0 <_sk_callback_avx+0x45e>
   .byte  197,252,89,195                      // vmulps        %ymm3,%ymm0,%ymm0
-  .byte  196,226,125,24,29,238,18,0,0        // vbroadcastss  0x12ee(%rip),%ymm3        # 66a8 <_sk_callback_avx+0x462>
+  .byte  196,226,125,24,29,238,18,0,0        // vbroadcastss  0x12ee(%rip),%ymm3        # 66c4 <_sk_callback_avx+0x462>
   .byte  197,244,89,203                      // vmulps        %ymm3,%ymm1,%ymm1
   .byte  197,252,88,193                      // vaddps        %ymm1,%ymm0,%ymm0
-  .byte  196,226,125,24,13,225,18,0,0        // vbroadcastss  0x12e1(%rip),%ymm1        # 66ac <_sk_callback_avx+0x466>
+  .byte  196,226,125,24,13,225,18,0,0        // vbroadcastss  0x12e1(%rip),%ymm1        # 66c8 <_sk_callback_avx+0x466>
   .byte  197,236,89,201                      // vmulps        %ymm1,%ymm2,%ymm1
   .byte  197,252,88,217                      // vaddps        %ymm1,%ymm0,%ymm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19112,9 +19145,9 @@ _sk_evenly_spaced_gradient_avx:
   .byte  72,139,24                           // mov           (%rax),%rbx
   .byte  72,139,104,8                        // mov           0x8(%rax),%rbp
   .byte  72,255,203                          // dec           %rbx
-  .byte  120,7                               // js            5688 <_sk_evenly_spaced_gradient_avx+0x1f>
+  .byte  120,7                               // js            56a4 <_sk_evenly_spaced_gradient_avx+0x1f>
   .byte  196,225,242,42,203                  // vcvtsi2ss     %rbx,%xmm1,%xmm1
-  .byte  235,21                              // jmp           569d <_sk_evenly_spaced_gradient_avx+0x34>
+  .byte  235,21                              // jmp           56b9 <_sk_evenly_spaced_gradient_avx+0x34>
   .byte  73,137,216                          // mov           %rbx,%r8
   .byte  73,209,232                          // shr           %r8
   .byte  131,227,1                           // and           $0x1,%ebx
@@ -19281,12 +19314,12 @@ _sk_gradient_avx:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  197,244,87,201                      // vxorps        %ymm1,%ymm1,%ymm1
   .byte  73,131,248,2                        // cmp           $0x2,%r8
-  .byte  114,80                              // jb            5a2b <_sk_gradient_avx+0x69>
+  .byte  114,80                              // jb            5a47 <_sk_gradient_avx+0x69>
   .byte  72,139,88,72                        // mov           0x48(%rax),%rbx
   .byte  73,255,200                          // dec           %r8
   .byte  72,131,195,4                        // add           $0x4,%rbx
   .byte  196,65,52,87,201                    // vxorps        %ymm9,%ymm9,%ymm9
-  .byte  196,98,125,24,21,188,12,0,0         // vbroadcastss  0xcbc(%rip),%ymm10        # 66b0 <_sk_callback_avx+0x46a>
+  .byte  196,98,125,24,21,188,12,0,0         // vbroadcastss  0xcbc(%rip),%ymm10        # 66cc <_sk_callback_avx+0x46a>
   .byte  197,244,87,201                      // vxorps        %ymm1,%ymm1,%ymm1
   .byte  196,98,125,24,3                     // vbroadcastss  (%rbx),%ymm8
   .byte  197,60,194,192,2                    // vcmpleps      %ymm0,%ymm8,%ymm8
@@ -19298,7 +19331,7 @@ _sk_gradient_avx:
   .byte  196,227,117,24,202,1                // vinsertf128   $0x1,%xmm2,%ymm1,%ymm1
   .byte  72,131,195,4                        // add           $0x4,%rbx
   .byte  73,255,200                          // dec           %r8
-  .byte  117,205                             // jne           59f8 <_sk_gradient_avx+0x36>
+  .byte  117,205                             // jne           5a14 <_sk_gradient_avx+0x36>
   .byte  196,195,249,22,200,1                // vpextrq       $0x1,%xmm1,%r8
   .byte  69,137,193                          // mov           %r8d,%r9d
   .byte  73,193,232,32                       // shr           $0x20,%r8
@@ -19480,27 +19513,27 @@ _sk_xy_to_unit_angle_avx:
   .byte  196,65,52,95,226                    // vmaxps        %ymm10,%ymm9,%ymm12
   .byte  196,65,36,94,220                    // vdivps        %ymm12,%ymm11,%ymm11
   .byte  196,65,36,89,227                    // vmulps        %ymm11,%ymm11,%ymm12
-  .byte  196,98,125,24,45,224,8,0,0          // vbroadcastss  0x8e0(%rip),%ymm13        # 66b4 <_sk_callback_avx+0x46e>
+  .byte  196,98,125,24,45,224,8,0,0          // vbroadcastss  0x8e0(%rip),%ymm13        # 66d0 <_sk_callback_avx+0x46e>
   .byte  196,65,28,89,237                    // vmulps        %ymm13,%ymm12,%ymm13
-  .byte  196,98,125,24,53,214,8,0,0          // vbroadcastss  0x8d6(%rip),%ymm14        # 66b8 <_sk_callback_avx+0x472>
+  .byte  196,98,125,24,53,214,8,0,0          // vbroadcastss  0x8d6(%rip),%ymm14        # 66d4 <_sk_callback_avx+0x472>
   .byte  196,65,20,88,238                    // vaddps        %ymm14,%ymm13,%ymm13
   .byte  196,65,28,89,237                    // vmulps        %ymm13,%ymm12,%ymm13
-  .byte  196,98,125,24,53,199,8,0,0          // vbroadcastss  0x8c7(%rip),%ymm14        # 66bc <_sk_callback_avx+0x476>
+  .byte  196,98,125,24,53,199,8,0,0          // vbroadcastss  0x8c7(%rip),%ymm14        # 66d8 <_sk_callback_avx+0x476>
   .byte  196,65,20,88,238                    // vaddps        %ymm14,%ymm13,%ymm13
   .byte  196,65,28,89,229                    // vmulps        %ymm13,%ymm12,%ymm12
-  .byte  196,98,125,24,45,184,8,0,0          // vbroadcastss  0x8b8(%rip),%ymm13        # 66c0 <_sk_callback_avx+0x47a>
+  .byte  196,98,125,24,45,184,8,0,0          // vbroadcastss  0x8b8(%rip),%ymm13        # 66dc <_sk_callback_avx+0x47a>
   .byte  196,65,28,88,229                    // vaddps        %ymm13,%ymm12,%ymm12
   .byte  196,65,36,89,220                    // vmulps        %ymm12,%ymm11,%ymm11
   .byte  196,65,52,194,202,1                 // vcmpltps      %ymm10,%ymm9,%ymm9
-  .byte  196,98,125,24,21,163,8,0,0          // vbroadcastss  0x8a3(%rip),%ymm10        # 66c4 <_sk_callback_avx+0x47e>
+  .byte  196,98,125,24,21,163,8,0,0          // vbroadcastss  0x8a3(%rip),%ymm10        # 66e0 <_sk_callback_avx+0x47e>
   .byte  196,65,44,92,211                    // vsubps        %ymm11,%ymm10,%ymm10
   .byte  196,67,37,74,202,144                // vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   .byte  196,193,124,194,192,1               // vcmpltps      %ymm8,%ymm0,%ymm0
-  .byte  196,98,125,24,21,141,8,0,0          // vbroadcastss  0x88d(%rip),%ymm10        # 66c8 <_sk_callback_avx+0x482>
+  .byte  196,98,125,24,21,141,8,0,0          // vbroadcastss  0x88d(%rip),%ymm10        # 66e4 <_sk_callback_avx+0x482>
   .byte  196,65,44,92,209                    // vsubps        %ymm9,%ymm10,%ymm10
   .byte  196,195,53,74,194,0                 // vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   .byte  196,65,116,194,200,1                // vcmpltps      %ymm8,%ymm1,%ymm9
-  .byte  196,98,125,24,21,119,8,0,0          // vbroadcastss  0x877(%rip),%ymm10        # 66cc <_sk_callback_avx+0x486>
+  .byte  196,98,125,24,21,119,8,0,0          // vbroadcastss  0x877(%rip),%ymm10        # 66e8 <_sk_callback_avx+0x486>
   .byte  197,44,92,208                       // vsubps        %ymm0,%ymm10,%ymm10
   .byte  196,195,125,74,194,144              // vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   .byte  196,65,124,194,200,3                // vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -19524,7 +19557,7 @@ HIDDEN _sk_save_xy_avx
 FUNCTION(_sk_save_xy_avx)
 _sk_save_xy_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,65,8,0,0            // vbroadcastss  0x841(%rip),%ymm8        # 66d0 <_sk_callback_avx+0x48a>
+  .byte  196,98,125,24,5,65,8,0,0            // vbroadcastss  0x841(%rip),%ymm8        # 66ec <_sk_callback_avx+0x48a>
   .byte  196,65,124,88,200                   // vaddps        %ymm8,%ymm0,%ymm9
   .byte  196,67,125,8,209,1                  // vroundps      $0x1,%ymm9,%ymm10
   .byte  196,65,52,92,202                    // vsubps        %ymm10,%ymm9,%ymm9
@@ -19561,9 +19594,9 @@ HIDDEN _sk_bilinear_nx_avx
 FUNCTION(_sk_bilinear_nx_avx)
 _sk_bilinear_nx_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,205,7,0,0          // vbroadcastss  0x7cd(%rip),%ymm0        # 66d4 <_sk_callback_avx+0x48e>
+  .byte  196,226,125,24,5,205,7,0,0          // vbroadcastss  0x7cd(%rip),%ymm0        # 66f0 <_sk_callback_avx+0x48e>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,196,7,0,0           // vbroadcastss  0x7c4(%rip),%ymm8        # 66d8 <_sk_callback_avx+0x492>
+  .byte  196,98,125,24,5,196,7,0,0           // vbroadcastss  0x7c4(%rip),%ymm8        # 66f4 <_sk_callback_avx+0x492>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19574,7 +19607,7 @@ HIDDEN _sk_bilinear_px_avx
 FUNCTION(_sk_bilinear_px_avx)
 _sk_bilinear_px_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,172,7,0,0          // vbroadcastss  0x7ac(%rip),%ymm0        # 66dc <_sk_callback_avx+0x496>
+  .byte  196,226,125,24,5,172,7,0,0          // vbroadcastss  0x7ac(%rip),%ymm0        # 66f8 <_sk_callback_avx+0x496>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -19586,9 +19619,9 @@ HIDDEN _sk_bilinear_ny_avx
 FUNCTION(_sk_bilinear_ny_avx)
 _sk_bilinear_ny_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,144,7,0,0         // vbroadcastss  0x790(%rip),%ymm1        # 66e0 <_sk_callback_avx+0x49a>
+  .byte  196,226,125,24,13,144,7,0,0         // vbroadcastss  0x790(%rip),%ymm1        # 66fc <_sk_callback_avx+0x49a>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,134,7,0,0           // vbroadcastss  0x786(%rip),%ymm8        # 66e4 <_sk_callback_avx+0x49e>
+  .byte  196,98,125,24,5,134,7,0,0           // vbroadcastss  0x786(%rip),%ymm8        # 6700 <_sk_callback_avx+0x49e>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19599,7 +19632,7 @@ HIDDEN _sk_bilinear_py_avx
 FUNCTION(_sk_bilinear_py_avx)
 _sk_bilinear_py_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,110,7,0,0         // vbroadcastss  0x76e(%rip),%ymm1        # 66e8 <_sk_callback_avx+0x4a2>
+  .byte  196,226,125,24,13,110,7,0,0         // vbroadcastss  0x76e(%rip),%ymm1        # 6704 <_sk_callback_avx+0x4a2>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -19611,14 +19644,14 @@ HIDDEN _sk_bicubic_n3x_avx
 FUNCTION(_sk_bicubic_n3x_avx)
 _sk_bicubic_n3x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,81,7,0,0           // vbroadcastss  0x751(%rip),%ymm0        # 66ec <_sk_callback_avx+0x4a6>
+  .byte  196,226,125,24,5,81,7,0,0           // vbroadcastss  0x751(%rip),%ymm0        # 6708 <_sk_callback_avx+0x4a6>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,72,7,0,0            // vbroadcastss  0x748(%rip),%ymm8        # 66f0 <_sk_callback_avx+0x4aa>
+  .byte  196,98,125,24,5,72,7,0,0            // vbroadcastss  0x748(%rip),%ymm8        # 670c <_sk_callback_avx+0x4aa>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,57,7,0,0           // vbroadcastss  0x739(%rip),%ymm10        # 66f4 <_sk_callback_avx+0x4ae>
+  .byte  196,98,125,24,21,57,7,0,0           // vbroadcastss  0x739(%rip),%ymm10        # 6710 <_sk_callback_avx+0x4ae>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,47,7,0,0           // vbroadcastss  0x72f(%rip),%ymm10        # 66f8 <_sk_callback_avx+0x4b2>
+  .byte  196,98,125,24,21,47,7,0,0           // vbroadcastss  0x72f(%rip),%ymm10        # 6714 <_sk_callback_avx+0x4b2>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -19630,19 +19663,19 @@ HIDDEN _sk_bicubic_n1x_avx
 FUNCTION(_sk_bicubic_n1x_avx)
 _sk_bicubic_n1x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,18,7,0,0           // vbroadcastss  0x712(%rip),%ymm0        # 66fc <_sk_callback_avx+0x4b6>
+  .byte  196,226,125,24,5,18,7,0,0           // vbroadcastss  0x712(%rip),%ymm0        # 6718 <_sk_callback_avx+0x4b6>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
-  .byte  196,98,125,24,5,9,7,0,0             // vbroadcastss  0x709(%rip),%ymm8        # 6700 <_sk_callback_avx+0x4ba>
+  .byte  196,98,125,24,5,9,7,0,0             // vbroadcastss  0x709(%rip),%ymm8        # 671c <_sk_callback_avx+0x4ba>
   .byte  197,60,92,64,64                     // vsubps        0x40(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,255,6,0,0          // vbroadcastss  0x6ff(%rip),%ymm9        # 6704 <_sk_callback_avx+0x4be>
+  .byte  196,98,125,24,13,255,6,0,0          // vbroadcastss  0x6ff(%rip),%ymm9        # 6720 <_sk_callback_avx+0x4be>
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,245,6,0,0          // vbroadcastss  0x6f5(%rip),%ymm10        # 6708 <_sk_callback_avx+0x4c2>
+  .byte  196,98,125,24,21,245,6,0,0          // vbroadcastss  0x6f5(%rip),%ymm10        # 6724 <_sk_callback_avx+0x4c2>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,230,6,0,0          // vbroadcastss  0x6e6(%rip),%ymm10        # 670c <_sk_callback_avx+0x4c6>
+  .byte  196,98,125,24,21,230,6,0,0          // vbroadcastss  0x6e6(%rip),%ymm10        # 6728 <_sk_callback_avx+0x4c6>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,215,6,0,0          // vbroadcastss  0x6d7(%rip),%ymm9        # 6710 <_sk_callback_avx+0x4ca>
+  .byte  196,98,125,24,13,215,6,0,0          // vbroadcastss  0x6d7(%rip),%ymm9        # 672c <_sk_callback_avx+0x4ca>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19653,17 +19686,17 @@ HIDDEN _sk_bicubic_p1x_avx
 FUNCTION(_sk_bicubic_p1x_avx)
 _sk_bicubic_p1x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,191,6,0,0           // vbroadcastss  0x6bf(%rip),%ymm8        # 6714 <_sk_callback_avx+0x4ce>
+  .byte  196,98,125,24,5,191,6,0,0           // vbroadcastss  0x6bf(%rip),%ymm8        # 6730 <_sk_callback_avx+0x4ce>
   .byte  197,188,88,0                        // vaddps        (%rax),%ymm8,%ymm0
   .byte  197,124,16,72,64                    // vmovups       0x40(%rax),%ymm9
-  .byte  196,98,125,24,21,177,6,0,0          // vbroadcastss  0x6b1(%rip),%ymm10        # 6718 <_sk_callback_avx+0x4d2>
+  .byte  196,98,125,24,21,177,6,0,0          // vbroadcastss  0x6b1(%rip),%ymm10        # 6734 <_sk_callback_avx+0x4d2>
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
-  .byte  196,98,125,24,29,167,6,0,0          // vbroadcastss  0x6a7(%rip),%ymm11        # 671c <_sk_callback_avx+0x4d6>
+  .byte  196,98,125,24,29,167,6,0,0          // vbroadcastss  0x6a7(%rip),%ymm11        # 6738 <_sk_callback_avx+0x4d6>
   .byte  196,65,44,88,211                    // vaddps        %ymm11,%ymm10,%ymm10
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
   .byte  196,65,44,88,192                    // vaddps        %ymm8,%ymm10,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
-  .byte  196,98,125,24,13,142,6,0,0          // vbroadcastss  0x68e(%rip),%ymm9        # 6720 <_sk_callback_avx+0x4da>
+  .byte  196,98,125,24,13,142,6,0,0          // vbroadcastss  0x68e(%rip),%ymm9        # 673c <_sk_callback_avx+0x4da>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19674,13 +19707,13 @@ HIDDEN _sk_bicubic_p3x_avx
 FUNCTION(_sk_bicubic_p3x_avx)
 _sk_bicubic_p3x_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,5,118,6,0,0          // vbroadcastss  0x676(%rip),%ymm0        # 6724 <_sk_callback_avx+0x4de>
+  .byte  196,226,125,24,5,118,6,0,0          // vbroadcastss  0x676(%rip),%ymm0        # 6740 <_sk_callback_avx+0x4de>
   .byte  197,252,88,0                        // vaddps        (%rax),%ymm0,%ymm0
   .byte  197,124,16,64,64                    // vmovups       0x40(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,99,6,0,0           // vbroadcastss  0x663(%rip),%ymm10        # 6728 <_sk_callback_avx+0x4e2>
+  .byte  196,98,125,24,21,99,6,0,0           // vbroadcastss  0x663(%rip),%ymm10        # 6744 <_sk_callback_avx+0x4e2>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,89,6,0,0           // vbroadcastss  0x659(%rip),%ymm10        # 672c <_sk_callback_avx+0x4e6>
+  .byte  196,98,125,24,21,89,6,0,0           // vbroadcastss  0x659(%rip),%ymm10        # 6748 <_sk_callback_avx+0x4e6>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,128,0,0,0            // vmovups       %ymm8,0x80(%rax)
@@ -19692,14 +19725,14 @@ HIDDEN _sk_bicubic_n3y_avx
 FUNCTION(_sk_bicubic_n3y_avx)
 _sk_bicubic_n3y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,60,6,0,0          // vbroadcastss  0x63c(%rip),%ymm1        # 6730 <_sk_callback_avx+0x4ea>
+  .byte  196,226,125,24,13,60,6,0,0          // vbroadcastss  0x63c(%rip),%ymm1        # 674c <_sk_callback_avx+0x4ea>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,50,6,0,0            // vbroadcastss  0x632(%rip),%ymm8        # 6734 <_sk_callback_avx+0x4ee>
+  .byte  196,98,125,24,5,50,6,0,0            // vbroadcastss  0x632(%rip),%ymm8        # 6750 <_sk_callback_avx+0x4ee>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,35,6,0,0           // vbroadcastss  0x623(%rip),%ymm10        # 6738 <_sk_callback_avx+0x4f2>
+  .byte  196,98,125,24,21,35,6,0,0           // vbroadcastss  0x623(%rip),%ymm10        # 6754 <_sk_callback_avx+0x4f2>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,25,6,0,0           // vbroadcastss  0x619(%rip),%ymm10        # 673c <_sk_callback_avx+0x4f6>
+  .byte  196,98,125,24,21,25,6,0,0           // vbroadcastss  0x619(%rip),%ymm10        # 6758 <_sk_callback_avx+0x4f6>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -19711,19 +19744,19 @@ HIDDEN _sk_bicubic_n1y_avx
 FUNCTION(_sk_bicubic_n1y_avx)
 _sk_bicubic_n1y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,252,5,0,0         // vbroadcastss  0x5fc(%rip),%ymm1        # 6740 <_sk_callback_avx+0x4fa>
+  .byte  196,226,125,24,13,252,5,0,0         // vbroadcastss  0x5fc(%rip),%ymm1        # 675c <_sk_callback_avx+0x4fa>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
-  .byte  196,98,125,24,5,242,5,0,0           // vbroadcastss  0x5f2(%rip),%ymm8        # 6744 <_sk_callback_avx+0x4fe>
+  .byte  196,98,125,24,5,242,5,0,0           // vbroadcastss  0x5f2(%rip),%ymm8        # 6760 <_sk_callback_avx+0x4fe>
   .byte  197,60,92,64,96                     // vsubps        0x60(%rax),%ymm8,%ymm8
-  .byte  196,98,125,24,13,232,5,0,0          // vbroadcastss  0x5e8(%rip),%ymm9        # 6748 <_sk_callback_avx+0x502>
+  .byte  196,98,125,24,13,232,5,0,0          // vbroadcastss  0x5e8(%rip),%ymm9        # 6764 <_sk_callback_avx+0x502>
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,222,5,0,0          // vbroadcastss  0x5de(%rip),%ymm10        # 674c <_sk_callback_avx+0x506>
+  .byte  196,98,125,24,21,222,5,0,0          // vbroadcastss  0x5de(%rip),%ymm10        # 6768 <_sk_callback_avx+0x506>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,201                    // vmulps        %ymm9,%ymm8,%ymm9
-  .byte  196,98,125,24,21,207,5,0,0          // vbroadcastss  0x5cf(%rip),%ymm10        # 6750 <_sk_callback_avx+0x50a>
+  .byte  196,98,125,24,21,207,5,0,0          // vbroadcastss  0x5cf(%rip),%ymm10        # 676c <_sk_callback_avx+0x50a>
   .byte  196,65,52,88,202                    // vaddps        %ymm10,%ymm9,%ymm9
   .byte  196,65,60,89,193                    // vmulps        %ymm9,%ymm8,%ymm8
-  .byte  196,98,125,24,13,192,5,0,0          // vbroadcastss  0x5c0(%rip),%ymm9        # 6754 <_sk_callback_avx+0x50e>
+  .byte  196,98,125,24,13,192,5,0,0          // vbroadcastss  0x5c0(%rip),%ymm9        # 6770 <_sk_callback_avx+0x50e>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19734,17 +19767,17 @@ HIDDEN _sk_bicubic_p1y_avx
 FUNCTION(_sk_bicubic_p1y_avx)
 _sk_bicubic_p1y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,98,125,24,5,168,5,0,0           // vbroadcastss  0x5a8(%rip),%ymm8        # 6758 <_sk_callback_avx+0x512>
+  .byte  196,98,125,24,5,168,5,0,0           // vbroadcastss  0x5a8(%rip),%ymm8        # 6774 <_sk_callback_avx+0x512>
   .byte  197,188,88,72,32                    // vaddps        0x20(%rax),%ymm8,%ymm1
   .byte  197,124,16,72,96                    // vmovups       0x60(%rax),%ymm9
-  .byte  196,98,125,24,21,153,5,0,0          // vbroadcastss  0x599(%rip),%ymm10        # 675c <_sk_callback_avx+0x516>
+  .byte  196,98,125,24,21,153,5,0,0          // vbroadcastss  0x599(%rip),%ymm10        # 6778 <_sk_callback_avx+0x516>
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
-  .byte  196,98,125,24,29,143,5,0,0          // vbroadcastss  0x58f(%rip),%ymm11        # 6760 <_sk_callback_avx+0x51a>
+  .byte  196,98,125,24,29,143,5,0,0          // vbroadcastss  0x58f(%rip),%ymm11        # 677c <_sk_callback_avx+0x51a>
   .byte  196,65,44,88,211                    // vaddps        %ymm11,%ymm10,%ymm10
   .byte  196,65,52,89,210                    // vmulps        %ymm10,%ymm9,%ymm10
   .byte  196,65,44,88,192                    // vaddps        %ymm8,%ymm10,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
-  .byte  196,98,125,24,13,118,5,0,0          // vbroadcastss  0x576(%rip),%ymm9        # 6764 <_sk_callback_avx+0x51e>
+  .byte  196,98,125,24,13,118,5,0,0          // vbroadcastss  0x576(%rip),%ymm9        # 6780 <_sk_callback_avx+0x51e>
   .byte  196,65,60,88,193                    // vaddps        %ymm9,%ymm8,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -19755,13 +19788,13 @@ HIDDEN _sk_bicubic_p3y_avx
 FUNCTION(_sk_bicubic_p3y_avx)
 _sk_bicubic_p3y_avx:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  196,226,125,24,13,94,5,0,0          // vbroadcastss  0x55e(%rip),%ymm1        # 6768 <_sk_callback_avx+0x522>
+  .byte  196,226,125,24,13,94,5,0,0          // vbroadcastss  0x55e(%rip),%ymm1        # 6784 <_sk_callback_avx+0x522>
   .byte  197,244,88,72,32                    // vaddps        0x20(%rax),%ymm1,%ymm1
   .byte  197,124,16,64,96                    // vmovups       0x60(%rax),%ymm8
   .byte  196,65,60,89,200                    // vmulps        %ymm8,%ymm8,%ymm9
-  .byte  196,98,125,24,21,74,5,0,0           // vbroadcastss  0x54a(%rip),%ymm10        # 676c <_sk_callback_avx+0x526>
+  .byte  196,98,125,24,21,74,5,0,0           // vbroadcastss  0x54a(%rip),%ymm10        # 6788 <_sk_callback_avx+0x526>
   .byte  196,65,60,89,194                    // vmulps        %ymm10,%ymm8,%ymm8
-  .byte  196,98,125,24,21,64,5,0,0           // vbroadcastss  0x540(%rip),%ymm10        # 6770 <_sk_callback_avx+0x52a>
+  .byte  196,98,125,24,21,64,5,0,0           // vbroadcastss  0x540(%rip),%ymm10        # 678c <_sk_callback_avx+0x52a>
   .byte  196,65,60,88,194                    // vaddps        %ymm10,%ymm8,%ymm8
   .byte  196,65,52,89,192                    // vmulps        %ymm8,%ymm9,%ymm8
   .byte  197,124,17,128,160,0,0,0            // vmovups       %ymm8,0xa0(%rax)
@@ -19885,25 +19918,25 @@ BALIGN4
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 641d <.literal4+0xb1>
+  .byte  71,225,61                           // rex.RXB       loope 6439 <.literal4+0xb1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 642d <.literal4+0xc1>
+  .byte  71,225,61                           // rex.RXB       loope 6449 <.literal4+0xc1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 643d <.literal4+0xd1>
+  .byte  71,225,61                           // rex.RXB       loope 6459 <.literal4+0xd1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,154                          // cmpb          $0x9a,(%rdi)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
   .byte  62,61,10,23,63,174                  // ds            cmp $0xae3f170a,%eax
-  .byte  71,225,61                           // rex.RXB       loope 644d <.literal4+0xe1>
+  .byte  71,225,61                           // rex.RXB       loope 6469 <.literal4+0xe1>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -19953,7 +19986,7 @@ BALIGN4
   .byte  190,129,128,128,59                  // mov           $0x3b808081,%esi
   .byte  129,128,128,59,0,248,0,0,8,33       // addl          $0x21080000,-0x7ffc480(%rax)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        6499 <.literal4+0x12d>
+  .byte  224,7                               // loopne        64b5 <.literal4+0x12d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -19969,10 +20002,10 @@ BALIGN4
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
   .byte  0,52,255                            // add           %dh,(%rdi,%rdi,8)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            64c0 <.literal4+0x154>
+  .byte  127,0                               // jg            64dc <.literal4+0x154>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            6539 <.literal4+0x1cd>
+  .byte  119,115                             // ja            6555 <.literal4+0x1cd>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -19986,10 +20019,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            64f4 <.literal4+0x188>
+  .byte  127,0                               // jg            6510 <.literal4+0x188>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            656d <.literal4+0x201>
+  .byte  119,115                             // ja            6589 <.literal4+0x201>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -20003,10 +20036,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            6528 <.literal4+0x1bc>
+  .byte  127,0                               // jg            6544 <.literal4+0x1bc>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            65a1 <.literal4+0x235>
+  .byte  119,115                             // ja            65bd <.literal4+0x235>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -20020,10 +20053,10 @@ BALIGN4
   .byte  0,128,63,0,0,0                      // add           %al,0x3f(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            655c <.literal4+0x1f0>
+  .byte  127,0                               // jg            6578 <.literal4+0x1f0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            65d5 <.literal4+0x269>
+  .byte  119,115                             // ja            65f1 <.literal4+0x269>
   .byte  248                                 // clc
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,249,68,180                   // mov           $0xb444f93f,%edi
@@ -20036,7 +20069,7 @@ BALIGN4
   .byte  0,75,0                              // add           %cl,0x0(%rbx)
   .byte  0,128,63,0,0,200                    // add           %al,-0x37ffffc1(%rax)
   .byte  66,0,0                              // rex.X         add %al,(%rax)
-  .byte  127,67                              // jg            65d3 <.literal4+0x267>
+  .byte  127,67                              // jg            65ef <.literal4+0x267>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -20048,10 +20081,10 @@ BALIGN4
   .byte  190,80,128,3,62                     // mov           $0x3e038050,%esi
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           65f3 <.literal4+0x287>
+  .byte  118,63                              // jbe           660f <.literal4+0x287>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            6607 <.literal4+0x29b>
+  .byte  127,67                              // jg            6623 <.literal4+0x29b>
   .byte  129,128,128,59,0,0,128,63,129,128   // addl          $0x80813f80,0x3b80(%rax)
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,128,63,129,128,128                // add           %al,-0x7f7f7ec1(%rax)
@@ -20060,7 +20093,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        65e9 <.literal4+0x27d>
+  .byte  224,7                               // loopne        6605 <.literal4+0x27d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -20072,7 +20105,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        6605 <.literal4+0x299>
+  .byte  224,7                               // loopne        6621 <.literal4+0x299>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -20083,7 +20116,7 @@ BALIGN4
   .byte  0,0                                 // add           %al,(%rax)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            665a <.literal4+0x2ee>
+  .byte  124,66                              // jl            6676 <.literal4+0x2ee>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,55,0,15                 // mov           %ecx,0xf003788(%rax)
@@ -20101,9 +20134,9 @@ BALIGN4
   .byte  137,136,136,59,15,0                 // mov           %ecx,0xf3b88(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  137,136,136,61,0,0                  // mov           %ecx,0x3d88(%rax)
-  .byte  112,65                              // jo            669d <.literal4+0x331>
+  .byte  112,65                              // jo            66b9 <.literal4+0x331>
   .byte  129,128,128,59,129,128,128,59,0,0   // addl          $0x3b80,-0x7f7ec480(%rax)
-  .byte  127,67                              // jg            66ab <.literal4+0x33f>
+  .byte  127,67                              // jg            66c7 <.literal4+0x33f>
   .byte  0,128,0,0,0,0                       // add           %al,0x0(%rax)
   .byte  0,128,0,4,0,128                     // add           %al,-0x7ffffc00(%rax)
   .byte  0,0                                 // add           %al,(%rax)
@@ -20119,7 +20152,7 @@ BALIGN4
   .byte  0,128,55,0,0,128                    // add           %al,-0x7fffffc9(%rax)
   .byte  63                                  // (bad)
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            66eb <.literal4+0x37f>
+  .byte  127,71                              // jg            6707 <.literal4+0x37f>
   .byte  208                                 // (bad)
   .byte  179,89                              // mov           $0x59,%bl
   .byte  62,89                               // ds            pop %rcx
@@ -20349,7 +20382,7 @@ _sk_seed_shader_sse41:
   .byte  102,15,110,199                      // movd          %edi,%xmm0
   .byte  102,15,112,192,0                    // pshufd        $0x0,%xmm0,%xmm0
   .byte  15,91,200                           // cvtdq2ps      %xmm0,%xmm1
-  .byte  15,40,21,116,70,0,0                 // movaps        0x4674(%rip),%xmm2        # 46f0 <_sk_callback_sse41+0xe3>
+  .byte  15,40,21,148,70,0,0                 // movaps        0x4694(%rip),%xmm2        # 4710 <_sk_callback_sse41+0xd9>
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  15,16,2                             // movups        (%rdx),%xmm0
   .byte  15,88,193                           // addps         %xmm1,%xmm0
@@ -20358,7 +20391,7 @@ _sk_seed_shader_sse41:
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,21,99,70,0,0                  // movaps        0x4663(%rip),%xmm2        # 4700 <_sk_callback_sse41+0xf3>
+  .byte  15,40,21,131,70,0,0                 // movaps        0x4683(%rip),%xmm2        # 4720 <_sk_callback_sse41+0xe9>
   .byte  15,87,219                           // xorps         %xmm3,%xmm3
   .byte  15,87,228                           // xorps         %xmm4,%xmm4
   .byte  15,87,237                           // xorps         %xmm5,%xmm5
@@ -20381,14 +20414,14 @@ _sk_dither_sse41:
   .byte  102,68,15,110,1                     // movd          (%rcx),%xmm8
   .byte  102,69,15,112,192,0                 // pshufd        $0x0,%xmm8,%xmm8
   .byte  102,69,15,239,193                   // pxor          %xmm9,%xmm8
-  .byte  102,68,15,111,21,40,70,0,0          // movdqa        0x4628(%rip),%xmm10        # 4710 <_sk_callback_sse41+0x103>
+  .byte  102,68,15,111,21,72,70,0,0          // movdqa        0x4648(%rip),%xmm10        # 4730 <_sk_callback_sse41+0xf9>
   .byte  102,69,15,111,216                   // movdqa        %xmm8,%xmm11
   .byte  102,69,15,219,218                   // pand          %xmm10,%xmm11
   .byte  102,65,15,114,243,5                 // pslld         $0x5,%xmm11
   .byte  102,69,15,219,209                   // pand          %xmm9,%xmm10
   .byte  102,65,15,114,242,4                 // pslld         $0x4,%xmm10
-  .byte  102,68,15,111,37,20,70,0,0          // movdqa        0x4614(%rip),%xmm12        # 4720 <_sk_callback_sse41+0x113>
-  .byte  102,68,15,111,45,27,70,0,0          // movdqa        0x461b(%rip),%xmm13        # 4730 <_sk_callback_sse41+0x123>
+  .byte  102,68,15,111,37,52,70,0,0          // movdqa        0x4634(%rip),%xmm12        # 4740 <_sk_callback_sse41+0x109>
+  .byte  102,68,15,111,45,59,70,0,0          // movdqa        0x463b(%rip),%xmm13        # 4750 <_sk_callback_sse41+0x119>
   .byte  102,69,15,111,240                   // movdqa        %xmm8,%xmm14
   .byte  102,69,15,219,245                   // pand          %xmm13,%xmm14
   .byte  102,65,15,114,246,2                 // pslld         $0x2,%xmm14
@@ -20404,15 +20437,26 @@ _sk_dither_sse41:
   .byte  102,69,15,235,245                   // por           %xmm13,%xmm14
   .byte  102,69,15,235,240                   // por           %xmm8,%xmm14
   .byte  69,15,91,198                        // cvtdq2ps      %xmm14,%xmm8
-  .byte  68,15,89,5,214,69,0,0               // mulps         0x45d6(%rip),%xmm8        # 4740 <_sk_callback_sse41+0x133>
-  .byte  68,15,88,5,222,69,0,0               // addps         0x45de(%rip),%xmm8        # 4750 <_sk_callback_sse41+0x143>
-  .byte  243,68,15,16,72,8                   // movss         0x8(%rax),%xmm9
-  .byte  69,15,198,201,0                     // shufps        $0x0,%xmm9,%xmm9
-  .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
-  .byte  65,15,88,193                        // addps         %xmm9,%xmm0
-  .byte  65,15,88,201                        // addps         %xmm9,%xmm1
-  .byte  65,15,88,209                        // addps         %xmm9,%xmm2
+  .byte  68,15,89,5,246,69,0,0               // mulps         0x45f6(%rip),%xmm8        # 4760 <_sk_callback_sse41+0x129>
+  .byte  68,15,88,5,254,69,0,0               // addps         0x45fe(%rip),%xmm8        # 4770 <_sk_callback_sse41+0x139>
+  .byte  243,68,15,16,80,8                   // movss         0x8(%rax),%xmm10
+  .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
+  .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
+  .byte  65,15,88,194                        // addps         %xmm10,%xmm0
+  .byte  65,15,88,202                        // addps         %xmm10,%xmm1
+  .byte  68,15,88,210                        // addps         %xmm2,%xmm10
+  .byte  15,93,195                           // minps         %xmm3,%xmm0
+  .byte  15,87,210                           // xorps         %xmm2,%xmm2
+  .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
+  .byte  68,15,95,192                        // maxps         %xmm0,%xmm8
+  .byte  15,93,203                           // minps         %xmm3,%xmm1
+  .byte  102,69,15,239,201                   // pxor          %xmm9,%xmm9
+  .byte  68,15,95,201                        // maxps         %xmm1,%xmm9
+  .byte  68,15,93,211                        // minps         %xmm3,%xmm10
+  .byte  65,15,95,210                        // maxps         %xmm10,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
+  .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
+  .byte  65,15,40,201                        // movaps        %xmm9,%xmm1
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_constant_color_sse41
@@ -20471,7 +20515,7 @@ HIDDEN _sk_srcatop_sse41
 FUNCTION(_sk_srcatop_sse41)
 _sk_srcatop_sse41:
   .byte  15,89,199                           // mulps         %xmm7,%xmm0
-  .byte  68,15,40,5,97,69,0,0                // movaps        0x4561(%rip),%xmm8        # 4760 <_sk_callback_sse41+0x153>
+  .byte  68,15,40,5,87,69,0,0                // movaps        0x4557(%rip),%xmm8        # 4780 <_sk_callback_sse41+0x149>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -20496,7 +20540,7 @@ FUNCTION(_sk_dstatop_sse41)
 _sk_dstatop_sse41:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
   .byte  68,15,89,196                        // mulps         %xmm4,%xmm8
-  .byte  68,15,40,13,36,69,0,0               // movaps        0x4524(%rip),%xmm9        # 4770 <_sk_callback_sse41+0x163>
+  .byte  68,15,40,13,26,69,0,0               // movaps        0x451a(%rip),%xmm9        # 4790 <_sk_callback_sse41+0x159>
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
@@ -20543,7 +20587,7 @@ HIDDEN _sk_srcout_sse41
 .globl _sk_srcout_sse41
 FUNCTION(_sk_srcout_sse41)
 _sk_srcout_sse41:
-  .byte  68,15,40,5,200,68,0,0               // movaps        0x44c8(%rip),%xmm8        # 4780 <_sk_callback_sse41+0x173>
+  .byte  68,15,40,5,190,68,0,0               // movaps        0x44be(%rip),%xmm8        # 47a0 <_sk_callback_sse41+0x169>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
@@ -20556,7 +20600,7 @@ HIDDEN _sk_dstout_sse41
 .globl _sk_dstout_sse41
 FUNCTION(_sk_dstout_sse41)
 _sk_dstout_sse41:
-  .byte  68,15,40,5,184,68,0,0               // movaps        0x44b8(%rip),%xmm8        # 4790 <_sk_callback_sse41+0x183>
+  .byte  68,15,40,5,174,68,0,0               // movaps        0x44ae(%rip),%xmm8        # 47b0 <_sk_callback_sse41+0x179>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  15,89,196                           // mulps         %xmm4,%xmm0
@@ -20573,7 +20617,7 @@ HIDDEN _sk_srcover_sse41
 .globl _sk_srcover_sse41
 FUNCTION(_sk_srcover_sse41)
 _sk_srcover_sse41:
-  .byte  68,15,40,5,155,68,0,0               // movaps        0x449b(%rip),%xmm8        # 47a0 <_sk_callback_sse41+0x193>
+  .byte  68,15,40,5,145,68,0,0               // movaps        0x4491(%rip),%xmm8        # 47c0 <_sk_callback_sse41+0x189>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -20593,7 +20637,7 @@ HIDDEN _sk_dstover_sse41
 .globl _sk_dstover_sse41
 FUNCTION(_sk_dstover_sse41)
 _sk_dstover_sse41:
-  .byte  68,15,40,5,111,68,0,0               // movaps        0x446f(%rip),%xmm8        # 47b0 <_sk_callback_sse41+0x1a3>
+  .byte  68,15,40,5,101,68,0,0               // movaps        0x4465(%rip),%xmm8        # 47d0 <_sk_callback_sse41+0x199>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -20621,7 +20665,7 @@ HIDDEN _sk_multiply_sse41
 .globl _sk_multiply_sse41
 FUNCTION(_sk_multiply_sse41)
 _sk_multiply_sse41:
-  .byte  68,15,40,5,67,68,0,0                // movaps        0x4443(%rip),%xmm8        # 47c0 <_sk_callback_sse41+0x1b3>
+  .byte  68,15,40,5,57,68,0,0                // movaps        0x4439(%rip),%xmm8        # 47e0 <_sk_callback_sse41+0x1a9>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
@@ -20697,7 +20741,7 @@ HIDDEN _sk_xor__sse41
 FUNCTION(_sk_xor__sse41)
 _sk_xor__sse41:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
-  .byte  15,40,29,116,67,0,0                 // movaps        0x4374(%rip),%xmm3        # 47d0 <_sk_callback_sse41+0x1c3>
+  .byte  15,40,29,106,67,0,0                 // movaps        0x436a(%rip),%xmm3        # 47f0 <_sk_callback_sse41+0x1b9>
   .byte  68,15,40,203                        // movaps        %xmm3,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
@@ -20745,7 +20789,7 @@ _sk_darken_sse41:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,95,209                        // maxps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,223,66,0,0                 // movaps        0x42df(%rip),%xmm2        # 47e0 <_sk_callback_sse41+0x1d3>
+  .byte  15,40,21,213,66,0,0                 // movaps        0x42d5(%rip),%xmm2        # 4800 <_sk_callback_sse41+0x1c9>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -20779,7 +20823,7 @@ _sk_lighten_sse41:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,132,66,0,0                 // movaps        0x4284(%rip),%xmm2        # 47f0 <_sk_callback_sse41+0x1e3>
+  .byte  15,40,21,122,66,0,0                 // movaps        0x427a(%rip),%xmm2        # 4810 <_sk_callback_sse41+0x1d9>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -20816,7 +20860,7 @@ _sk_difference_sse41:
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,30,66,0,0                  // movaps        0x421e(%rip),%xmm2        # 4800 <_sk_callback_sse41+0x1f3>
+  .byte  15,40,21,20,66,0,0                  // movaps        0x4214(%rip),%xmm2        # 4820 <_sk_callback_sse41+0x1e9>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -20843,7 +20887,7 @@ _sk_exclusion_sse41:
   .byte  15,89,214                           // mulps         %xmm6,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,202                        // subps         %xmm2,%xmm9
-  .byte  15,40,13,223,65,0,0                 // movaps        0x41df(%rip),%xmm1        # 4810 <_sk_callback_sse41+0x203>
+  .byte  15,40,13,213,65,0,0                 // movaps        0x41d5(%rip),%xmm1        # 4830 <_sk_callback_sse41+0x1f9>
   .byte  15,92,203                           // subps         %xmm3,%xmm1
   .byte  15,89,207                           // mulps         %xmm7,%xmm1
   .byte  15,88,217                           // addps         %xmm1,%xmm3
@@ -20857,7 +20901,7 @@ HIDDEN _sk_colorburn_sse41
 FUNCTION(_sk_colorburn_sse41)
 _sk_colorburn_sse41:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,40,21,206,65,0,0              // movaps        0x41ce(%rip),%xmm10        # 4820 <_sk_callback_sse41+0x213>
+  .byte  68,15,40,21,196,65,0,0              // movaps        0x41c4(%rip),%xmm10        # 4840 <_sk_callback_sse41+0x209>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  69,15,40,203                        // movaps        %xmm11,%xmm9
@@ -20939,7 +20983,7 @@ HIDDEN _sk_colordodge_sse41
 FUNCTION(_sk_colordodge_sse41)
 _sk_colordodge_sse41:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,40,21,172,64,0,0              // movaps        0x40ac(%rip),%xmm10        # 4830 <_sk_callback_sse41+0x223>
+  .byte  68,15,40,21,162,64,0,0              // movaps        0x40a2(%rip),%xmm10        # 4850 <_sk_callback_sse41+0x219>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  69,15,40,227                        // movaps        %xmm11,%xmm12
@@ -21021,7 +21065,7 @@ _sk_hardlight_sse41:
   .byte  15,40,244                           // movaps        %xmm4,%xmm6
   .byte  15,40,227                           // movaps        %xmm3,%xmm4
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
-  .byte  68,15,40,21,133,63,0,0              // movaps        0x3f85(%rip),%xmm10        # 4840 <_sk_callback_sse41+0x233>
+  .byte  68,15,40,21,123,63,0,0              // movaps        0x3f7b(%rip),%xmm10        # 4860 <_sk_callback_sse41+0x229>
   .byte  65,15,40,234                        // movaps        %xmm10,%xmm5
   .byte  15,92,239                           // subps         %xmm7,%xmm5
   .byte  15,40,197                           // movaps        %xmm5,%xmm0
@@ -21104,7 +21148,7 @@ FUNCTION(_sk_overlay_sse41)
 _sk_overlay_sse41:
   .byte  68,15,40,201                        // movaps        %xmm1,%xmm9
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
-  .byte  68,15,40,21,106,62,0,0              // movaps        0x3e6a(%rip),%xmm10        # 4850 <_sk_callback_sse41+0x243>
+  .byte  68,15,40,21,96,62,0,0               // movaps        0x3e60(%rip),%xmm10        # 4870 <_sk_callback_sse41+0x239>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
@@ -21189,7 +21233,7 @@ _sk_softlight_sse41:
   .byte  15,40,198                           // movaps        %xmm6,%xmm0
   .byte  15,94,199                           // divps         %xmm7,%xmm0
   .byte  65,15,84,193                        // andps         %xmm9,%xmm0
-  .byte  15,40,13,65,61,0,0                  // movaps        0x3d41(%rip),%xmm1        # 4860 <_sk_callback_sse41+0x253>
+  .byte  15,40,13,55,61,0,0                  // movaps        0x3d37(%rip),%xmm1        # 4880 <_sk_callback_sse41+0x249>
   .byte  68,15,40,209                        // movaps        %xmm1,%xmm10
   .byte  68,15,92,208                        // subps         %xmm0,%xmm10
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
@@ -21202,10 +21246,10 @@ _sk_softlight_sse41:
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  15,89,210                           // mulps         %xmm2,%xmm2
   .byte  15,88,208                           // addps         %xmm0,%xmm2
-  .byte  68,15,40,45,31,61,0,0               // movaps        0x3d1f(%rip),%xmm13        # 4870 <_sk_callback_sse41+0x263>
+  .byte  68,15,40,45,21,61,0,0               // movaps        0x3d15(%rip),%xmm13        # 4890 <_sk_callback_sse41+0x259>
   .byte  69,15,88,245                        // addps         %xmm13,%xmm14
   .byte  68,15,89,242                        // mulps         %xmm2,%xmm14
-  .byte  68,15,40,37,31,61,0,0               // movaps        0x3d1f(%rip),%xmm12        # 4880 <_sk_callback_sse41+0x273>
+  .byte  68,15,40,37,21,61,0,0               // movaps        0x3d15(%rip),%xmm12        # 48a0 <_sk_callback_sse41+0x269>
   .byte  69,15,89,252                        // mulps         %xmm12,%xmm15
   .byte  69,15,88,254                        // addps         %xmm14,%xmm15
   .byte  15,40,198                           // movaps        %xmm6,%xmm0
@@ -21391,12 +21435,12 @@ _sk_hue_sse41:
   .byte  68,15,84,208                        // andps         %xmm0,%xmm10
   .byte  15,84,200                           // andps         %xmm0,%xmm1
   .byte  68,15,84,232                        // andps         %xmm0,%xmm13
-  .byte  15,40,5,138,58,0,0                  // movaps        0x3a8a(%rip),%xmm0        # 4890 <_sk_callback_sse41+0x283>
+  .byte  15,40,5,128,58,0,0                  // movaps        0x3a80(%rip),%xmm0        # 48b0 <_sk_callback_sse41+0x279>
   .byte  68,15,89,224                        // mulps         %xmm0,%xmm12
-  .byte  15,40,21,143,58,0,0                 // movaps        0x3a8f(%rip),%xmm2        # 48a0 <_sk_callback_sse41+0x293>
+  .byte  15,40,21,133,58,0,0                 // movaps        0x3a85(%rip),%xmm2        # 48c0 <_sk_callback_sse41+0x289>
   .byte  15,89,250                           // mulps         %xmm2,%xmm7
   .byte  65,15,88,252                        // addps         %xmm12,%xmm7
-  .byte  68,15,40,53,144,58,0,0              // movaps        0x3a90(%rip),%xmm14        # 48b0 <_sk_callback_sse41+0x2a3>
+  .byte  68,15,40,53,134,58,0,0              // movaps        0x3a86(%rip),%xmm14        # 48d0 <_sk_callback_sse41+0x299>
   .byte  68,15,40,252                        // movaps        %xmm4,%xmm15
   .byte  69,15,89,254                        // mulps         %xmm14,%xmm15
   .byte  68,15,88,255                        // addps         %xmm7,%xmm15
@@ -21479,7 +21523,7 @@ _sk_hue_sse41:
   .byte  65,15,88,214                        // addps         %xmm14,%xmm2
   .byte  15,40,196                           // movaps        %xmm4,%xmm0
   .byte  102,15,56,20,202                    // blendvps      %xmm0,%xmm2,%xmm1
-  .byte  68,15,40,13,84,57,0,0               // movaps        0x3954(%rip),%xmm9        # 48c0 <_sk_callback_sse41+0x2b3>
+  .byte  68,15,40,13,74,57,0,0               // movaps        0x394a(%rip),%xmm9        # 48e0 <_sk_callback_sse41+0x2a9>
   .byte  65,15,40,225                        // movaps        %xmm9,%xmm4
   .byte  15,92,229                           // subps         %xmm5,%xmm4
   .byte  15,40,68,36,200                     // movaps        -0x38(%rsp),%xmm0
@@ -21573,14 +21617,14 @@ _sk_saturation_sse41:
   .byte  68,15,84,215                        // andps         %xmm7,%xmm10
   .byte  68,15,84,223                        // andps         %xmm7,%xmm11
   .byte  68,15,84,199                        // andps         %xmm7,%xmm8
-  .byte  15,40,21,14,56,0,0                  // movaps        0x380e(%rip),%xmm2        # 48d0 <_sk_callback_sse41+0x2c3>
+  .byte  15,40,21,4,56,0,0                   // movaps        0x3804(%rip),%xmm2        # 48f0 <_sk_callback_sse41+0x2b9>
   .byte  15,40,221                           // movaps        %xmm5,%xmm3
   .byte  15,89,218                           // mulps         %xmm2,%xmm3
-  .byte  15,40,13,17,56,0,0                  // movaps        0x3811(%rip),%xmm1        # 48e0 <_sk_callback_sse41+0x2d3>
+  .byte  15,40,13,7,56,0,0                   // movaps        0x3807(%rip),%xmm1        # 4900 <_sk_callback_sse41+0x2c9>
   .byte  15,40,254                           // movaps        %xmm6,%xmm7
   .byte  15,89,249                           // mulps         %xmm1,%xmm7
   .byte  15,88,251                           // addps         %xmm3,%xmm7
-  .byte  68,15,40,45,16,56,0,0               // movaps        0x3810(%rip),%xmm13        # 48f0 <_sk_callback_sse41+0x2e3>
+  .byte  68,15,40,45,6,56,0,0                // movaps        0x3806(%rip),%xmm13        # 4910 <_sk_callback_sse41+0x2d9>
   .byte  69,15,89,245                        // mulps         %xmm13,%xmm14
   .byte  68,15,88,247                        // addps         %xmm7,%xmm14
   .byte  65,15,40,218                        // movaps        %xmm10,%xmm3
@@ -21661,7 +21705,7 @@ _sk_saturation_sse41:
   .byte  65,15,88,253                        // addps         %xmm13,%xmm7
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  102,68,15,56,20,223                 // blendvps      %xmm0,%xmm7,%xmm11
-  .byte  68,15,40,13,214,54,0,0              // movaps        0x36d6(%rip),%xmm9        # 4900 <_sk_callback_sse41+0x2f3>
+  .byte  68,15,40,13,204,54,0,0              // movaps        0x36cc(%rip),%xmm9        # 4920 <_sk_callback_sse41+0x2e9>
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  68,15,92,204                        // subps         %xmm4,%xmm9
   .byte  15,40,124,36,168                    // movaps        -0x58(%rsp),%xmm7
@@ -21716,14 +21760,14 @@ _sk_color_sse41:
   .byte  15,40,231                           // movaps        %xmm7,%xmm4
   .byte  68,15,89,244                        // mulps         %xmm4,%xmm14
   .byte  15,89,204                           // mulps         %xmm4,%xmm1
-  .byte  68,15,40,13,33,54,0,0               // movaps        0x3621(%rip),%xmm9        # 4910 <_sk_callback_sse41+0x303>
+  .byte  68,15,40,13,23,54,0,0               // movaps        0x3617(%rip),%xmm9        # 4930 <_sk_callback_sse41+0x2f9>
   .byte  65,15,40,250                        // movaps        %xmm10,%xmm7
   .byte  65,15,89,249                        // mulps         %xmm9,%xmm7
-  .byte  68,15,40,21,33,54,0,0               // movaps        0x3621(%rip),%xmm10        # 4920 <_sk_callback_sse41+0x313>
+  .byte  68,15,40,21,23,54,0,0               // movaps        0x3617(%rip),%xmm10        # 4940 <_sk_callback_sse41+0x309>
   .byte  65,15,40,219                        // movaps        %xmm11,%xmm3
   .byte  65,15,89,218                        // mulps         %xmm10,%xmm3
   .byte  15,88,223                           // addps         %xmm7,%xmm3
-  .byte  68,15,40,29,30,54,0,0               // movaps        0x361e(%rip),%xmm11        # 4930 <_sk_callback_sse41+0x323>
+  .byte  68,15,40,29,20,54,0,0               // movaps        0x3614(%rip),%xmm11        # 4950 <_sk_callback_sse41+0x319>
   .byte  69,15,40,236                        // movaps        %xmm12,%xmm13
   .byte  69,15,89,235                        // mulps         %xmm11,%xmm13
   .byte  68,15,88,235                        // addps         %xmm3,%xmm13
@@ -21808,7 +21852,7 @@ _sk_color_sse41:
   .byte  65,15,88,251                        // addps         %xmm11,%xmm7
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  102,15,56,20,207                    // blendvps      %xmm0,%xmm7,%xmm1
-  .byte  68,15,40,13,218,52,0,0              // movaps        0x34da(%rip),%xmm9        # 4940 <_sk_callback_sse41+0x333>
+  .byte  68,15,40,13,208,52,0,0              // movaps        0x34d0(%rip),%xmm9        # 4960 <_sk_callback_sse41+0x329>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  68,15,89,192                        // mulps         %xmm0,%xmm8
@@ -21860,13 +21904,13 @@ _sk_luminosity_sse41:
   .byte  69,15,89,216                        // mulps         %xmm8,%xmm11
   .byte  68,15,40,203                        // movaps        %xmm3,%xmm9
   .byte  68,15,89,205                        // mulps         %xmm5,%xmm9
-  .byte  68,15,40,5,50,52,0,0                // movaps        0x3432(%rip),%xmm8        # 4950 <_sk_callback_sse41+0x343>
+  .byte  68,15,40,5,40,52,0,0                // movaps        0x3428(%rip),%xmm8        # 4970 <_sk_callback_sse41+0x339>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
-  .byte  68,15,40,21,54,52,0,0               // movaps        0x3436(%rip),%xmm10        # 4960 <_sk_callback_sse41+0x353>
+  .byte  68,15,40,21,44,52,0,0               // movaps        0x342c(%rip),%xmm10        # 4980 <_sk_callback_sse41+0x349>
   .byte  15,40,233                           // movaps        %xmm1,%xmm5
   .byte  65,15,89,234                        // mulps         %xmm10,%xmm5
   .byte  15,88,232                           // addps         %xmm0,%xmm5
-  .byte  68,15,40,37,52,52,0,0               // movaps        0x3434(%rip),%xmm12        # 4970 <_sk_callback_sse41+0x363>
+  .byte  68,15,40,37,42,52,0,0               // movaps        0x342a(%rip),%xmm12        # 4990 <_sk_callback_sse41+0x359>
   .byte  68,15,40,242                        // movaps        %xmm2,%xmm14
   .byte  69,15,89,244                        // mulps         %xmm12,%xmm14
   .byte  68,15,88,245                        // addps         %xmm5,%xmm14
@@ -21951,7 +21995,7 @@ _sk_luminosity_sse41:
   .byte  65,15,88,244                        // addps         %xmm12,%xmm6
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  102,68,15,56,20,206                 // blendvps      %xmm0,%xmm6,%xmm9
-  .byte  15,40,5,234,50,0,0                  // movaps        0x32ea(%rip),%xmm0        # 4980 <_sk_callback_sse41+0x373>
+  .byte  15,40,5,224,50,0,0                  // movaps        0x32e0(%rip),%xmm0        # 49a0 <_sk_callback_sse41+0x369>
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  15,92,215                           // subps         %xmm7,%xmm2
   .byte  15,89,226                           // mulps         %xmm2,%xmm4
@@ -22000,7 +22044,7 @@ HIDDEN _sk_clamp_1_sse41
 .globl _sk_clamp_1_sse41
 FUNCTION(_sk_clamp_1_sse41)
 _sk_clamp_1_sse41:
-  .byte  68,15,40,5,109,50,0,0               // movaps        0x326d(%rip),%xmm8        # 4990 <_sk_callback_sse41+0x383>
+  .byte  68,15,40,5,99,50,0,0                // movaps        0x3263(%rip),%xmm8        # 49b0 <_sk_callback_sse41+0x379>
   .byte  65,15,93,192                        // minps         %xmm8,%xmm0
   .byte  65,15,93,200                        // minps         %xmm8,%xmm1
   .byte  65,15,93,208                        // minps         %xmm8,%xmm2
@@ -22012,7 +22056,7 @@ HIDDEN _sk_clamp_a_sse41
 .globl _sk_clamp_a_sse41
 FUNCTION(_sk_clamp_a_sse41)
 _sk_clamp_a_sse41:
-  .byte  15,93,29,98,50,0,0                  // minps         0x3262(%rip),%xmm3        # 49a0 <_sk_callback_sse41+0x393>
+  .byte  15,93,29,88,50,0,0                  // minps         0x3258(%rip),%xmm3        # 49c0 <_sk_callback_sse41+0x389>
   .byte  15,93,195                           // minps         %xmm3,%xmm0
   .byte  15,93,203                           // minps         %xmm3,%xmm1
   .byte  15,93,211                           // minps         %xmm3,%xmm2
@@ -22099,7 +22143,7 @@ HIDDEN _sk_unpremul_sse41
 FUNCTION(_sk_unpremul_sse41)
 _sk_unpremul_sse41:
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
-  .byte  68,15,40,13,205,49,0,0              // movaps        0x31cd(%rip),%xmm9        # 49b0 <_sk_callback_sse41+0x3a3>
+  .byte  68,15,40,13,195,49,0,0              // movaps        0x31c3(%rip),%xmm9        # 49d0 <_sk_callback_sse41+0x399>
   .byte  68,15,94,203                        // divps         %xmm3,%xmm9
   .byte  68,15,194,195,4                     // cmpneqps      %xmm3,%xmm8
   .byte  69,15,84,193                        // andps         %xmm9,%xmm8
@@ -22113,20 +22157,20 @@ HIDDEN _sk_from_srgb_sse41
 .globl _sk_from_srgb_sse41
 FUNCTION(_sk_from_srgb_sse41)
 _sk_from_srgb_sse41:
-  .byte  68,15,40,29,184,49,0,0              // movaps        0x31b8(%rip),%xmm11        # 49c0 <_sk_callback_sse41+0x3b3>
+  .byte  68,15,40,29,174,49,0,0              // movaps        0x31ae(%rip),%xmm11        # 49e0 <_sk_callback_sse41+0x3a9>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
   .byte  68,15,40,208                        // movaps        %xmm0,%xmm10
   .byte  69,15,89,210                        // mulps         %xmm10,%xmm10
-  .byte  68,15,40,37,176,49,0,0              // movaps        0x31b0(%rip),%xmm12        # 49d0 <_sk_callback_sse41+0x3c3>
+  .byte  68,15,40,37,166,49,0,0              // movaps        0x31a6(%rip),%xmm12        # 49f0 <_sk_callback_sse41+0x3b9>
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,89,196                        // mulps         %xmm12,%xmm8
-  .byte  68,15,40,45,176,49,0,0              // movaps        0x31b0(%rip),%xmm13        # 49e0 <_sk_callback_sse41+0x3d3>
+  .byte  68,15,40,45,166,49,0,0              // movaps        0x31a6(%rip),%xmm13        # 4a00 <_sk_callback_sse41+0x3c9>
   .byte  69,15,88,197                        // addps         %xmm13,%xmm8
   .byte  69,15,89,194                        // mulps         %xmm10,%xmm8
-  .byte  68,15,40,53,176,49,0,0              // movaps        0x31b0(%rip),%xmm14        # 49f0 <_sk_callback_sse41+0x3e3>
+  .byte  68,15,40,53,166,49,0,0              // movaps        0x31a6(%rip),%xmm14        # 4a10 <_sk_callback_sse41+0x3d9>
   .byte  69,15,88,198                        // addps         %xmm14,%xmm8
-  .byte  68,15,40,61,180,49,0,0              // movaps        0x31b4(%rip),%xmm15        # 4a00 <_sk_callback_sse41+0x3f3>
+  .byte  68,15,40,61,170,49,0,0              // movaps        0x31aa(%rip),%xmm15        # 4a20 <_sk_callback_sse41+0x3e9>
   .byte  65,15,194,199,1                     // cmpltps       %xmm15,%xmm0
   .byte  102,69,15,56,20,193                 // blendvps      %xmm0,%xmm9,%xmm8
   .byte  68,15,40,209                        // movaps        %xmm1,%xmm10
@@ -22171,20 +22215,20 @@ _sk_to_srgb_sse41:
   .byte  68,15,82,192                        // rsqrtps       %xmm0,%xmm8
   .byte  69,15,83,200                        // rcpps         %xmm8,%xmm9
   .byte  69,15,82,208                        // rsqrtps       %xmm8,%xmm10
-  .byte  68,15,40,29,36,49,0,0               // movaps        0x3124(%rip),%xmm11        # 4a10 <_sk_callback_sse41+0x403>
+  .byte  68,15,40,29,26,49,0,0               // movaps        0x311a(%rip),%xmm11        # 4a30 <_sk_callback_sse41+0x3f9>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
-  .byte  68,15,40,37,37,49,0,0               // movaps        0x3125(%rip),%xmm12        # 4a20 <_sk_callback_sse41+0x413>
+  .byte  68,15,40,37,27,49,0,0               // movaps        0x311b(%rip),%xmm12        # 4a40 <_sk_callback_sse41+0x409>
   .byte  69,15,89,204                        // mulps         %xmm12,%xmm9
-  .byte  68,15,40,45,41,49,0,0               // movaps        0x3129(%rip),%xmm13        # 4a30 <_sk_callback_sse41+0x423>
+  .byte  68,15,40,45,31,49,0,0               // movaps        0x311f(%rip),%xmm13        # 4a50 <_sk_callback_sse41+0x419>
   .byte  69,15,88,205                        // addps         %xmm13,%xmm9
-  .byte  68,15,40,53,45,49,0,0               // movaps        0x312d(%rip),%xmm14        # 4a40 <_sk_callback_sse41+0x433>
+  .byte  68,15,40,53,35,49,0,0               // movaps        0x3123(%rip),%xmm14        # 4a60 <_sk_callback_sse41+0x429>
   .byte  69,15,89,214                        // mulps         %xmm14,%xmm10
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
-  .byte  68,15,40,5,45,49,0,0                // movaps        0x312d(%rip),%xmm8        # 4a50 <_sk_callback_sse41+0x443>
+  .byte  68,15,40,5,35,49,0,0                // movaps        0x3123(%rip),%xmm8        # 4a70 <_sk_callback_sse41+0x439>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,93,202                        // minps         %xmm10,%xmm9
-  .byte  68,15,40,61,45,49,0,0               // movaps        0x312d(%rip),%xmm15        # 4a60 <_sk_callback_sse41+0x453>
+  .byte  68,15,40,61,35,49,0,0               // movaps        0x3123(%rip),%xmm15        # 4a80 <_sk_callback_sse41+0x449>
   .byte  65,15,194,199,1                     // cmpltps       %xmm15,%xmm0
   .byte  102,68,15,56,20,201                 // blendvps      %xmm0,%xmm1,%xmm9
   .byte  15,82,194                           // rsqrtps       %xmm2,%xmm0
@@ -22238,7 +22282,7 @@ _sk_rgb_to_hsl_sse41:
   .byte  68,15,93,226                        // minps         %xmm2,%xmm12
   .byte  65,15,40,203                        // movaps        %xmm11,%xmm1
   .byte  65,15,92,204                        // subps         %xmm12,%xmm1
-  .byte  68,15,40,53,126,48,0,0              // movaps        0x307e(%rip),%xmm14        # 4a70 <_sk_callback_sse41+0x463>
+  .byte  68,15,40,53,116,48,0,0              // movaps        0x3074(%rip),%xmm14        # 4a90 <_sk_callback_sse41+0x459>
   .byte  68,15,94,241                        // divps         %xmm1,%xmm14
   .byte  69,15,40,211                        // movaps        %xmm11,%xmm10
   .byte  69,15,194,208,0                     // cmpeqps       %xmm8,%xmm10
@@ -22247,27 +22291,27 @@ _sk_rgb_to_hsl_sse41:
   .byte  65,15,89,198                        // mulps         %xmm14,%xmm0
   .byte  69,15,40,249                        // movaps        %xmm9,%xmm15
   .byte  68,15,194,250,1                     // cmpltps       %xmm2,%xmm15
-  .byte  68,15,84,61,101,48,0,0              // andps         0x3065(%rip),%xmm15        # 4a80 <_sk_callback_sse41+0x473>
+  .byte  68,15,84,61,91,48,0,0               // andps         0x305b(%rip),%xmm15        # 4aa0 <_sk_callback_sse41+0x469>
   .byte  68,15,88,248                        // addps         %xmm0,%xmm15
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  65,15,194,193,0                     // cmpeqps       %xmm9,%xmm0
   .byte  65,15,92,208                        // subps         %xmm8,%xmm2
   .byte  65,15,89,214                        // mulps         %xmm14,%xmm2
-  .byte  68,15,40,45,88,48,0,0               // movaps        0x3058(%rip),%xmm13        # 4a90 <_sk_callback_sse41+0x483>
+  .byte  68,15,40,45,78,48,0,0               // movaps        0x304e(%rip),%xmm13        # 4ab0 <_sk_callback_sse41+0x479>
   .byte  65,15,88,213                        // addps         %xmm13,%xmm2
   .byte  69,15,92,193                        // subps         %xmm9,%xmm8
   .byte  69,15,89,198                        // mulps         %xmm14,%xmm8
-  .byte  68,15,88,5,84,48,0,0                // addps         0x3054(%rip),%xmm8        # 4aa0 <_sk_callback_sse41+0x493>
+  .byte  68,15,88,5,74,48,0,0                // addps         0x304a(%rip),%xmm8        # 4ac0 <_sk_callback_sse41+0x489>
   .byte  102,68,15,56,20,194                 // blendvps      %xmm0,%xmm2,%xmm8
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  102,69,15,56,20,199                 // blendvps      %xmm0,%xmm15,%xmm8
-  .byte  68,15,89,5,76,48,0,0                // mulps         0x304c(%rip),%xmm8        # 4ab0 <_sk_callback_sse41+0x4a3>
+  .byte  68,15,89,5,66,48,0,0                // mulps         0x3042(%rip),%xmm8        # 4ad0 <_sk_callback_sse41+0x499>
   .byte  69,15,40,203                        // movaps        %xmm11,%xmm9
   .byte  69,15,194,204,4                     // cmpneqps      %xmm12,%xmm9
   .byte  69,15,84,193                        // andps         %xmm9,%xmm8
   .byte  69,15,92,235                        // subps         %xmm11,%xmm13
   .byte  69,15,88,220                        // addps         %xmm12,%xmm11
-  .byte  15,40,5,64,48,0,0                   // movaps        0x3040(%rip),%xmm0        # 4ac0 <_sk_callback_sse41+0x4b3>
+  .byte  15,40,5,54,48,0,0                   // movaps        0x3036(%rip),%xmm0        # 4ae0 <_sk_callback_sse41+0x4a9>
   .byte  65,15,40,211                        // movaps        %xmm11,%xmm2
   .byte  15,89,208                           // mulps         %xmm0,%xmm2
   .byte  15,194,194,1                        // cmpltps       %xmm2,%xmm0
@@ -22289,7 +22333,7 @@ _sk_hsl_to_rgb_sse41:
   .byte  15,41,100,36,184                    // movaps        %xmm4,-0x48(%rsp)
   .byte  15,41,92,36,168                     // movaps        %xmm3,-0x58(%rsp)
   .byte  68,15,40,208                        // movaps        %xmm0,%xmm10
-  .byte  68,15,40,13,6,48,0,0                // movaps        0x3006(%rip),%xmm9        # 4ad0 <_sk_callback_sse41+0x4c3>
+  .byte  68,15,40,13,252,47,0,0              // movaps        0x2ffc(%rip),%xmm9        # 4af0 <_sk_callback_sse41+0x4b9>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  15,194,194,2                        // cmpleps       %xmm2,%xmm0
   .byte  15,40,217                           // movaps        %xmm1,%xmm3
@@ -22302,19 +22346,19 @@ _sk_hsl_to_rgb_sse41:
   .byte  15,41,84,36,152                     // movaps        %xmm2,-0x68(%rsp)
   .byte  69,15,88,192                        // addps         %xmm8,%xmm8
   .byte  68,15,92,197                        // subps         %xmm5,%xmm8
-  .byte  68,15,40,53,225,47,0,0              // movaps        0x2fe1(%rip),%xmm14        # 4ae0 <_sk_callback_sse41+0x4d3>
+  .byte  68,15,40,53,215,47,0,0              // movaps        0x2fd7(%rip),%xmm14        # 4b00 <_sk_callback_sse41+0x4c9>
   .byte  69,15,88,242                        // addps         %xmm10,%xmm14
   .byte  102,65,15,58,8,198,1                // roundps       $0x1,%xmm14,%xmm0
   .byte  68,15,92,240                        // subps         %xmm0,%xmm14
-  .byte  68,15,40,29,218,47,0,0              // movaps        0x2fda(%rip),%xmm11        # 4af0 <_sk_callback_sse41+0x4e3>
+  .byte  68,15,40,29,208,47,0,0              // movaps        0x2fd0(%rip),%xmm11        # 4b10 <_sk_callback_sse41+0x4d9>
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  65,15,194,198,2                     // cmpleps       %xmm14,%xmm0
   .byte  15,40,245                           // movaps        %xmm5,%xmm6
   .byte  65,15,92,240                        // subps         %xmm8,%xmm6
-  .byte  15,40,61,211,47,0,0                 // movaps        0x2fd3(%rip),%xmm7        # 4b00 <_sk_callback_sse41+0x4f3>
+  .byte  15,40,61,201,47,0,0                 // movaps        0x2fc9(%rip),%xmm7        # 4b20 <_sk_callback_sse41+0x4e9>
   .byte  69,15,40,238                        // movaps        %xmm14,%xmm13
   .byte  68,15,89,239                        // mulps         %xmm7,%xmm13
-  .byte  15,40,29,212,47,0,0                 // movaps        0x2fd4(%rip),%xmm3        # 4b10 <_sk_callback_sse41+0x503>
+  .byte  15,40,29,202,47,0,0                 // movaps        0x2fca(%rip),%xmm3        # 4b30 <_sk_callback_sse41+0x4f9>
   .byte  68,15,40,227                        // movaps        %xmm3,%xmm12
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  68,15,89,230                        // mulps         %xmm6,%xmm12
@@ -22324,7 +22368,7 @@ _sk_hsl_to_rgb_sse41:
   .byte  65,15,194,198,2                     // cmpleps       %xmm14,%xmm0
   .byte  68,15,40,253                        // movaps        %xmm5,%xmm15
   .byte  102,69,15,56,20,252                 // blendvps      %xmm0,%xmm12,%xmm15
-  .byte  68,15,40,37,179,47,0,0              // movaps        0x2fb3(%rip),%xmm12        # 4b20 <_sk_callback_sse41+0x513>
+  .byte  68,15,40,37,169,47,0,0              // movaps        0x2fa9(%rip),%xmm12        # 4b40 <_sk_callback_sse41+0x509>
   .byte  65,15,40,196                        // movaps        %xmm12,%xmm0
   .byte  65,15,194,198,2                     // cmpleps       %xmm14,%xmm0
   .byte  68,15,89,238                        // mulps         %xmm6,%xmm13
@@ -22358,7 +22402,7 @@ _sk_hsl_to_rgb_sse41:
   .byte  65,15,40,198                        // movaps        %xmm14,%xmm0
   .byte  15,40,84,36,152                     // movaps        -0x68(%rsp),%xmm2
   .byte  102,15,56,20,202                    // blendvps      %xmm0,%xmm2,%xmm1
-  .byte  68,15,88,21,43,47,0,0               // addps         0x2f2b(%rip),%xmm10        # 4b30 <_sk_callback_sse41+0x523>
+  .byte  68,15,88,21,33,47,0,0               // addps         0x2f21(%rip),%xmm10        # 4b50 <_sk_callback_sse41+0x519>
   .byte  102,65,15,58,8,194,1                // roundps       $0x1,%xmm10,%xmm0
   .byte  68,15,92,208                        // subps         %xmm0,%xmm10
   .byte  69,15,194,218,2                     // cmpleps       %xmm10,%xmm11
@@ -22410,7 +22454,7 @@ _sk_scale_u8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,68,15,56,49,4,56                // pmovzxbd      (%rax,%rdi,1),%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,136,46,0,0               // mulps         0x2e88(%rip),%xmm8        # 4b40 <_sk_callback_sse41+0x533>
+  .byte  68,15,89,5,126,46,0,0               // mulps         0x2e7e(%rip),%xmm8        # 4b60 <_sk_callback_sse41+0x529>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
@@ -22448,7 +22492,7 @@ _sk_lerp_u8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,68,15,56,49,4,56                // pmovzxbd      (%rax,%rdi,1),%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,52,46,0,0                // mulps         0x2e34(%rip),%xmm8        # 4b50 <_sk_callback_sse41+0x543>
+  .byte  68,15,89,5,42,46,0,0                // mulps         0x2e2a(%rip),%xmm8        # 4b70 <_sk_callback_sse41+0x539>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -22471,17 +22515,17 @@ _sk_lerp_565_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,68,15,56,51,20,120              // pmovzxwd      (%rax,%rdi,2),%xmm10
-  .byte  102,68,15,111,5,3,46,0,0            // movdqa        0x2e03(%rip),%xmm8        # 4b60 <_sk_callback_sse41+0x553>
+  .byte  102,68,15,111,5,249,45,0,0          // movdqa        0x2df9(%rip),%xmm8        # 4b80 <_sk_callback_sse41+0x549>
   .byte  102,69,15,219,194                   // pand          %xmm10,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,2,46,0,0                 // mulps         0x2e02(%rip),%xmm8        # 4b70 <_sk_callback_sse41+0x563>
-  .byte  102,68,15,111,13,9,46,0,0           // movdqa        0x2e09(%rip),%xmm9        # 4b80 <_sk_callback_sse41+0x573>
+  .byte  68,15,89,5,248,45,0,0               // mulps         0x2df8(%rip),%xmm8        # 4b90 <_sk_callback_sse41+0x559>
+  .byte  102,68,15,111,13,255,45,0,0         // movdqa        0x2dff(%rip),%xmm9        # 4ba0 <_sk_callback_sse41+0x569>
   .byte  102,69,15,219,202                   // pand          %xmm10,%xmm9
   .byte  69,15,91,201                        // cvtdq2ps      %xmm9,%xmm9
-  .byte  68,15,89,13,8,46,0,0                // mulps         0x2e08(%rip),%xmm9        # 4b90 <_sk_callback_sse41+0x583>
-  .byte  102,68,15,219,21,15,46,0,0          // pand          0x2e0f(%rip),%xmm10        # 4ba0 <_sk_callback_sse41+0x593>
+  .byte  68,15,89,13,254,45,0,0              // mulps         0x2dfe(%rip),%xmm9        # 4bb0 <_sk_callback_sse41+0x579>
+  .byte  102,68,15,219,21,5,46,0,0           // pand          0x2e05(%rip),%xmm10        # 4bc0 <_sk_callback_sse41+0x589>
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
-  .byte  68,15,89,21,19,46,0,0               // mulps         0x2e13(%rip),%xmm10        # 4bb0 <_sk_callback_sse41+0x5a3>
+  .byte  68,15,89,21,9,46,0,0                // mulps         0x2e09(%rip),%xmm10        # 4bd0 <_sk_callback_sse41+0x599>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -22512,7 +22556,7 @@ _sk_load_tables_sse41:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,139,72,8                         // mov           0x8(%rax),%r9
   .byte  243,69,15,111,4,184                 // movdqu        (%r8,%rdi,4),%xmm8
-  .byte  102,15,111,5,196,45,0,0             // movdqa        0x2dc4(%rip),%xmm0        # 4bc0 <_sk_callback_sse41+0x5b3>
+  .byte  102,15,111,5,186,45,0,0             // movdqa        0x2dba(%rip),%xmm0        # 4be0 <_sk_callback_sse41+0x5a9>
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,73,15,58,22,192,1               // pextrq        $0x1,%xmm0,%r8
   .byte  102,72,15,126,193                   // movq          %xmm0,%rcx
@@ -22527,7 +22571,7 @@ _sk_load_tables_sse41:
   .byte  102,15,58,33,193,48                 // insertps      $0x30,%xmm1,%xmm0
   .byte  76,139,64,16                        // mov           0x10(%rax),%r8
   .byte  102,65,15,111,200                   // movdqa        %xmm8,%xmm1
-  .byte  102,15,56,0,13,127,45,0,0           // pshufb        0x2d7f(%rip),%xmm1        # 4bd0 <_sk_callback_sse41+0x5c3>
+  .byte  102,15,56,0,13,117,45,0,0           // pshufb        0x2d75(%rip),%xmm1        # 4bf0 <_sk_callback_sse41+0x5b9>
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
   .byte  102,72,15,126,201                   // movq          %xmm1,%rcx
   .byte  68,15,182,209                       // movzbl        %cl,%r10d
@@ -22542,7 +22586,7 @@ _sk_load_tables_sse41:
   .byte  102,15,58,33,202,48                 // insertps      $0x30,%xmm2,%xmm1
   .byte  76,139,64,24                        // mov           0x18(%rax),%r8
   .byte  102,65,15,111,208                   // movdqa        %xmm8,%xmm2
-  .byte  102,15,56,0,21,59,45,0,0            // pshufb        0x2d3b(%rip),%xmm2        # 4be0 <_sk_callback_sse41+0x5d3>
+  .byte  102,15,56,0,21,49,45,0,0            // pshufb        0x2d31(%rip),%xmm2        # 4c00 <_sk_callback_sse41+0x5c9>
   .byte  102,72,15,58,22,209,1               // pextrq        $0x1,%xmm2,%rcx
   .byte  102,72,15,126,208                   // movq          %xmm2,%rax
   .byte  68,15,182,200                       // movzbl        %al,%r9d
@@ -22557,7 +22601,7 @@ _sk_load_tables_sse41:
   .byte  102,15,58,33,211,48                 // insertps      $0x30,%xmm3,%xmm2
   .byte  102,65,15,114,208,24                // psrld         $0x18,%xmm8
   .byte  65,15,91,216                        // cvtdq2ps      %xmm8,%xmm3
-  .byte  15,89,29,248,44,0,0                 // mulps         0x2cf8(%rip),%xmm3        # 4bf0 <_sk_callback_sse41+0x5e3>
+  .byte  15,89,29,238,44,0,0                 // mulps         0x2cee(%rip),%xmm3        # 4c10 <_sk_callback_sse41+0x5d9>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -22576,7 +22620,7 @@ _sk_load_tables_u16_be_sse41:
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,97,200                       // punpcklwd     %xmm0,%xmm1
   .byte  102,68,15,105,200                   // punpckhwd     %xmm0,%xmm9
-  .byte  102,68,15,111,5,203,44,0,0          // movdqa        0x2ccb(%rip),%xmm8        # 4c00 <_sk_callback_sse41+0x5f3>
+  .byte  102,68,15,111,5,193,44,0,0          // movdqa        0x2cc1(%rip),%xmm8        # 4c20 <_sk_callback_sse41+0x5e9>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,15,56,51,192                    // pmovzxwd      %xmm0,%xmm0
@@ -22593,7 +22637,7 @@ _sk_load_tables_u16_be_sse41:
   .byte  243,67,15,16,20,8                   // movss         (%r8,%r9,1),%xmm2
   .byte  102,15,58,33,194,48                 // insertps      $0x30,%xmm2,%xmm0
   .byte  76,139,64,16                        // mov           0x10(%rax),%r8
-  .byte  102,15,56,0,13,126,44,0,0           // pshufb        0x2c7e(%rip),%xmm1        # 4c10 <_sk_callback_sse41+0x603>
+  .byte  102,15,56,0,13,116,44,0,0           // pshufb        0x2c74(%rip),%xmm1        # 4c30 <_sk_callback_sse41+0x5f9>
   .byte  102,15,56,51,201                    // pmovzxwd      %xmm1,%xmm1
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
   .byte  102,72,15,126,201                   // movq          %xmm1,%rcx
@@ -22629,7 +22673,7 @@ _sk_load_tables_u16_be_sse41:
   .byte  102,65,15,235,216                   // por           %xmm8,%xmm3
   .byte  102,15,56,51,219                    // pmovzxwd      %xmm3,%xmm3
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,204,43,0,0                 // mulps         0x2bcc(%rip),%xmm3        # 4c20 <_sk_callback_sse41+0x613>
+  .byte  15,89,29,194,43,0,0                 // mulps         0x2bc2(%rip),%xmm3        # 4c40 <_sk_callback_sse41+0x609>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -22651,7 +22695,7 @@ _sk_load_tables_rgb_u16_be_sse41:
   .byte  102,68,15,97,200                    // punpcklwd     %xmm0,%xmm9
   .byte  102,15,111,202                      // movdqa        %xmm2,%xmm1
   .byte  102,65,15,97,201                    // punpcklwd     %xmm9,%xmm1
-  .byte  102,68,15,111,5,142,43,0,0          // movdqa        0x2b8e(%rip),%xmm8        # 4c30 <_sk_callback_sse41+0x623>
+  .byte  102,68,15,111,5,132,43,0,0          // movdqa        0x2b84(%rip),%xmm8        # 4c50 <_sk_callback_sse41+0x619>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,15,56,51,192                    // pmovzxwd      %xmm0,%xmm0
@@ -22668,7 +22712,7 @@ _sk_load_tables_rgb_u16_be_sse41:
   .byte  243,67,15,16,28,8                   // movss         (%r8,%r9,1),%xmm3
   .byte  102,15,58,33,195,48                 // insertps      $0x30,%xmm3,%xmm0
   .byte  76,139,64,16                        // mov           0x10(%rax),%r8
-  .byte  102,15,56,0,13,65,43,0,0            // pshufb        0x2b41(%rip),%xmm1        # 4c40 <_sk_callback_sse41+0x633>
+  .byte  102,15,56,0,13,55,43,0,0            // pshufb        0x2b37(%rip),%xmm1        # 4c60 <_sk_callback_sse41+0x629>
   .byte  102,15,56,51,201                    // pmovzxwd      %xmm1,%xmm1
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
   .byte  102,72,15,126,201                   // movq          %xmm1,%rcx
@@ -22699,7 +22743,7 @@ _sk_load_tables_rgb_u16_be_sse41:
   .byte  243,65,15,16,28,8                   // movss         (%r8,%rcx,1),%xmm3
   .byte  102,15,58,33,211,48                 // insertps      $0x30,%xmm3,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,172,42,0,0                 // movaps        0x2aac(%rip),%xmm3        # 4c50 <_sk_callback_sse41+0x643>
+  .byte  15,40,29,162,42,0,0                 // movaps        0x2aa2(%rip),%xmm3        # 4c70 <_sk_callback_sse41+0x639>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_byte_tables_sse41
@@ -22709,7 +22753,7 @@ _sk_byte_tables_sse41:
   .byte  65,86                               // push          %r14
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,173,42,0,0               // movaps        0x2aad(%rip),%xmm8        # 4c60 <_sk_callback_sse41+0x653>
+  .byte  68,15,40,5,163,42,0,0               // movaps        0x2aa3(%rip),%xmm8        # 4c80 <_sk_callback_sse41+0x649>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,91,192                       // cvtps2dq      %xmm0,%xmm0
   .byte  102,72,15,58,22,193,1               // pextrq        $0x1,%xmm0,%rcx
@@ -22728,7 +22772,7 @@ _sk_byte_tables_sse41:
   .byte  102,15,58,32,193,3                  // pinsrb        $0x3,%ecx,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,13,94,42,0,0               // movaps        0x2a5e(%rip),%xmm9        # 4c70 <_sk_callback_sse41+0x663>
+  .byte  68,15,40,13,84,42,0,0               // movaps        0x2a54(%rip),%xmm9        # 4c90 <_sk_callback_sse41+0x659>
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -22819,7 +22863,7 @@ _sk_byte_tables_rgb_sse41:
   .byte  102,15,58,32,193,3                  // pinsrb        $0x3,%ecx,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,13,230,40,0,0              // movaps        0x28e6(%rip),%xmm9        # 4c80 <_sk_callback_sse41+0x673>
+  .byte  68,15,40,13,220,40,0,0              // movaps        0x28dc(%rip),%xmm9        # 4ca0 <_sk_callback_sse41+0x669>
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -22996,31 +23040,31 @@ _sk_parametric_r_sse41:
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,194                        // cvtdq2ps      %xmm10,%xmm8
-  .byte  68,15,89,5,61,38,0,0                // mulps         0x263d(%rip),%xmm8        # 4c90 <_sk_callback_sse41+0x683>
-  .byte  68,15,84,21,69,38,0,0               // andps         0x2645(%rip),%xmm10        # 4ca0 <_sk_callback_sse41+0x693>
-  .byte  68,15,86,21,77,38,0,0               // orps          0x264d(%rip),%xmm10        # 4cb0 <_sk_callback_sse41+0x6a3>
-  .byte  68,15,88,5,85,38,0,0                // addps         0x2655(%rip),%xmm8        # 4cc0 <_sk_callback_sse41+0x6b3>
-  .byte  68,15,40,37,93,38,0,0               // movaps        0x265d(%rip),%xmm12        # 4cd0 <_sk_callback_sse41+0x6c3>
+  .byte  68,15,89,5,51,38,0,0                // mulps         0x2633(%rip),%xmm8        # 4cb0 <_sk_callback_sse41+0x679>
+  .byte  68,15,84,21,59,38,0,0               // andps         0x263b(%rip),%xmm10        # 4cc0 <_sk_callback_sse41+0x689>
+  .byte  68,15,86,21,67,38,0,0               // orps          0x2643(%rip),%xmm10        # 4cd0 <_sk_callback_sse41+0x699>
+  .byte  68,15,88,5,75,38,0,0                // addps         0x264b(%rip),%xmm8        # 4ce0 <_sk_callback_sse41+0x6a9>
+  .byte  68,15,40,37,83,38,0,0               // movaps        0x2653(%rip),%xmm12        # 4cf0 <_sk_callback_sse41+0x6b9>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,196                        // subps         %xmm12,%xmm8
-  .byte  68,15,88,21,93,38,0,0               // addps         0x265d(%rip),%xmm10        # 4ce0 <_sk_callback_sse41+0x6d3>
-  .byte  68,15,40,37,101,38,0,0              // movaps        0x2665(%rip),%xmm12        # 4cf0 <_sk_callback_sse41+0x6e3>
+  .byte  68,15,88,21,83,38,0,0               // addps         0x2653(%rip),%xmm10        # 4d00 <_sk_callback_sse41+0x6c9>
+  .byte  68,15,40,37,91,38,0,0               // movaps        0x265b(%rip),%xmm12        # 4d10 <_sk_callback_sse41+0x6d9>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,196                        // subps         %xmm12,%xmm8
   .byte  69,15,89,195                        // mulps         %xmm11,%xmm8
   .byte  102,69,15,58,8,208,1                // roundps       $0x1,%xmm8,%xmm10
   .byte  69,15,40,216                        // movaps        %xmm8,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,5,82,38,0,0                // addps         0x2652(%rip),%xmm8        # 4d00 <_sk_callback_sse41+0x6f3>
-  .byte  68,15,40,21,90,38,0,0               // movaps        0x265a(%rip),%xmm10        # 4d10 <_sk_callback_sse41+0x703>
+  .byte  68,15,88,5,72,38,0,0                // addps         0x2648(%rip),%xmm8        # 4d20 <_sk_callback_sse41+0x6e9>
+  .byte  68,15,40,21,80,38,0,0               // movaps        0x2650(%rip),%xmm10        # 4d30 <_sk_callback_sse41+0x6f9>
   .byte  69,15,89,211                        // mulps         %xmm11,%xmm10
   .byte  69,15,92,194                        // subps         %xmm10,%xmm8
-  .byte  68,15,40,21,90,38,0,0               // movaps        0x265a(%rip),%xmm10        # 4d20 <_sk_callback_sse41+0x713>
+  .byte  68,15,40,21,80,38,0,0               // movaps        0x2650(%rip),%xmm10        # 4d40 <_sk_callback_sse41+0x709>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  68,15,40,29,94,38,0,0               // movaps        0x265e(%rip),%xmm11        # 4d30 <_sk_callback_sse41+0x723>
+  .byte  68,15,40,29,84,38,0,0               // movaps        0x2654(%rip),%xmm11        # 4d50 <_sk_callback_sse41+0x719>
   .byte  69,15,94,218                        // divps         %xmm10,%xmm11
   .byte  69,15,88,216                        // addps         %xmm8,%xmm11
-  .byte  68,15,89,29,94,38,0,0               // mulps         0x265e(%rip),%xmm11        # 4d40 <_sk_callback_sse41+0x733>
+  .byte  68,15,89,29,84,38,0,0               // mulps         0x2654(%rip),%xmm11        # 4d60 <_sk_callback_sse41+0x729>
   .byte  102,69,15,91,211                    // cvtps2dq      %xmm11,%xmm10
   .byte  243,68,15,16,64,20                  // movss         0x14(%rax),%xmm8
   .byte  69,15,198,192,0                     // shufps        $0x0,%xmm8,%xmm8
@@ -23028,7 +23072,7 @@ _sk_parametric_r_sse41:
   .byte  102,69,15,56,20,193                 // blendvps      %xmm0,%xmm9,%xmm8
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  68,15,95,192                        // maxps         %xmm0,%xmm8
-  .byte  68,15,93,5,69,38,0,0                // minps         0x2645(%rip),%xmm8        # 4d50 <_sk_callback_sse41+0x743>
+  .byte  68,15,93,5,59,38,0,0                // minps         0x263b(%rip),%xmm8        # 4d70 <_sk_callback_sse41+0x739>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -23058,31 +23102,31 @@ _sk_parametric_g_sse41:
   .byte  68,15,88,217                        // addps         %xmm1,%xmm11
   .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
   .byte  69,15,91,227                        // cvtdq2ps      %xmm11,%xmm12
-  .byte  68,15,89,37,230,37,0,0              // mulps         0x25e6(%rip),%xmm12        # 4d60 <_sk_callback_sse41+0x753>
-  .byte  68,15,84,29,238,37,0,0              // andps         0x25ee(%rip),%xmm11        # 4d70 <_sk_callback_sse41+0x763>
-  .byte  68,15,86,29,246,37,0,0              // orps          0x25f6(%rip),%xmm11        # 4d80 <_sk_callback_sse41+0x773>
-  .byte  68,15,88,37,254,37,0,0              // addps         0x25fe(%rip),%xmm12        # 4d90 <_sk_callback_sse41+0x783>
-  .byte  15,40,13,7,38,0,0                   // movaps        0x2607(%rip),%xmm1        # 4da0 <_sk_callback_sse41+0x793>
+  .byte  68,15,89,37,220,37,0,0              // mulps         0x25dc(%rip),%xmm12        # 4d80 <_sk_callback_sse41+0x749>
+  .byte  68,15,84,29,228,37,0,0              // andps         0x25e4(%rip),%xmm11        # 4d90 <_sk_callback_sse41+0x759>
+  .byte  68,15,86,29,236,37,0,0              // orps          0x25ec(%rip),%xmm11        # 4da0 <_sk_callback_sse41+0x769>
+  .byte  68,15,88,37,244,37,0,0              // addps         0x25f4(%rip),%xmm12        # 4db0 <_sk_callback_sse41+0x779>
+  .byte  15,40,13,253,37,0,0                 // movaps        0x25fd(%rip),%xmm1        # 4dc0 <_sk_callback_sse41+0x789>
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
   .byte  68,15,92,225                        // subps         %xmm1,%xmm12
-  .byte  68,15,88,29,7,38,0,0                // addps         0x2607(%rip),%xmm11        # 4db0 <_sk_callback_sse41+0x7a3>
-  .byte  15,40,13,16,38,0,0                  // movaps        0x2610(%rip),%xmm1        # 4dc0 <_sk_callback_sse41+0x7b3>
+  .byte  68,15,88,29,253,37,0,0              // addps         0x25fd(%rip),%xmm11        # 4dd0 <_sk_callback_sse41+0x799>
+  .byte  15,40,13,6,38,0,0                   // movaps        0x2606(%rip),%xmm1        # 4de0 <_sk_callback_sse41+0x7a9>
   .byte  65,15,94,203                        // divps         %xmm11,%xmm1
   .byte  68,15,92,225                        // subps         %xmm1,%xmm12
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  102,69,15,58,8,212,1                // roundps       $0x1,%xmm12,%xmm10
   .byte  69,15,40,220                        // movaps        %xmm12,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,37,253,37,0,0              // addps         0x25fd(%rip),%xmm12        # 4dd0 <_sk_callback_sse41+0x7c3>
-  .byte  15,40,13,6,38,0,0                   // movaps        0x2606(%rip),%xmm1        # 4de0 <_sk_callback_sse41+0x7d3>
+  .byte  68,15,88,37,243,37,0,0              // addps         0x25f3(%rip),%xmm12        # 4df0 <_sk_callback_sse41+0x7b9>
+  .byte  15,40,13,252,37,0,0                 // movaps        0x25fc(%rip),%xmm1        # 4e00 <_sk_callback_sse41+0x7c9>
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
   .byte  68,15,92,225                        // subps         %xmm1,%xmm12
-  .byte  68,15,40,21,6,38,0,0                // movaps        0x2606(%rip),%xmm10        # 4df0 <_sk_callback_sse41+0x7e3>
+  .byte  68,15,40,21,252,37,0,0              // movaps        0x25fc(%rip),%xmm10        # 4e10 <_sk_callback_sse41+0x7d9>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  15,40,13,11,38,0,0                  // movaps        0x260b(%rip),%xmm1        # 4e00 <_sk_callback_sse41+0x7f3>
+  .byte  15,40,13,1,38,0,0                   // movaps        0x2601(%rip),%xmm1        # 4e20 <_sk_callback_sse41+0x7e9>
   .byte  65,15,94,202                        // divps         %xmm10,%xmm1
   .byte  65,15,88,204                        // addps         %xmm12,%xmm1
-  .byte  15,89,13,12,38,0,0                  // mulps         0x260c(%rip),%xmm1        # 4e10 <_sk_callback_sse41+0x803>
+  .byte  15,89,13,2,38,0,0                   // mulps         0x2602(%rip),%xmm1        # 4e30 <_sk_callback_sse41+0x7f9>
   .byte  102,68,15,91,209                    // cvtps2dq      %xmm1,%xmm10
   .byte  243,15,16,72,20                     // movss         0x14(%rax),%xmm1
   .byte  15,198,201,0                        // shufps        $0x0,%xmm1,%xmm1
@@ -23090,7 +23134,7 @@ _sk_parametric_g_sse41:
   .byte  102,65,15,56,20,201                 // blendvps      %xmm0,%xmm9,%xmm1
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,95,200                           // maxps         %xmm0,%xmm1
-  .byte  15,93,13,247,37,0,0                 // minps         0x25f7(%rip),%xmm1        # 4e20 <_sk_callback_sse41+0x813>
+  .byte  15,93,13,237,37,0,0                 // minps         0x25ed(%rip),%xmm1        # 4e40 <_sk_callback_sse41+0x809>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -23120,31 +23164,31 @@ _sk_parametric_b_sse41:
   .byte  68,15,88,218                        // addps         %xmm2,%xmm11
   .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
   .byte  69,15,91,227                        // cvtdq2ps      %xmm11,%xmm12
-  .byte  68,15,89,37,152,37,0,0              // mulps         0x2598(%rip),%xmm12        # 4e30 <_sk_callback_sse41+0x823>
-  .byte  68,15,84,29,160,37,0,0              // andps         0x25a0(%rip),%xmm11        # 4e40 <_sk_callback_sse41+0x833>
-  .byte  68,15,86,29,168,37,0,0              // orps          0x25a8(%rip),%xmm11        # 4e50 <_sk_callback_sse41+0x843>
-  .byte  68,15,88,37,176,37,0,0              // addps         0x25b0(%rip),%xmm12        # 4e60 <_sk_callback_sse41+0x853>
-  .byte  15,40,21,185,37,0,0                 // movaps        0x25b9(%rip),%xmm2        # 4e70 <_sk_callback_sse41+0x863>
+  .byte  68,15,89,37,142,37,0,0              // mulps         0x258e(%rip),%xmm12        # 4e50 <_sk_callback_sse41+0x819>
+  .byte  68,15,84,29,150,37,0,0              // andps         0x2596(%rip),%xmm11        # 4e60 <_sk_callback_sse41+0x829>
+  .byte  68,15,86,29,158,37,0,0              // orps          0x259e(%rip),%xmm11        # 4e70 <_sk_callback_sse41+0x839>
+  .byte  68,15,88,37,166,37,0,0              // addps         0x25a6(%rip),%xmm12        # 4e80 <_sk_callback_sse41+0x849>
+  .byte  15,40,21,175,37,0,0                 // movaps        0x25af(%rip),%xmm2        # 4e90 <_sk_callback_sse41+0x859>
   .byte  65,15,89,211                        // mulps         %xmm11,%xmm2
   .byte  68,15,92,226                        // subps         %xmm2,%xmm12
-  .byte  68,15,88,29,185,37,0,0              // addps         0x25b9(%rip),%xmm11        # 4e80 <_sk_callback_sse41+0x873>
-  .byte  15,40,21,194,37,0,0                 // movaps        0x25c2(%rip),%xmm2        # 4e90 <_sk_callback_sse41+0x883>
+  .byte  68,15,88,29,175,37,0,0              // addps         0x25af(%rip),%xmm11        # 4ea0 <_sk_callback_sse41+0x869>
+  .byte  15,40,21,184,37,0,0                 // movaps        0x25b8(%rip),%xmm2        # 4eb0 <_sk_callback_sse41+0x879>
   .byte  65,15,94,211                        // divps         %xmm11,%xmm2
   .byte  68,15,92,226                        // subps         %xmm2,%xmm12
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  102,69,15,58,8,212,1                // roundps       $0x1,%xmm12,%xmm10
   .byte  69,15,40,220                        // movaps        %xmm12,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,37,175,37,0,0              // addps         0x25af(%rip),%xmm12        # 4ea0 <_sk_callback_sse41+0x893>
-  .byte  15,40,21,184,37,0,0                 // movaps        0x25b8(%rip),%xmm2        # 4eb0 <_sk_callback_sse41+0x8a3>
+  .byte  68,15,88,37,165,37,0,0              // addps         0x25a5(%rip),%xmm12        # 4ec0 <_sk_callback_sse41+0x889>
+  .byte  15,40,21,174,37,0,0                 // movaps        0x25ae(%rip),%xmm2        # 4ed0 <_sk_callback_sse41+0x899>
   .byte  65,15,89,211                        // mulps         %xmm11,%xmm2
   .byte  68,15,92,226                        // subps         %xmm2,%xmm12
-  .byte  68,15,40,21,184,37,0,0              // movaps        0x25b8(%rip),%xmm10        # 4ec0 <_sk_callback_sse41+0x8b3>
+  .byte  68,15,40,21,174,37,0,0              // movaps        0x25ae(%rip),%xmm10        # 4ee0 <_sk_callback_sse41+0x8a9>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  15,40,21,189,37,0,0                 // movaps        0x25bd(%rip),%xmm2        # 4ed0 <_sk_callback_sse41+0x8c3>
+  .byte  15,40,21,179,37,0,0                 // movaps        0x25b3(%rip),%xmm2        # 4ef0 <_sk_callback_sse41+0x8b9>
   .byte  65,15,94,210                        // divps         %xmm10,%xmm2
   .byte  65,15,88,212                        // addps         %xmm12,%xmm2
-  .byte  15,89,21,190,37,0,0                 // mulps         0x25be(%rip),%xmm2        # 4ee0 <_sk_callback_sse41+0x8d3>
+  .byte  15,89,21,180,37,0,0                 // mulps         0x25b4(%rip),%xmm2        # 4f00 <_sk_callback_sse41+0x8c9>
   .byte  102,68,15,91,210                    // cvtps2dq      %xmm2,%xmm10
   .byte  243,15,16,80,20                     // movss         0x14(%rax),%xmm2
   .byte  15,198,210,0                        // shufps        $0x0,%xmm2,%xmm2
@@ -23152,7 +23196,7 @@ _sk_parametric_b_sse41:
   .byte  102,65,15,56,20,209                 // blendvps      %xmm0,%xmm9,%xmm2
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,95,208                           // maxps         %xmm0,%xmm2
-  .byte  15,93,21,169,37,0,0                 // minps         0x25a9(%rip),%xmm2        # 4ef0 <_sk_callback_sse41+0x8e3>
+  .byte  15,93,21,159,37,0,0                 // minps         0x259f(%rip),%xmm2        # 4f10 <_sk_callback_sse41+0x8d9>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -23182,31 +23226,31 @@ _sk_parametric_a_sse41:
   .byte  68,15,88,219                        // addps         %xmm3,%xmm11
   .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
   .byte  69,15,91,227                        // cvtdq2ps      %xmm11,%xmm12
-  .byte  68,15,89,37,74,37,0,0               // mulps         0x254a(%rip),%xmm12        # 4f00 <_sk_callback_sse41+0x8f3>
-  .byte  68,15,84,29,82,37,0,0               // andps         0x2552(%rip),%xmm11        # 4f10 <_sk_callback_sse41+0x903>
-  .byte  68,15,86,29,90,37,0,0               // orps          0x255a(%rip),%xmm11        # 4f20 <_sk_callback_sse41+0x913>
-  .byte  68,15,88,37,98,37,0,0               // addps         0x2562(%rip),%xmm12        # 4f30 <_sk_callback_sse41+0x923>
-  .byte  15,40,29,107,37,0,0                 // movaps        0x256b(%rip),%xmm3        # 4f40 <_sk_callback_sse41+0x933>
+  .byte  68,15,89,37,64,37,0,0               // mulps         0x2540(%rip),%xmm12        # 4f20 <_sk_callback_sse41+0x8e9>
+  .byte  68,15,84,29,72,37,0,0               // andps         0x2548(%rip),%xmm11        # 4f30 <_sk_callback_sse41+0x8f9>
+  .byte  68,15,86,29,80,37,0,0               // orps          0x2550(%rip),%xmm11        # 4f40 <_sk_callback_sse41+0x909>
+  .byte  68,15,88,37,88,37,0,0               // addps         0x2558(%rip),%xmm12        # 4f50 <_sk_callback_sse41+0x919>
+  .byte  15,40,29,97,37,0,0                  // movaps        0x2561(%rip),%xmm3        # 4f60 <_sk_callback_sse41+0x929>
   .byte  65,15,89,219                        // mulps         %xmm11,%xmm3
   .byte  68,15,92,227                        // subps         %xmm3,%xmm12
-  .byte  68,15,88,29,107,37,0,0              // addps         0x256b(%rip),%xmm11        # 4f50 <_sk_callback_sse41+0x943>
-  .byte  15,40,29,116,37,0,0                 // movaps        0x2574(%rip),%xmm3        # 4f60 <_sk_callback_sse41+0x953>
+  .byte  68,15,88,29,97,37,0,0               // addps         0x2561(%rip),%xmm11        # 4f70 <_sk_callback_sse41+0x939>
+  .byte  15,40,29,106,37,0,0                 // movaps        0x256a(%rip),%xmm3        # 4f80 <_sk_callback_sse41+0x949>
   .byte  65,15,94,219                        // divps         %xmm11,%xmm3
   .byte  68,15,92,227                        // subps         %xmm3,%xmm12
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  102,69,15,58,8,212,1                // roundps       $0x1,%xmm12,%xmm10
   .byte  69,15,40,220                        // movaps        %xmm12,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  68,15,88,37,97,37,0,0               // addps         0x2561(%rip),%xmm12        # 4f70 <_sk_callback_sse41+0x963>
-  .byte  15,40,29,106,37,0,0                 // movaps        0x256a(%rip),%xmm3        # 4f80 <_sk_callback_sse41+0x973>
+  .byte  68,15,88,37,87,37,0,0               // addps         0x2557(%rip),%xmm12        # 4f90 <_sk_callback_sse41+0x959>
+  .byte  15,40,29,96,37,0,0                  // movaps        0x2560(%rip),%xmm3        # 4fa0 <_sk_callback_sse41+0x969>
   .byte  65,15,89,219                        // mulps         %xmm11,%xmm3
   .byte  68,15,92,227                        // subps         %xmm3,%xmm12
-  .byte  68,15,40,21,106,37,0,0              // movaps        0x256a(%rip),%xmm10        # 4f90 <_sk_callback_sse41+0x983>
+  .byte  68,15,40,21,96,37,0,0               // movaps        0x2560(%rip),%xmm10        # 4fb0 <_sk_callback_sse41+0x979>
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
-  .byte  15,40,29,111,37,0,0                 // movaps        0x256f(%rip),%xmm3        # 4fa0 <_sk_callback_sse41+0x993>
+  .byte  15,40,29,101,37,0,0                 // movaps        0x2565(%rip),%xmm3        # 4fc0 <_sk_callback_sse41+0x989>
   .byte  65,15,94,218                        // divps         %xmm10,%xmm3
   .byte  65,15,88,220                        // addps         %xmm12,%xmm3
-  .byte  15,89,29,112,37,0,0                 // mulps         0x2570(%rip),%xmm3        # 4fb0 <_sk_callback_sse41+0x9a3>
+  .byte  15,89,29,102,37,0,0                 // mulps         0x2566(%rip),%xmm3        # 4fd0 <_sk_callback_sse41+0x999>
   .byte  102,68,15,91,211                    // cvtps2dq      %xmm3,%xmm10
   .byte  243,15,16,88,20                     // movss         0x14(%rax),%xmm3
   .byte  15,198,219,0                        // shufps        $0x0,%xmm3,%xmm3
@@ -23214,7 +23258,7 @@ _sk_parametric_a_sse41:
   .byte  102,65,15,56,20,217                 // blendvps      %xmm0,%xmm9,%xmm3
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,95,216                           // maxps         %xmm0,%xmm3
-  .byte  15,93,29,91,37,0,0                  // minps         0x255b(%rip),%xmm3        # 4fc0 <_sk_callback_sse41+0x9b3>
+  .byte  15,93,29,81,37,0,0                  // minps         0x2551(%rip),%xmm3        # 4fe0 <_sk_callback_sse41+0x9a9>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -23224,29 +23268,29 @@ HIDDEN _sk_lab_to_xyz_sse41
 FUNCTION(_sk_lab_to_xyz_sse41)
 _sk_lab_to_xyz_sse41:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,89,5,87,37,0,0                // mulps         0x2557(%rip),%xmm8        # 4fd0 <_sk_callback_sse41+0x9c3>
-  .byte  68,15,40,13,95,37,0,0               // movaps        0x255f(%rip),%xmm9        # 4fe0 <_sk_callback_sse41+0x9d3>
+  .byte  68,15,89,5,77,37,0,0                // mulps         0x254d(%rip),%xmm8        # 4ff0 <_sk_callback_sse41+0x9b9>
+  .byte  68,15,40,13,85,37,0,0               // movaps        0x2555(%rip),%xmm9        # 5000 <_sk_callback_sse41+0x9c9>
   .byte  65,15,89,201                        // mulps         %xmm9,%xmm1
-  .byte  15,40,5,100,37,0,0                  // movaps        0x2564(%rip),%xmm0        # 4ff0 <_sk_callback_sse41+0x9e3>
+  .byte  15,40,5,90,37,0,0                   // movaps        0x255a(%rip),%xmm0        # 5010 <_sk_callback_sse41+0x9d9>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
   .byte  15,88,208                           // addps         %xmm0,%xmm2
-  .byte  68,15,88,5,98,37,0,0                // addps         0x2562(%rip),%xmm8        # 5000 <_sk_callback_sse41+0x9f3>
-  .byte  68,15,89,5,106,37,0,0               // mulps         0x256a(%rip),%xmm8        # 5010 <_sk_callback_sse41+0xa03>
-  .byte  15,89,13,115,37,0,0                 // mulps         0x2573(%rip),%xmm1        # 5020 <_sk_callback_sse41+0xa13>
+  .byte  68,15,88,5,88,37,0,0                // addps         0x2558(%rip),%xmm8        # 5020 <_sk_callback_sse41+0x9e9>
+  .byte  68,15,89,5,96,37,0,0                // mulps         0x2560(%rip),%xmm8        # 5030 <_sk_callback_sse41+0x9f9>
+  .byte  15,89,13,105,37,0,0                 // mulps         0x2569(%rip),%xmm1        # 5040 <_sk_callback_sse41+0xa09>
   .byte  65,15,88,200                        // addps         %xmm8,%xmm1
-  .byte  15,89,21,120,37,0,0                 // mulps         0x2578(%rip),%xmm2        # 5030 <_sk_callback_sse41+0xa23>
+  .byte  15,89,21,110,37,0,0                 // mulps         0x256e(%rip),%xmm2        # 5050 <_sk_callback_sse41+0xa19>
   .byte  69,15,40,208                        // movaps        %xmm8,%xmm10
   .byte  68,15,92,210                        // subps         %xmm2,%xmm10
   .byte  68,15,40,217                        // movaps        %xmm1,%xmm11
   .byte  69,15,89,219                        // mulps         %xmm11,%xmm11
   .byte  68,15,89,217                        // mulps         %xmm1,%xmm11
-  .byte  68,15,40,13,108,37,0,0              // movaps        0x256c(%rip),%xmm9        # 5040 <_sk_callback_sse41+0xa33>
+  .byte  68,15,40,13,98,37,0,0               // movaps        0x2562(%rip),%xmm9        # 5060 <_sk_callback_sse41+0xa29>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  65,15,194,195,1                     // cmpltps       %xmm11,%xmm0
-  .byte  15,40,21,108,37,0,0                 // movaps        0x256c(%rip),%xmm2        # 5050 <_sk_callback_sse41+0xa43>
+  .byte  15,40,21,98,37,0,0                  // movaps        0x2562(%rip),%xmm2        # 5070 <_sk_callback_sse41+0xa39>
   .byte  15,88,202                           // addps         %xmm2,%xmm1
-  .byte  68,15,40,37,113,37,0,0              // movaps        0x2571(%rip),%xmm12        # 5060 <_sk_callback_sse41+0xa53>
+  .byte  68,15,40,37,103,37,0,0              // movaps        0x2567(%rip),%xmm12        # 5080 <_sk_callback_sse41+0xa49>
   .byte  65,15,89,204                        // mulps         %xmm12,%xmm1
   .byte  102,65,15,56,20,203                 // blendvps      %xmm0,%xmm11,%xmm1
   .byte  69,15,40,216                        // movaps        %xmm8,%xmm11
@@ -23265,8 +23309,8 @@ _sk_lab_to_xyz_sse41:
   .byte  65,15,89,212                        // mulps         %xmm12,%xmm2
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  102,65,15,56,20,211                 // blendvps      %xmm0,%xmm11,%xmm2
-  .byte  15,89,13,42,37,0,0                  // mulps         0x252a(%rip),%xmm1        # 5070 <_sk_callback_sse41+0xa63>
-  .byte  15,89,21,51,37,0,0                  // mulps         0x2533(%rip),%xmm2        # 5080 <_sk_callback_sse41+0xa73>
+  .byte  15,89,13,32,37,0,0                  // mulps         0x2520(%rip),%xmm1        # 5090 <_sk_callback_sse41+0xa59>
+  .byte  15,89,21,41,37,0,0                  // mulps         0x2529(%rip),%xmm2        # 50a0 <_sk_callback_sse41+0xa69>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,40,193                           // movaps        %xmm1,%xmm0
   .byte  65,15,40,200                        // movaps        %xmm8,%xmm1
@@ -23280,7 +23324,7 @@ _sk_load_a8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,49,4,56                   // pmovzxbd      (%rax,%rdi,1),%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,35,37,0,0                  // mulps         0x2523(%rip),%xmm3        # 5090 <_sk_callback_sse41+0xa83>
+  .byte  15,89,29,25,37,0,0                  // mulps         0x2519(%rip),%xmm3        # 50b0 <_sk_callback_sse41+0xa79>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  15,87,201                           // xorps         %xmm1,%xmm1
@@ -23313,7 +23357,7 @@ _sk_gather_a8_sse41:
   .byte  102,15,58,32,192,3                  // pinsrb        $0x3,%eax,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,183,36,0,0                 // mulps         0x24b7(%rip),%xmm3        # 50a0 <_sk_callback_sse41+0xa93>
+  .byte  15,89,29,173,36,0,0                 // mulps         0x24ad(%rip),%xmm3        # 50c0 <_sk_callback_sse41+0xa89>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
@@ -23326,7 +23370,7 @@ FUNCTION(_sk_store_a8_sse41)
 _sk_store_a8_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,171,36,0,0               // movaps        0x24ab(%rip),%xmm8        # 50b0 <_sk_callback_sse41+0xaa3>
+  .byte  68,15,40,5,161,36,0,0               // movaps        0x24a1(%rip),%xmm8        # 50d0 <_sk_callback_sse41+0xa99>
   .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
   .byte  102,69,15,56,43,192                 // packusdw      %xmm8,%xmm8
@@ -23343,9 +23387,9 @@ _sk_load_g8_sse41:
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,49,4,56                   // pmovzxbd      (%rax,%rdi,1),%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,136,36,0,0                  // mulps         0x2488(%rip),%xmm0        # 50c0 <_sk_callback_sse41+0xab3>
+  .byte  15,89,5,126,36,0,0                  // mulps         0x247e(%rip),%xmm0        # 50e0 <_sk_callback_sse41+0xaa9>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,143,36,0,0                 // movaps        0x248f(%rip),%xmm3        # 50d0 <_sk_callback_sse41+0xac3>
+  .byte  15,40,29,133,36,0,0                 // movaps        0x2485(%rip),%xmm3        # 50f0 <_sk_callback_sse41+0xab9>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -23376,9 +23420,9 @@ _sk_gather_g8_sse41:
   .byte  102,15,58,32,192,3                  // pinsrb        $0x3,%eax,%xmm0
   .byte  102,15,56,49,192                    // pmovzxbd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,40,36,0,0                   // mulps         0x2428(%rip),%xmm0        # 50e0 <_sk_callback_sse41+0xad3>
+  .byte  15,89,5,30,36,0,0                   // mulps         0x241e(%rip),%xmm0        # 5100 <_sk_callback_sse41+0xac9>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,47,36,0,0                  // movaps        0x242f(%rip),%xmm3        # 50f0 <_sk_callback_sse41+0xae3>
+  .byte  15,40,29,37,36,0,0                  // movaps        0x2425(%rip),%xmm3        # 5110 <_sk_callback_sse41+0xad9>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -23390,9 +23434,9 @@ _sk_gather_i8_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            2cd8 <_sk_gather_i8_sse41+0xf>
+  .byte  116,5                               // je            2d02 <_sk_gather_i8_sse41+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           2cda <_sk_gather_i8_sse41+0x11>
+  .byte  235,2                               // jmp           2d04 <_sk_gather_i8_sse41+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  243,15,91,201                       // cvttps2dq     %xmm1,%xmm1
@@ -23423,17 +23467,17 @@ _sk_gather_i8_sse41:
   .byte  102,15,58,34,28,8,1                 // pinsrd        $0x1,(%rax,%rcx,1),%xmm3
   .byte  102,66,15,58,34,28,144,2            // pinsrd        $0x2,(%rax,%r10,4),%xmm3
   .byte  102,66,15,58,34,28,8,3              // pinsrd        $0x3,(%rax,%r9,1),%xmm3
-  .byte  102,15,111,5,134,35,0,0             // movdqa        0x2386(%rip),%xmm0        # 5100 <_sk_callback_sse41+0xaf3>
+  .byte  102,15,111,5,124,35,0,0             // movdqa        0x237c(%rip),%xmm0        # 5120 <_sk_callback_sse41+0xae9>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,135,35,0,0               // movaps        0x2387(%rip),%xmm8        # 5110 <_sk_callback_sse41+0xb03>
+  .byte  68,15,40,5,125,35,0,0               // movaps        0x237d(%rip),%xmm8        # 5130 <_sk_callback_sse41+0xaf9>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
-  .byte  102,15,56,0,13,134,35,0,0           // pshufb        0x2386(%rip),%xmm1        # 5120 <_sk_callback_sse41+0xb13>
+  .byte  102,15,56,0,13,124,35,0,0           // pshufb        0x237c(%rip),%xmm1        # 5140 <_sk_callback_sse41+0xb09>
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,111,211                      // movdqa        %xmm3,%xmm2
-  .byte  102,15,56,0,21,130,35,0,0           // pshufb        0x2382(%rip),%xmm2        # 5130 <_sk_callback_sse41+0xb23>
+  .byte  102,15,56,0,21,120,35,0,0           // pshufb        0x2378(%rip),%xmm2        # 5150 <_sk_callback_sse41+0xb19>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  102,15,114,211,24                   // psrld         $0x18,%xmm3
@@ -23449,19 +23493,19 @@ _sk_load_565_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,51,20,120                 // pmovzxwd      (%rax,%rdi,2),%xmm2
-  .byte  102,15,111,5,104,35,0,0             // movdqa        0x2368(%rip),%xmm0        # 5140 <_sk_callback_sse41+0xb33>
+  .byte  102,15,111,5,94,35,0,0              // movdqa        0x235e(%rip),%xmm0        # 5160 <_sk_callback_sse41+0xb29>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,106,35,0,0                  // mulps         0x236a(%rip),%xmm0        # 5150 <_sk_callback_sse41+0xb43>
-  .byte  102,15,111,13,114,35,0,0            // movdqa        0x2372(%rip),%xmm1        # 5160 <_sk_callback_sse41+0xb53>
+  .byte  15,89,5,96,35,0,0                   // mulps         0x2360(%rip),%xmm0        # 5170 <_sk_callback_sse41+0xb39>
+  .byte  102,15,111,13,104,35,0,0            // movdqa        0x2368(%rip),%xmm1        # 5180 <_sk_callback_sse41+0xb49>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,116,35,0,0                 // mulps         0x2374(%rip),%xmm1        # 5170 <_sk_callback_sse41+0xb63>
-  .byte  102,15,219,21,124,35,0,0            // pand          0x237c(%rip),%xmm2        # 5180 <_sk_callback_sse41+0xb73>
+  .byte  15,89,13,106,35,0,0                 // mulps         0x236a(%rip),%xmm1        # 5190 <_sk_callback_sse41+0xb59>
+  .byte  102,15,219,21,114,35,0,0            // pand          0x2372(%rip),%xmm2        # 51a0 <_sk_callback_sse41+0xb69>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,130,35,0,0                 // mulps         0x2382(%rip),%xmm2        # 5190 <_sk_callback_sse41+0xb83>
+  .byte  15,89,21,120,35,0,0                 // mulps         0x2378(%rip),%xmm2        # 51b0 <_sk_callback_sse41+0xb79>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,137,35,0,0                 // movaps        0x2389(%rip),%xmm3        # 51a0 <_sk_callback_sse41+0xb93>
+  .byte  15,40,29,127,35,0,0                 // movaps        0x237f(%rip),%xmm3        # 51c0 <_sk_callback_sse41+0xb89>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_gather_565_sse41
@@ -23489,19 +23533,19 @@ _sk_gather_565_sse41:
   .byte  65,15,183,4,65                      // movzwl        (%r9,%rax,2),%eax
   .byte  102,15,196,192,3                    // pinsrw        $0x3,%eax,%xmm0
   .byte  102,15,56,51,208                    // pmovzxwd      %xmm0,%xmm2
-  .byte  102,15,111,5,46,35,0,0              // movdqa        0x232e(%rip),%xmm0        # 51b0 <_sk_callback_sse41+0xba3>
+  .byte  102,15,111,5,36,35,0,0              // movdqa        0x2324(%rip),%xmm0        # 51d0 <_sk_callback_sse41+0xb99>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,48,35,0,0                   // mulps         0x2330(%rip),%xmm0        # 51c0 <_sk_callback_sse41+0xbb3>
-  .byte  102,15,111,13,56,35,0,0             // movdqa        0x2338(%rip),%xmm1        # 51d0 <_sk_callback_sse41+0xbc3>
+  .byte  15,89,5,38,35,0,0                   // mulps         0x2326(%rip),%xmm0        # 51e0 <_sk_callback_sse41+0xba9>
+  .byte  102,15,111,13,46,35,0,0             // movdqa        0x232e(%rip),%xmm1        # 51f0 <_sk_callback_sse41+0xbb9>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,58,35,0,0                  // mulps         0x233a(%rip),%xmm1        # 51e0 <_sk_callback_sse41+0xbd3>
-  .byte  102,15,219,21,66,35,0,0             // pand          0x2342(%rip),%xmm2        # 51f0 <_sk_callback_sse41+0xbe3>
+  .byte  15,89,13,48,35,0,0                  // mulps         0x2330(%rip),%xmm1        # 5200 <_sk_callback_sse41+0xbc9>
+  .byte  102,15,219,21,56,35,0,0             // pand          0x2338(%rip),%xmm2        # 5210 <_sk_callback_sse41+0xbd9>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,72,35,0,0                  // mulps         0x2348(%rip),%xmm2        # 5200 <_sk_callback_sse41+0xbf3>
+  .byte  15,89,21,62,35,0,0                  // mulps         0x233e(%rip),%xmm2        # 5220 <_sk_callback_sse41+0xbe9>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,79,35,0,0                  // movaps        0x234f(%rip),%xmm3        # 5210 <_sk_callback_sse41+0xc03>
+  .byte  15,40,29,69,35,0,0                  // movaps        0x2345(%rip),%xmm3        # 5230 <_sk_callback_sse41+0xbf9>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_565_sse41
@@ -23510,12 +23554,12 @@ FUNCTION(_sk_store_565_sse41)
 _sk_store_565_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,80,35,0,0                // movaps        0x2350(%rip),%xmm8        # 5220 <_sk_callback_sse41+0xc13>
+  .byte  68,15,40,5,70,35,0,0                // movaps        0x2346(%rip),%xmm8        # 5240 <_sk_callback_sse41+0xc09>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
   .byte  102,65,15,114,241,11                // pslld         $0xb,%xmm9
-  .byte  68,15,40,21,69,35,0,0               // movaps        0x2345(%rip),%xmm10        # 5230 <_sk_callback_sse41+0xc23>
+  .byte  68,15,40,21,59,35,0,0               // movaps        0x233b(%rip),%xmm10        # 5250 <_sk_callback_sse41+0xc19>
   .byte  68,15,89,209                        // mulps         %xmm1,%xmm10
   .byte  102,69,15,91,210                    // cvtps2dq      %xmm10,%xmm10
   .byte  102,65,15,114,242,5                 // pslld         $0x5,%xmm10
@@ -23535,21 +23579,21 @@ _sk_load_4444_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  102,15,56,51,28,120                 // pmovzxwd      (%rax,%rdi,2),%xmm3
-  .byte  102,15,111,5,16,35,0,0              // movdqa        0x2310(%rip),%xmm0        # 5240 <_sk_callback_sse41+0xc33>
+  .byte  102,15,111,5,6,35,0,0               // movdqa        0x2306(%rip),%xmm0        # 5260 <_sk_callback_sse41+0xc29>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,18,35,0,0                   // mulps         0x2312(%rip),%xmm0        # 5250 <_sk_callback_sse41+0xc43>
-  .byte  102,15,111,13,26,35,0,0             // movdqa        0x231a(%rip),%xmm1        # 5260 <_sk_callback_sse41+0xc53>
+  .byte  15,89,5,8,35,0,0                    // mulps         0x2308(%rip),%xmm0        # 5270 <_sk_callback_sse41+0xc39>
+  .byte  102,15,111,13,16,35,0,0             // movdqa        0x2310(%rip),%xmm1        # 5280 <_sk_callback_sse41+0xc49>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,28,35,0,0                  // mulps         0x231c(%rip),%xmm1        # 5270 <_sk_callback_sse41+0xc63>
-  .byte  102,15,111,21,36,35,0,0             // movdqa        0x2324(%rip),%xmm2        # 5280 <_sk_callback_sse41+0xc73>
+  .byte  15,89,13,18,35,0,0                  // mulps         0x2312(%rip),%xmm1        # 5290 <_sk_callback_sse41+0xc59>
+  .byte  102,15,111,21,26,35,0,0             // movdqa        0x231a(%rip),%xmm2        # 52a0 <_sk_callback_sse41+0xc69>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,38,35,0,0                  // mulps         0x2326(%rip),%xmm2        # 5290 <_sk_callback_sse41+0xc83>
-  .byte  102,15,219,29,46,35,0,0             // pand          0x232e(%rip),%xmm3        # 52a0 <_sk_callback_sse41+0xc93>
+  .byte  15,89,21,28,35,0,0                  // mulps         0x231c(%rip),%xmm2        # 52b0 <_sk_callback_sse41+0xc79>
+  .byte  102,15,219,29,36,35,0,0             // pand          0x2324(%rip),%xmm3        # 52c0 <_sk_callback_sse41+0xc89>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,52,35,0,0                  // mulps         0x2334(%rip),%xmm3        # 52b0 <_sk_callback_sse41+0xca3>
+  .byte  15,89,29,42,35,0,0                  // mulps         0x232a(%rip),%xmm3        # 52d0 <_sk_callback_sse41+0xc99>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -23578,21 +23622,21 @@ _sk_gather_4444_sse41:
   .byte  65,15,183,4,65                      // movzwl        (%r9,%rax,2),%eax
   .byte  102,15,196,192,3                    // pinsrw        $0x3,%eax,%xmm0
   .byte  102,15,56,51,216                    // pmovzxwd      %xmm0,%xmm3
-  .byte  102,15,111,5,215,34,0,0             // movdqa        0x22d7(%rip),%xmm0        # 52c0 <_sk_callback_sse41+0xcb3>
+  .byte  102,15,111,5,205,34,0,0             // movdqa        0x22cd(%rip),%xmm0        # 52e0 <_sk_callback_sse41+0xca9>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,217,34,0,0                  // mulps         0x22d9(%rip),%xmm0        # 52d0 <_sk_callback_sse41+0xcc3>
-  .byte  102,15,111,13,225,34,0,0            // movdqa        0x22e1(%rip),%xmm1        # 52e0 <_sk_callback_sse41+0xcd3>
+  .byte  15,89,5,207,34,0,0                  // mulps         0x22cf(%rip),%xmm0        # 52f0 <_sk_callback_sse41+0xcb9>
+  .byte  102,15,111,13,215,34,0,0            // movdqa        0x22d7(%rip),%xmm1        # 5300 <_sk_callback_sse41+0xcc9>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,227,34,0,0                 // mulps         0x22e3(%rip),%xmm1        # 52f0 <_sk_callback_sse41+0xce3>
-  .byte  102,15,111,21,235,34,0,0            // movdqa        0x22eb(%rip),%xmm2        # 5300 <_sk_callback_sse41+0xcf3>
+  .byte  15,89,13,217,34,0,0                 // mulps         0x22d9(%rip),%xmm1        # 5310 <_sk_callback_sse41+0xcd9>
+  .byte  102,15,111,21,225,34,0,0            // movdqa        0x22e1(%rip),%xmm2        # 5320 <_sk_callback_sse41+0xce9>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,237,34,0,0                 // mulps         0x22ed(%rip),%xmm2        # 5310 <_sk_callback_sse41+0xd03>
-  .byte  102,15,219,29,245,34,0,0            // pand          0x22f5(%rip),%xmm3        # 5320 <_sk_callback_sse41+0xd13>
+  .byte  15,89,21,227,34,0,0                 // mulps         0x22e3(%rip),%xmm2        # 5330 <_sk_callback_sse41+0xcf9>
+  .byte  102,15,219,29,235,34,0,0            // pand          0x22eb(%rip),%xmm3        # 5340 <_sk_callback_sse41+0xd09>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,251,34,0,0                 // mulps         0x22fb(%rip),%xmm3        # 5330 <_sk_callback_sse41+0xd23>
+  .byte  15,89,29,241,34,0,0                 // mulps         0x22f1(%rip),%xmm3        # 5350 <_sk_callback_sse41+0xd19>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -23602,7 +23646,7 @@ FUNCTION(_sk_store_4444_sse41)
 _sk_store_4444_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,250,34,0,0               // movaps        0x22fa(%rip),%xmm8        # 5340 <_sk_callback_sse41+0xd33>
+  .byte  68,15,40,5,240,34,0,0               // movaps        0x22f0(%rip),%xmm8        # 5360 <_sk_callback_sse41+0xd29>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -23632,17 +23676,17 @@ _sk_load_8888_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  15,16,28,184                        // movups        (%rax,%rdi,4),%xmm3
-  .byte  15,40,5,153,34,0,0                  // movaps        0x2299(%rip),%xmm0        # 5350 <_sk_callback_sse41+0xd43>
+  .byte  15,40,5,143,34,0,0                  // movaps        0x228f(%rip),%xmm0        # 5370 <_sk_callback_sse41+0xd39>
   .byte  15,84,195                           // andps         %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,155,34,0,0               // movaps        0x229b(%rip),%xmm8        # 5360 <_sk_callback_sse41+0xd53>
+  .byte  68,15,40,5,145,34,0,0               // movaps        0x2291(%rip),%xmm8        # 5380 <_sk_callback_sse41+0xd49>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,40,203                           // movaps        %xmm3,%xmm1
-  .byte  102,15,56,0,13,155,34,0,0           // pshufb        0x229b(%rip),%xmm1        # 5370 <_sk_callback_sse41+0xd63>
+  .byte  102,15,56,0,13,145,34,0,0           // pshufb        0x2291(%rip),%xmm1        # 5390 <_sk_callback_sse41+0xd59>
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  15,40,211                           // movaps        %xmm3,%xmm2
-  .byte  102,15,56,0,21,152,34,0,0           // pshufb        0x2298(%rip),%xmm2        # 5380 <_sk_callback_sse41+0xd73>
+  .byte  102,15,56,0,21,142,34,0,0           // pshufb        0x228e(%rip),%xmm2        # 53a0 <_sk_callback_sse41+0xd69>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  102,15,114,211,24                   // psrld         $0x18,%xmm3
@@ -23673,17 +23717,17 @@ _sk_gather_8888_sse41:
   .byte  102,65,15,58,34,28,129,1            // pinsrd        $0x1,(%r9,%rax,4),%xmm3
   .byte  102,67,15,58,34,28,145,2            // pinsrd        $0x2,(%r9,%r10,4),%xmm3
   .byte  102,65,15,58,34,28,137,3            // pinsrd        $0x3,(%r9,%rcx,4),%xmm3
-  .byte  102,15,111,5,49,34,0,0              // movdqa        0x2231(%rip),%xmm0        # 5390 <_sk_callback_sse41+0xd83>
+  .byte  102,15,111,5,39,34,0,0              // movdqa        0x2227(%rip),%xmm0        # 53b0 <_sk_callback_sse41+0xd79>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,50,34,0,0                // movaps        0x2232(%rip),%xmm8        # 53a0 <_sk_callback_sse41+0xd93>
+  .byte  68,15,40,5,40,34,0,0                // movaps        0x2228(%rip),%xmm8        # 53c0 <_sk_callback_sse41+0xd89>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
-  .byte  102,15,56,0,13,49,34,0,0            // pshufb        0x2231(%rip),%xmm1        # 53b0 <_sk_callback_sse41+0xda3>
+  .byte  102,15,56,0,13,39,34,0,0            // pshufb        0x2227(%rip),%xmm1        # 53d0 <_sk_callback_sse41+0xd99>
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,111,211                      // movdqa        %xmm3,%xmm2
-  .byte  102,15,56,0,21,45,34,0,0            // pshufb        0x222d(%rip),%xmm2        # 53c0 <_sk_callback_sse41+0xdb3>
+  .byte  102,15,56,0,21,35,34,0,0            // pshufb        0x2223(%rip),%xmm2        # 53e0 <_sk_callback_sse41+0xda9>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  102,15,114,211,24                   // psrld         $0x18,%xmm3
@@ -23698,7 +23742,7 @@ FUNCTION(_sk_store_8888_sse41)
 _sk_store_8888_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,25,34,0,0                // movaps        0x2219(%rip),%xmm8        # 53d0 <_sk_callback_sse41+0xdc3>
+  .byte  68,15,40,5,15,34,0,0                // movaps        0x220f(%rip),%xmm8        # 53f0 <_sk_callback_sse41+0xdb9>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -23735,18 +23779,18 @@ _sk_load_f16_sse41:
   .byte  102,68,15,97,216                    // punpcklwd     %xmm0,%xmm11
   .byte  102,68,15,105,200                   // punpckhwd     %xmm0,%xmm9
   .byte  102,65,15,56,51,203                 // pmovzxwd      %xmm11,%xmm1
-  .byte  102,68,15,111,5,146,33,0,0          // movdqa        0x2192(%rip),%xmm8        # 53e0 <_sk_callback_sse41+0xdd3>
+  .byte  102,68,15,111,5,136,33,0,0          // movdqa        0x2188(%rip),%xmm8        # 5400 <_sk_callback_sse41+0xdc9>
   .byte  102,15,111,209                      // movdqa        %xmm1,%xmm2
   .byte  102,65,15,219,208                   // pand          %xmm8,%xmm2
   .byte  102,15,239,202                      // pxor          %xmm2,%xmm1
-  .byte  102,15,111,29,141,33,0,0            // movdqa        0x218d(%rip),%xmm3        # 53f0 <_sk_callback_sse41+0xde3>
+  .byte  102,15,111,29,131,33,0,0            // movdqa        0x2183(%rip),%xmm3        # 5410 <_sk_callback_sse41+0xdd9>
   .byte  102,15,114,242,16                   // pslld         $0x10,%xmm2
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,15,56,63,195                    // pmaxud        %xmm3,%xmm0
   .byte  102,15,118,193                      // pcmpeqd       %xmm1,%xmm0
   .byte  102,15,114,241,13                   // pslld         $0xd,%xmm1
   .byte  102,15,235,202                      // por           %xmm2,%xmm1
-  .byte  102,68,15,111,21,121,33,0,0         // movdqa        0x2179(%rip),%xmm10        # 5400 <_sk_callback_sse41+0xdf3>
+  .byte  102,68,15,111,21,111,33,0,0         // movdqa        0x216f(%rip),%xmm10        # 5420 <_sk_callback_sse41+0xde9>
   .byte  102,65,15,254,202                   // paddd         %xmm10,%xmm1
   .byte  102,15,219,193                      // pand          %xmm1,%xmm0
   .byte  102,65,15,115,219,8                 // psrldq        $0x8,%xmm11
@@ -23819,18 +23863,18 @@ _sk_gather_f16_sse41:
   .byte  102,68,15,97,218                    // punpcklwd     %xmm2,%xmm11
   .byte  102,68,15,105,202                   // punpckhwd     %xmm2,%xmm9
   .byte  102,65,15,56,51,203                 // pmovzxwd      %xmm11,%xmm1
-  .byte  102,68,15,111,5,55,32,0,0           // movdqa        0x2037(%rip),%xmm8        # 5410 <_sk_callback_sse41+0xe03>
+  .byte  102,68,15,111,5,45,32,0,0           // movdqa        0x202d(%rip),%xmm8        # 5430 <_sk_callback_sse41+0xdf9>
   .byte  102,15,111,209                      // movdqa        %xmm1,%xmm2
   .byte  102,65,15,219,208                   // pand          %xmm8,%xmm2
   .byte  102,15,239,202                      // pxor          %xmm2,%xmm1
-  .byte  102,15,111,29,50,32,0,0             // movdqa        0x2032(%rip),%xmm3        # 5420 <_sk_callback_sse41+0xe13>
+  .byte  102,15,111,29,40,32,0,0             // movdqa        0x2028(%rip),%xmm3        # 5440 <_sk_callback_sse41+0xe09>
   .byte  102,15,114,242,16                   // pslld         $0x10,%xmm2
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,15,56,63,195                    // pmaxud        %xmm3,%xmm0
   .byte  102,15,118,193                      // pcmpeqd       %xmm1,%xmm0
   .byte  102,15,114,241,13                   // pslld         $0xd,%xmm1
   .byte  102,15,235,202                      // por           %xmm2,%xmm1
-  .byte  102,68,15,111,21,30,32,0,0          // movdqa        0x201e(%rip),%xmm10        # 5430 <_sk_callback_sse41+0xe23>
+  .byte  102,68,15,111,21,20,32,0,0          // movdqa        0x2014(%rip),%xmm10        # 5450 <_sk_callback_sse41+0xe19>
   .byte  102,65,15,254,202                   // paddd         %xmm10,%xmm1
   .byte  102,15,219,193                      // pand          %xmm1,%xmm0
   .byte  102,65,15,115,219,8                 // psrldq        $0x8,%xmm11
@@ -23878,17 +23922,17 @@ FUNCTION(_sk_store_f16_sse41)
 _sk_store_f16_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  102,68,15,111,21,84,31,0,0          // movdqa        0x1f54(%rip),%xmm10        # 5440 <_sk_callback_sse41+0xe33>
+  .byte  102,68,15,111,21,74,31,0,0          // movdqa        0x1f4a(%rip),%xmm10        # 5460 <_sk_callback_sse41+0xe29>
   .byte  102,68,15,111,224                   // movdqa        %xmm0,%xmm12
   .byte  102,68,15,111,232                   // movdqa        %xmm0,%xmm13
   .byte  102,69,15,219,234                   // pand          %xmm10,%xmm13
   .byte  102,69,15,239,229                   // pxor          %xmm13,%xmm12
-  .byte  102,68,15,111,13,71,31,0,0          // movdqa        0x1f47(%rip),%xmm9        # 5450 <_sk_callback_sse41+0xe43>
+  .byte  102,68,15,111,13,61,31,0,0          // movdqa        0x1f3d(%rip),%xmm9        # 5470 <_sk_callback_sse41+0xe39>
   .byte  102,65,15,114,213,16                // psrld         $0x10,%xmm13
   .byte  102,69,15,111,193                   // movdqa        %xmm9,%xmm8
   .byte  102,69,15,102,196                   // pcmpgtd       %xmm12,%xmm8
   .byte  102,65,15,114,212,13                // psrld         $0xd,%xmm12
-  .byte  102,68,15,111,29,56,31,0,0          // movdqa        0x1f38(%rip),%xmm11        # 5460 <_sk_callback_sse41+0xe53>
+  .byte  102,68,15,111,29,46,31,0,0          // movdqa        0x1f2e(%rip),%xmm11        # 5480 <_sk_callback_sse41+0xe49>
   .byte  102,69,15,235,235                   // por           %xmm11,%xmm13
   .byte  102,69,15,254,236                   // paddd         %xmm12,%xmm13
   .byte  102,69,15,223,197                   // pandn         %xmm13,%xmm8
@@ -23958,7 +24002,7 @@ _sk_load_u16_be_sse41:
   .byte  102,15,235,200                      // por           %xmm0,%xmm1
   .byte  102,15,56,51,193                    // pmovzxwd      %xmm1,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,7,30,0,0                 // movaps        0x1e07(%rip),%xmm8        # 5470 <_sk_callback_sse41+0xe63>
+  .byte  68,15,40,5,253,29,0,0               // movaps        0x1dfd(%rip),%xmm8        # 5490 <_sk_callback_sse41+0xe59>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -24010,7 +24054,7 @@ _sk_load_rgb_u16_be_sse41:
   .byte  102,15,235,193                      // por           %xmm1,%xmm0
   .byte  102,15,56,51,192                    // pmovzxwd      %xmm0,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,72,29,0,0                // movaps        0x1d48(%rip),%xmm8        # 5480 <_sk_callback_sse41+0xe73>
+  .byte  68,15,40,5,62,29,0,0                // movaps        0x1d3e(%rip),%xmm8        # 54a0 <_sk_callback_sse41+0xe69>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -24027,7 +24071,7 @@ _sk_load_rgb_u16_be_sse41:
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,15,29,0,0                  // movaps        0x1d0f(%rip),%xmm3        # 5490 <_sk_callback_sse41+0xe83>
+  .byte  15,40,29,5,29,0,0                   // movaps        0x1d05(%rip),%xmm3        # 54b0 <_sk_callback_sse41+0xe79>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_u16_be_sse41
@@ -24036,7 +24080,7 @@ FUNCTION(_sk_store_u16_be_sse41)
 _sk_store_u16_be_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,13,16,29,0,0               // movaps        0x1d10(%rip),%xmm9        # 54a0 <_sk_callback_sse41+0xe93>
+  .byte  68,15,40,13,6,29,0,0                // movaps        0x1d06(%rip),%xmm9        # 54c0 <_sk_callback_sse41+0xe89>
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
@@ -24247,10 +24291,10 @@ HIDDEN _sk_luminance_to_alpha_sse41
 FUNCTION(_sk_luminance_to_alpha_sse41)
 _sk_luminance_to_alpha_sse41:
   .byte  15,40,218                           // movaps        %xmm2,%xmm3
-  .byte  15,89,5,108,26,0,0                  // mulps         0x1a6c(%rip),%xmm0        # 54b0 <_sk_callback_sse41+0xea3>
-  .byte  15,89,13,117,26,0,0                 // mulps         0x1a75(%rip),%xmm1        # 54c0 <_sk_callback_sse41+0xeb3>
+  .byte  15,89,5,98,26,0,0                   // mulps         0x1a62(%rip),%xmm0        # 54d0 <_sk_callback_sse41+0xe99>
+  .byte  15,89,13,107,26,0,0                 // mulps         0x1a6b(%rip),%xmm1        # 54e0 <_sk_callback_sse41+0xea9>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  15,89,29,123,26,0,0                 // mulps         0x1a7b(%rip),%xmm3        # 54d0 <_sk_callback_sse41+0xec3>
+  .byte  15,89,29,113,26,0,0                 // mulps         0x1a71(%rip),%xmm3        # 54f0 <_sk_callback_sse41+0xeb9>
   .byte  15,88,217                           // addps         %xmm1,%xmm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
@@ -24476,9 +24520,9 @@ _sk_evenly_spaced_gradient_sse41:
   .byte  72,139,8                            // mov           (%rax),%rcx
   .byte  76,139,88,8                         // mov           0x8(%rax),%r11
   .byte  72,255,201                          // dec           %rcx
-  .byte  120,7                               // js            3dd4 <_sk_evenly_spaced_gradient_sse41+0x15>
+  .byte  120,7                               // js            3dfe <_sk_evenly_spaced_gradient_sse41+0x15>
   .byte  243,72,15,42,201                    // cvtsi2ss      %rcx,%xmm1
-  .byte  235,21                              // jmp           3de9 <_sk_evenly_spaced_gradient_sse41+0x2a>
+  .byte  235,21                              // jmp           3e13 <_sk_evenly_spaced_gradient_sse41+0x2a>
   .byte  73,137,200                          // mov           %rcx,%r8
   .byte  73,209,232                          // shr           %r8
   .byte  131,225,1                           // and           $0x1,%ecx
@@ -24569,12 +24613,12 @@ _sk_gradient_sse41:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
   .byte  73,131,248,2                        // cmp           $0x2,%r8
-  .byte  114,50                              // jb            3fcc <_sk_gradient_sse41+0x41>
+  .byte  114,50                              // jb            3ff6 <_sk_gradient_sse41+0x41>
   .byte  72,139,72,72                        // mov           0x48(%rax),%rcx
   .byte  73,255,200                          // dec           %r8
   .byte  72,131,193,4                        // add           $0x4,%rcx
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
-  .byte  15,40,21,48,21,0,0                  // movaps        0x1530(%rip),%xmm2        # 54e0 <_sk_callback_sse41+0xed3>
+  .byte  15,40,21,38,21,0,0                  // movaps        0x1526(%rip),%xmm2        # 5500 <_sk_callback_sse41+0xec9>
   .byte  243,15,16,25                        // movss         (%rcx),%xmm3
   .byte  15,198,219,0                        // shufps        $0x0,%xmm3,%xmm3
   .byte  15,194,216,2                        // cmpleps       %xmm0,%xmm3
@@ -24582,7 +24626,7 @@ _sk_gradient_sse41:
   .byte  102,15,254,203                      // paddd         %xmm3,%xmm1
   .byte  72,131,193,4                        // add           $0x4,%rcx
   .byte  73,255,200                          // dec           %r8
-  .byte  117,228                             // jne           3fb0 <_sk_gradient_sse41+0x25>
+  .byte  117,228                             // jne           3fda <_sk_gradient_sse41+0x25>
   .byte  65,86                               // push          %r14
   .byte  83                                  // push          %rbx
   .byte  102,73,15,58,22,201,1               // pextrq        $0x1,%xmm1,%r9
@@ -24713,26 +24757,26 @@ _sk_xy_to_unit_angle_sse41:
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,40,236                        // movaps        %xmm12,%xmm13
   .byte  69,15,89,237                        // mulps         %xmm13,%xmm13
-  .byte  68,15,40,21,210,18,0,0              // movaps        0x12d2(%rip),%xmm10        # 54f0 <_sk_callback_sse41+0xee3>
+  .byte  68,15,40,21,200,18,0,0              // movaps        0x12c8(%rip),%xmm10        # 5510 <_sk_callback_sse41+0xed9>
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,88,21,214,18,0,0              // addps         0x12d6(%rip),%xmm10        # 5500 <_sk_callback_sse41+0xef3>
+  .byte  68,15,88,21,204,18,0,0              // addps         0x12cc(%rip),%xmm10        # 5520 <_sk_callback_sse41+0xee9>
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,88,21,218,18,0,0              // addps         0x12da(%rip),%xmm10        # 5510 <_sk_callback_sse41+0xf03>
+  .byte  68,15,88,21,208,18,0,0              // addps         0x12d0(%rip),%xmm10        # 5530 <_sk_callback_sse41+0xef9>
   .byte  69,15,89,213                        // mulps         %xmm13,%xmm10
-  .byte  68,15,88,21,222,18,0,0              // addps         0x12de(%rip),%xmm10        # 5520 <_sk_callback_sse41+0xf13>
+  .byte  68,15,88,21,212,18,0,0              // addps         0x12d4(%rip),%xmm10        # 5540 <_sk_callback_sse41+0xf09>
   .byte  69,15,89,212                        // mulps         %xmm12,%xmm10
   .byte  65,15,194,195,1                     // cmpltps       %xmm11,%xmm0
-  .byte  68,15,40,29,221,18,0,0              // movaps        0x12dd(%rip),%xmm11        # 5530 <_sk_callback_sse41+0xf23>
+  .byte  68,15,40,29,211,18,0,0              // movaps        0x12d3(%rip),%xmm11        # 5550 <_sk_callback_sse41+0xf19>
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  102,69,15,56,20,211                 // blendvps      %xmm0,%xmm11,%xmm10
   .byte  69,15,194,200,1                     // cmpltps       %xmm8,%xmm9
-  .byte  68,15,40,29,214,18,0,0              // movaps        0x12d6(%rip),%xmm11        # 5540 <_sk_callback_sse41+0xf33>
+  .byte  68,15,40,29,204,18,0,0              // movaps        0x12cc(%rip),%xmm11        # 5560 <_sk_callback_sse41+0xf29>
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  102,69,15,56,20,211                 // blendvps      %xmm0,%xmm11,%xmm10
   .byte  15,40,193                           // movaps        %xmm1,%xmm0
   .byte  65,15,194,192,1                     // cmpltps       %xmm8,%xmm0
-  .byte  68,15,40,13,200,18,0,0              // movaps        0x12c8(%rip),%xmm9        # 5550 <_sk_callback_sse41+0xf43>
+  .byte  68,15,40,13,190,18,0,0              // movaps        0x12be(%rip),%xmm9        # 5570 <_sk_callback_sse41+0xf39>
   .byte  69,15,92,202                        // subps         %xmm10,%xmm9
   .byte  102,69,15,56,20,209                 // blendvps      %xmm0,%xmm9,%xmm10
   .byte  69,15,194,194,7                     // cmpordps      %xmm10,%xmm8
@@ -24758,7 +24802,7 @@ HIDDEN _sk_save_xy_sse41
 FUNCTION(_sk_save_xy_sse41)
 _sk_save_xy_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,156,18,0,0               // movaps        0x129c(%rip),%xmm8        # 5560 <_sk_callback_sse41+0xf53>
+  .byte  68,15,40,5,146,18,0,0               // movaps        0x1292(%rip),%xmm8        # 5580 <_sk_callback_sse41+0xf49>
   .byte  15,17,0                             // movups        %xmm0,(%rax)
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,88,200                        // addps         %xmm8,%xmm9
@@ -24802,8 +24846,8 @@ _sk_bilinear_nx_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,30,18,0,0                   // addps         0x121e(%rip),%xmm0        # 5570 <_sk_callback_sse41+0xf63>
-  .byte  68,15,40,13,38,18,0,0               // movaps        0x1226(%rip),%xmm9        # 5580 <_sk_callback_sse41+0xf73>
+  .byte  15,88,5,20,18,0,0                   // addps         0x1214(%rip),%xmm0        # 5590 <_sk_callback_sse41+0xf59>
+  .byte  68,15,40,13,28,18,0,0               // movaps        0x121c(%rip),%xmm9        # 55a0 <_sk_callback_sse41+0xf69>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -24816,7 +24860,7 @@ _sk_bilinear_px_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,21,18,0,0                   // addps         0x1215(%rip),%xmm0        # 5590 <_sk_callback_sse41+0xf83>
+  .byte  15,88,5,11,18,0,0                   // addps         0x120b(%rip),%xmm0        # 55b0 <_sk_callback_sse41+0xf79>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24828,8 +24872,8 @@ _sk_bilinear_ny_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,7,18,0,0                   // addps         0x1207(%rip),%xmm1        # 55a0 <_sk_callback_sse41+0xf93>
-  .byte  68,15,40,13,15,18,0,0               // movaps        0x120f(%rip),%xmm9        # 55b0 <_sk_callback_sse41+0xfa3>
+  .byte  15,88,13,253,17,0,0                 // addps         0x11fd(%rip),%xmm1        # 55c0 <_sk_callback_sse41+0xf89>
+  .byte  68,15,40,13,5,18,0,0                // movaps        0x1205(%rip),%xmm9        # 55d0 <_sk_callback_sse41+0xf99>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -24842,7 +24886,7 @@ _sk_bilinear_py_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,253,17,0,0                 // addps         0x11fd(%rip),%xmm1        # 55c0 <_sk_callback_sse41+0xfb3>
+  .byte  15,88,13,243,17,0,0                 // addps         0x11f3(%rip),%xmm1        # 55e0 <_sk_callback_sse41+0xfa9>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24854,13 +24898,13 @@ _sk_bicubic_n3x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,240,17,0,0                  // addps         0x11f0(%rip),%xmm0        # 55d0 <_sk_callback_sse41+0xfc3>
-  .byte  68,15,40,13,248,17,0,0              // movaps        0x11f8(%rip),%xmm9        # 55e0 <_sk_callback_sse41+0xfd3>
+  .byte  15,88,5,230,17,0,0                  // addps         0x11e6(%rip),%xmm0        # 55f0 <_sk_callback_sse41+0xfb9>
+  .byte  68,15,40,13,238,17,0,0              // movaps        0x11ee(%rip),%xmm9        # 5600 <_sk_callback_sse41+0xfc9>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,244,17,0,0              // mulps         0x11f4(%rip),%xmm9        # 55f0 <_sk_callback_sse41+0xfe3>
-  .byte  68,15,88,13,252,17,0,0              // addps         0x11fc(%rip),%xmm9        # 5600 <_sk_callback_sse41+0xff3>
+  .byte  68,15,89,13,234,17,0,0              // mulps         0x11ea(%rip),%xmm9        # 5610 <_sk_callback_sse41+0xfd9>
+  .byte  68,15,88,13,242,17,0,0              // addps         0x11f2(%rip),%xmm9        # 5620 <_sk_callback_sse41+0xfe9>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -24873,16 +24917,16 @@ _sk_bicubic_n1x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,235,17,0,0                  // addps         0x11eb(%rip),%xmm0        # 5610 <_sk_callback_sse41+0x1003>
-  .byte  68,15,40,13,243,17,0,0              // movaps        0x11f3(%rip),%xmm9        # 5620 <_sk_callback_sse41+0x1013>
+  .byte  15,88,5,225,17,0,0                  // addps         0x11e1(%rip),%xmm0        # 5630 <_sk_callback_sse41+0xff9>
+  .byte  68,15,40,13,233,17,0,0              // movaps        0x11e9(%rip),%xmm9        # 5640 <_sk_callback_sse41+0x1009>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,247,17,0,0               // movaps        0x11f7(%rip),%xmm8        # 5630 <_sk_callback_sse41+0x1023>
+  .byte  68,15,40,5,237,17,0,0               // movaps        0x11ed(%rip),%xmm8        # 5650 <_sk_callback_sse41+0x1019>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,251,17,0,0               // addps         0x11fb(%rip),%xmm8        # 5640 <_sk_callback_sse41+0x1033>
+  .byte  68,15,88,5,241,17,0,0               // addps         0x11f1(%rip),%xmm8        # 5660 <_sk_callback_sse41+0x1029>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,255,17,0,0               // addps         0x11ff(%rip),%xmm8        # 5650 <_sk_callback_sse41+0x1043>
+  .byte  68,15,88,5,245,17,0,0               // addps         0x11f5(%rip),%xmm8        # 5670 <_sk_callback_sse41+0x1039>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,3,18,0,0                 // addps         0x1203(%rip),%xmm8        # 5660 <_sk_callback_sse41+0x1053>
+  .byte  68,15,88,5,249,17,0,0               // addps         0x11f9(%rip),%xmm8        # 5680 <_sk_callback_sse41+0x1049>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24892,17 +24936,17 @@ HIDDEN _sk_bicubic_p1x_sse41
 FUNCTION(_sk_bicubic_p1x_sse41)
 _sk_bicubic_p1x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,253,17,0,0               // movaps        0x11fd(%rip),%xmm8        # 5670 <_sk_callback_sse41+0x1063>
+  .byte  68,15,40,5,243,17,0,0               // movaps        0x11f3(%rip),%xmm8        # 5690 <_sk_callback_sse41+0x1059>
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,72,64                      // movups        0x40(%rax),%xmm9
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
-  .byte  68,15,40,21,249,17,0,0              // movaps        0x11f9(%rip),%xmm10        # 5680 <_sk_callback_sse41+0x1073>
+  .byte  68,15,40,21,239,17,0,0              // movaps        0x11ef(%rip),%xmm10        # 56a0 <_sk_callback_sse41+0x1069>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,253,17,0,0              // addps         0x11fd(%rip),%xmm10        # 5690 <_sk_callback_sse41+0x1083>
+  .byte  68,15,88,21,243,17,0,0              // addps         0x11f3(%rip),%xmm10        # 56b0 <_sk_callback_sse41+0x1079>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,249,17,0,0              // addps         0x11f9(%rip),%xmm10        # 56a0 <_sk_callback_sse41+0x1093>
+  .byte  68,15,88,21,239,17,0,0              // addps         0x11ef(%rip),%xmm10        # 56c0 <_sk_callback_sse41+0x1089>
   .byte  68,15,17,144,128,0,0,0              // movups        %xmm10,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24914,11 +24958,11 @@ _sk_bicubic_p3x_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,236,17,0,0                  // addps         0x11ec(%rip),%xmm0        # 56b0 <_sk_callback_sse41+0x10a3>
+  .byte  15,88,5,226,17,0,0                  // addps         0x11e2(%rip),%xmm0        # 56d0 <_sk_callback_sse41+0x1099>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,236,17,0,0               // mulps         0x11ec(%rip),%xmm8        # 56c0 <_sk_callback_sse41+0x10b3>
-  .byte  68,15,88,5,244,17,0,0               // addps         0x11f4(%rip),%xmm8        # 56d0 <_sk_callback_sse41+0x10c3>
+  .byte  68,15,89,5,226,17,0,0               // mulps         0x11e2(%rip),%xmm8        # 56e0 <_sk_callback_sse41+0x10a9>
+  .byte  68,15,88,5,234,17,0,0               // addps         0x11ea(%rip),%xmm8        # 56f0 <_sk_callback_sse41+0x10b9>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -24931,13 +24975,13 @@ _sk_bicubic_n3y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,226,17,0,0                 // addps         0x11e2(%rip),%xmm1        # 56e0 <_sk_callback_sse41+0x10d3>
-  .byte  68,15,40,13,234,17,0,0              // movaps        0x11ea(%rip),%xmm9        # 56f0 <_sk_callback_sse41+0x10e3>
+  .byte  15,88,13,216,17,0,0                 // addps         0x11d8(%rip),%xmm1        # 5700 <_sk_callback_sse41+0x10c9>
+  .byte  68,15,40,13,224,17,0,0              // movaps        0x11e0(%rip),%xmm9        # 5710 <_sk_callback_sse41+0x10d9>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,230,17,0,0              // mulps         0x11e6(%rip),%xmm9        # 5700 <_sk_callback_sse41+0x10f3>
-  .byte  68,15,88,13,238,17,0,0              // addps         0x11ee(%rip),%xmm9        # 5710 <_sk_callback_sse41+0x1103>
+  .byte  68,15,89,13,220,17,0,0              // mulps         0x11dc(%rip),%xmm9        # 5720 <_sk_callback_sse41+0x10e9>
+  .byte  68,15,88,13,228,17,0,0              // addps         0x11e4(%rip),%xmm9        # 5730 <_sk_callback_sse41+0x10f9>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -24950,16 +24994,16 @@ _sk_bicubic_n1y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,220,17,0,0                 // addps         0x11dc(%rip),%xmm1        # 5720 <_sk_callback_sse41+0x1113>
-  .byte  68,15,40,13,228,17,0,0              // movaps        0x11e4(%rip),%xmm9        # 5730 <_sk_callback_sse41+0x1123>
+  .byte  15,88,13,210,17,0,0                 // addps         0x11d2(%rip),%xmm1        # 5740 <_sk_callback_sse41+0x1109>
+  .byte  68,15,40,13,218,17,0,0              // movaps        0x11da(%rip),%xmm9        # 5750 <_sk_callback_sse41+0x1119>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,232,17,0,0               // movaps        0x11e8(%rip),%xmm8        # 5740 <_sk_callback_sse41+0x1133>
+  .byte  68,15,40,5,222,17,0,0               // movaps        0x11de(%rip),%xmm8        # 5760 <_sk_callback_sse41+0x1129>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,236,17,0,0               // addps         0x11ec(%rip),%xmm8        # 5750 <_sk_callback_sse41+0x1143>
+  .byte  68,15,88,5,226,17,0,0               // addps         0x11e2(%rip),%xmm8        # 5770 <_sk_callback_sse41+0x1139>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,240,17,0,0               // addps         0x11f0(%rip),%xmm8        # 5760 <_sk_callback_sse41+0x1153>
+  .byte  68,15,88,5,230,17,0,0               // addps         0x11e6(%rip),%xmm8        # 5780 <_sk_callback_sse41+0x1149>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,244,17,0,0               // addps         0x11f4(%rip),%xmm8        # 5770 <_sk_callback_sse41+0x1163>
+  .byte  68,15,88,5,234,17,0,0               // addps         0x11ea(%rip),%xmm8        # 5790 <_sk_callback_sse41+0x1159>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24969,17 +25013,17 @@ HIDDEN _sk_bicubic_p1y_sse41
 FUNCTION(_sk_bicubic_p1y_sse41)
 _sk_bicubic_p1y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,238,17,0,0               // movaps        0x11ee(%rip),%xmm8        # 5780 <_sk_callback_sse41+0x1173>
+  .byte  68,15,40,5,228,17,0,0               // movaps        0x11e4(%rip),%xmm8        # 57a0 <_sk_callback_sse41+0x1169>
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,72,96                      // movups        0x60(%rax),%xmm9
   .byte  65,15,88,200                        // addps         %xmm8,%xmm1
-  .byte  68,15,40,21,233,17,0,0              // movaps        0x11e9(%rip),%xmm10        # 5790 <_sk_callback_sse41+0x1183>
+  .byte  68,15,40,21,223,17,0,0              // movaps        0x11df(%rip),%xmm10        # 57b0 <_sk_callback_sse41+0x1179>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,237,17,0,0              // addps         0x11ed(%rip),%xmm10        # 57a0 <_sk_callback_sse41+0x1193>
+  .byte  68,15,88,21,227,17,0,0              // addps         0x11e3(%rip),%xmm10        # 57c0 <_sk_callback_sse41+0x1189>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,233,17,0,0              // addps         0x11e9(%rip),%xmm10        # 57b0 <_sk_callback_sse41+0x11a3>
+  .byte  68,15,88,21,223,17,0,0              // addps         0x11df(%rip),%xmm10        # 57d0 <_sk_callback_sse41+0x1199>
   .byte  68,15,17,144,160,0,0,0              // movups        %xmm10,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -24991,11 +25035,11 @@ _sk_bicubic_p3y_sse41:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,219,17,0,0                 // addps         0x11db(%rip),%xmm1        # 57c0 <_sk_callback_sse41+0x11b3>
+  .byte  15,88,13,209,17,0,0                 // addps         0x11d1(%rip),%xmm1        # 57e0 <_sk_callback_sse41+0x11a9>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,219,17,0,0               // mulps         0x11db(%rip),%xmm8        # 57d0 <_sk_callback_sse41+0x11c3>
-  .byte  68,15,88,5,227,17,0,0               // addps         0x11e3(%rip),%xmm8        # 57e0 <_sk_callback_sse41+0x11d3>
+  .byte  68,15,89,5,209,17,0,0               // mulps         0x11d1(%rip),%xmm8        # 57f0 <_sk_callback_sse41+0x11b9>
+  .byte  68,15,88,5,217,17,0,0               // addps         0x11d9(%rip),%xmm8        # 5800 <_sk_callback_sse41+0x11c9>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -25214,11 +25258,11 @@ BALIGN16
   .byte  128,191,0,0,128,191,0               // cmpb          $0x0,-0x40800000(%rdi)
   .byte  0,224                               // add           %ah,%al
   .byte  64,0,0                              // add           %al,(%rax)
-  .byte  224,64                              // loopne        48c8 <.literal16+0x1d8>
+  .byte  224,64                              // loopne        48e8 <.literal16+0x1d8>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        48cc <.literal16+0x1dc>
+  .byte  224,64                              // loopne        48ec <.literal16+0x1dc>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        48d0 <.literal16+0x1e0>
+  .byte  224,64                              // loopne        48f0 <.literal16+0x1e0>
   .byte  154                                 // (bad)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
@@ -25238,13 +25282,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 48f1 <.literal16+0x201>
+  .byte  71,225,61                           // rex.RXB       loope 4911 <.literal16+0x201>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 48f5 <.literal16+0x205>
+  .byte  71,225,61                           // rex.RXB       loope 4915 <.literal16+0x205>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 48f9 <.literal16+0x209>
+  .byte  71,225,61                           // rex.RXB       loope 4919 <.literal16+0x209>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 48fd <.literal16+0x20d>
+  .byte  71,225,61                           // rex.RXB       loope 491d <.literal16+0x20d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -25269,13 +25313,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4931 <.literal16+0x241>
+  .byte  71,225,61                           // rex.RXB       loope 4951 <.literal16+0x241>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4935 <.literal16+0x245>
+  .byte  71,225,61                           // rex.RXB       loope 4955 <.literal16+0x245>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4939 <.literal16+0x249>
+  .byte  71,225,61                           // rex.RXB       loope 4959 <.literal16+0x249>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 493d <.literal16+0x24d>
+  .byte  71,225,61                           // rex.RXB       loope 495d <.literal16+0x24d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -25300,13 +25344,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4971 <.literal16+0x281>
+  .byte  71,225,61                           // rex.RXB       loope 4991 <.literal16+0x281>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4975 <.literal16+0x285>
+  .byte  71,225,61                           // rex.RXB       loope 4995 <.literal16+0x285>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4979 <.literal16+0x289>
+  .byte  71,225,61                           // rex.RXB       loope 4999 <.literal16+0x289>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 497d <.literal16+0x28d>
+  .byte  71,225,61                           // rex.RXB       loope 499d <.literal16+0x28d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -25331,13 +25375,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 49b1 <.literal16+0x2c1>
+  .byte  71,225,61                           // rex.RXB       loope 49d1 <.literal16+0x2c1>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 49b5 <.literal16+0x2c5>
+  .byte  71,225,61                           // rex.RXB       loope 49d5 <.literal16+0x2c5>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 49b9 <.literal16+0x2c9>
+  .byte  71,225,61                           // rex.RXB       loope 49d9 <.literal16+0x2c9>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 49bd <.literal16+0x2cd>
+  .byte  71,225,61                           // rex.RXB       loope 49dd <.literal16+0x2cd>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -25561,13 +25605,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        4b89 <.literal16+0x499>
+  .byte  224,7                               // loopne        4ba9 <.literal16+0x499>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4b8d <.literal16+0x49d>
+  .byte  224,7                               // loopne        4bad <.literal16+0x49d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4b91 <.literal16+0x4a1>
+  .byte  224,7                               // loopne        4bb1 <.literal16+0x4a1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        4b95 <.literal16+0x4a5>
+  .byte  224,7                               // loopne        4bb5 <.literal16+0x4a5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -25601,10 +25645,10 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  1,255                               // add           %edi,%edi
   .byte  255                                 // (bad)
-  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004bd8 <_sk_callback_sse41+0xa0005cb>
+  .byte  255,5,255,255,255,9                 // incl          0x9ffffff(%rip)        # a004bf8 <_sk_callback_sse41+0xa0005c1>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004be0 <_sk_callback_sse41+0x30005d3>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3004c00 <_sk_callback_sse41+0x30005c9>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -25659,11 +25703,11 @@ BALIGN16
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4cab <.literal16+0x5bb>
+  .byte  127,67                              // jg            4ccb <.literal16+0x5bb>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4caf <.literal16+0x5bf>
+  .byte  127,67                              // jg            4ccf <.literal16+0x5bf>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            4cb3 <.literal16+0x5c3>
+  .byte  127,67                              // jg            4cd3 <.literal16+0x5c3>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,129,128,128,59           // addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -25678,16 +25722,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4ca4 <.literal16+0x5b4>
+  .byte  127,0                               // jg            4cc4 <.literal16+0x5b4>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4ca8 <.literal16+0x5b8>
+  .byte  127,0                               // jg            4cc8 <.literal16+0x5b8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4cac <.literal16+0x5bc>
+  .byte  127,0                               // jg            4ccc <.literal16+0x5bc>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4cb0 <.literal16+0x5c0>
+  .byte  127,0                               // jg            4cd0 <.literal16+0x5c0>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -25696,7 +25740,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4d35 <.literal16+0x645>
+  .byte  119,115                             // ja            4d55 <.literal16+0x645>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -25707,7 +25751,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4c99 <.literal16+0x5a9>
+  .byte  117,191                             // jne           4cb9 <.literal16+0x5a9>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -25719,7 +25763,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38cda <_sk_callback_sse41+0xffffffffe9a346cd>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38cfa <_sk_callback_sse41+0xffffffffe9a346c3>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -25774,16 +25818,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4d74 <.literal16+0x684>
+  .byte  127,0                               // jg            4d94 <.literal16+0x684>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4d78 <.literal16+0x688>
+  .byte  127,0                               // jg            4d98 <.literal16+0x688>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4d7c <.literal16+0x68c>
+  .byte  127,0                               // jg            4d9c <.literal16+0x68c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4d80 <.literal16+0x690>
+  .byte  127,0                               // jg            4da0 <.literal16+0x690>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -25792,7 +25836,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4e05 <.literal16+0x715>
+  .byte  119,115                             // ja            4e25 <.literal16+0x715>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -25803,7 +25847,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4d69 <.literal16+0x679>
+  .byte  117,191                             // jne           4d89 <.literal16+0x679>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -25815,7 +25859,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38daa <_sk_callback_sse41+0xffffffffe9a3479d>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38dca <_sk_callback_sse41+0xffffffffe9a34793>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -25870,16 +25914,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e44 <.literal16+0x754>
+  .byte  127,0                               // jg            4e64 <.literal16+0x754>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e48 <.literal16+0x758>
+  .byte  127,0                               // jg            4e68 <.literal16+0x758>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e4c <.literal16+0x75c>
+  .byte  127,0                               // jg            4e6c <.literal16+0x75c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4e50 <.literal16+0x760>
+  .byte  127,0                               // jg            4e70 <.literal16+0x760>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -25888,7 +25932,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4ed5 <.literal16+0x7e5>
+  .byte  119,115                             // ja            4ef5 <.literal16+0x7e5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -25899,7 +25943,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4e39 <.literal16+0x749>
+  .byte  117,191                             // jne           4e59 <.literal16+0x749>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -25911,7 +25955,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38e7a <_sk_callback_sse41+0xffffffffe9a3486d>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38e9a <_sk_callback_sse41+0xffffffffe9a34863>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -25966,16 +26010,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f14 <.literal16+0x824>
+  .byte  127,0                               // jg            4f34 <.literal16+0x824>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f18 <.literal16+0x828>
+  .byte  127,0                               // jg            4f38 <.literal16+0x828>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f1c <.literal16+0x82c>
+  .byte  127,0                               // jg            4f3c <.literal16+0x82c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            4f20 <.literal16+0x830>
+  .byte  127,0                               // jg            4f40 <.literal16+0x830>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -25984,7 +26028,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            4fa5 <.literal16+0x8b5>
+  .byte  119,115                             // ja            4fc5 <.literal16+0x8b5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -25995,7 +26039,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           4f09 <.literal16+0x819>
+  .byte  117,191                             // jne           4f29 <.literal16+0x819>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -26007,7 +26051,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38f4a <_sk_callback_sse41+0xffffffffe9a3493d>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a38f6a <_sk_callback_sse41+0xffffffffe9a34933>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  81                                  // push          %rcx
   .byte  140,242                             // mov           %?,%edx
@@ -26058,13 +26102,13 @@ BALIGN16
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
-  .byte  127,67                              // jg            5027 <.literal16+0x937>
+  .byte  127,67                              // jg            5047 <.literal16+0x937>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            502b <.literal16+0x93b>
+  .byte  127,67                              // jg            504b <.literal16+0x93b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            502f <.literal16+0x93f>
+  .byte  127,67                              // jg            504f <.literal16+0x93f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5033 <.literal16+0x943>
+  .byte  127,67                              // jg            5053 <.literal16+0x943>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -26111,16 +26155,16 @@ BALIGN16
   .byte  128,3,62                            // addb          $0x3e,(%rbx)
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           50b3 <.literal16+0x9c3>
+  .byte  118,63                              // jbe           50d3 <.literal16+0x9c3>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           50b7 <.literal16+0x9c7>
+  .byte  118,63                              // jbe           50d7 <.literal16+0x9c7>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           50bb <.literal16+0x9cb>
+  .byte  118,63                              // jbe           50db <.literal16+0x9cb>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           50bf <.literal16+0x9cf>
+  .byte  118,63                              // jbe           50df <.literal16+0x9cf>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
@@ -26132,11 +26176,11 @@ BALIGN16
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            50fb <.literal16+0xa0b>
+  .byte  127,67                              // jg            511b <.literal16+0xa0b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            50ff <.literal16+0xa0f>
+  .byte  127,67                              // jg            511f <.literal16+0xa0f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5103 <.literal16+0xa13>
+  .byte  127,67                              // jg            5123 <.literal16+0xa13>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,0,0,128,63               // addb          $0x3f,-0x7fffffc5(%rax)
@@ -26165,7 +26209,7 @@ BALIGN16
   .byte  5,255,255,255,9                     // add           $0x9ffffff,%eax
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3005130 <_sk_callback_sse41+0x3000b23>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3005150 <_sk_callback_sse41+0x3000b19>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -26194,13 +26238,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5169 <.literal16+0xa79>
+  .byte  224,7                               // loopne        5189 <.literal16+0xa79>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        516d <.literal16+0xa7d>
+  .byte  224,7                               // loopne        518d <.literal16+0xa7d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5171 <.literal16+0xa81>
+  .byte  224,7                               // loopne        5191 <.literal16+0xa81>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5175 <.literal16+0xa85>
+  .byte  224,7                               // loopne        5195 <.literal16+0xa85>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -26246,13 +26290,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        51d9 <.literal16+0xae9>
+  .byte  224,7                               // loopne        51f9 <.literal16+0xae9>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        51dd <.literal16+0xaed>
+  .byte  224,7                               // loopne        51fd <.literal16+0xaed>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        51e1 <.literal16+0xaf1>
+  .byte  224,7                               // loopne        5201 <.literal16+0xaf1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        51e5 <.literal16+0xaf5>
+  .byte  224,7                               // loopne        5205 <.literal16+0xaf5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -26290,13 +26334,13 @@ BALIGN16
   .byte  65,0,0                              // add           %al,(%r8)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            5276 <.literal16+0xb86>
+  .byte  124,66                              // jl            5296 <.literal16+0xb86>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            527a <.literal16+0xb8a>
+  .byte  124,66                              // jl            529a <.literal16+0xb8a>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            527e <.literal16+0xb8e>
+  .byte  124,66                              // jl            529e <.literal16+0xb8e>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            5282 <.literal16+0xb92>
+  .byte  124,66                              // jl            52a2 <.literal16+0xb92>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,240                               // add           %dh,%al
@@ -26386,13 +26430,13 @@ BALIGN16
   .byte  136,136,61,137,136,136              // mov           %cl,-0x777776c3(%rax)
   .byte  61,137,136,136,61                   // cmp           $0x3d888889,%eax
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5385 <.literal16+0xc95>
+  .byte  112,65                              // jo            53a5 <.literal16+0xc95>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5389 <.literal16+0xc99>
+  .byte  112,65                              // jo            53a9 <.literal16+0xc99>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            538d <.literal16+0xc9d>
+  .byte  112,65                              // jo            53ad <.literal16+0xc9d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            5391 <.literal16+0xca1>
+  .byte  112,65                              // jo            53b1 <.literal16+0xca1>
   .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  255,0                               // incl          (%rax)
@@ -26407,7 +26451,7 @@ BALIGN16
   .byte  5,255,255,255,9                     // add           $0x9ffffff,%eax
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 3005380 <_sk_callback_sse41+0x3000d73>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 30053a0 <_sk_callback_sse41+0x3000d69>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -26434,7 +26478,7 @@ BALIGN16
   .byte  5,255,255,255,9                     // add           $0x9ffffff,%eax
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 30053c0 <_sk_callback_sse41+0x3000db3>
+  .byte  255,13,255,255,255,2                // decl          0x2ffffff(%rip)        # 30053e0 <_sk_callback_sse41+0x3000da9>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
   .byte  255,6                               // incl          (%rsi)
@@ -26449,11 +26493,11 @@ BALIGN16
   .byte  255,0                               // incl          (%rax)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            541b <.literal16+0xd2b>
+  .byte  127,67                              // jg            543b <.literal16+0xd2b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            541f <.literal16+0xd2f>
+  .byte  127,67                              // jg            543f <.literal16+0xd2f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5423 <.literal16+0xd33>
+  .byte  127,67                              // jg            5443 <.literal16+0xd33>
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
@@ -26529,13 +26573,13 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  255                                 // (bad)
-  .byte  127,71                              // jg            54eb <.literal16+0xdfb>
+  .byte  127,71                              // jg            550b <.literal16+0xdfb>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            54ef <.literal16+0xdff>
+  .byte  127,71                              // jg            550f <.literal16+0xdff>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            54f3 <.literal16+0xe03>
+  .byte  127,71                              // jg            5513 <.literal16+0xe03>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            54f7 <.literal16+0xe07>
+  .byte  127,71                              // jg            5517 <.literal16+0xe07>
   .byte  208                                 // (bad)
   .byte  179,89                              // mov           $0x59,%bl
   .byte  62,208                              // ds            (bad)
@@ -26669,11 +26713,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5612 <.literal16+0xf22>
+  .byte  62,114,28                           // jb,pt         5632 <.literal16+0xf22>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5616 <.literal16+0xf26>
+  .byte  62,114,28                           // jb,pt         5636 <.literal16+0xf26>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         561a <.literal16+0xf2a>
+  .byte  62,114,28                           // jb,pt         563a <.literal16+0xf2a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -26717,7 +26761,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e4a5 <_sk_callback_sse41+0x3d639e98>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e4c5 <_sk_callback_sse41+0x3d639e8e>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -26743,7 +26787,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e4e5 <_sk_callback_sse41+0x3d639ed8>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e505 <_sk_callback_sse41+0x3d639ece>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -26752,13 +26796,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            56de <.literal16+0xfee>
+  .byte  114,28                              // jb            56fe <.literal16+0xfee>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         56e2 <.literal16+0xff2>
+  .byte  62,114,28                           // jb,pt         5702 <.literal16+0xff2>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         56e6 <.literal16+0xff6>
+  .byte  62,114,28                           // jb,pt         5706 <.literal16+0xff6>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         56ea <.literal16+0xffa>
+  .byte  62,114,28                           // jb,pt         570a <.literal16+0xffa>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -26779,11 +26823,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5722 <.literal16+0x1032>
+  .byte  62,114,28                           // jb,pt         5742 <.literal16+0x1032>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5726 <.literal16+0x1036>
+  .byte  62,114,28                           // jb,pt         5746 <.literal16+0x1036>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         572a <.literal16+0x103a>
+  .byte  62,114,28                           // jb,pt         574a <.literal16+0x103a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -26827,7 +26871,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e5b5 <_sk_callback_sse41+0x3d639fa8>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e5d5 <_sk_callback_sse41+0x3d639f9e>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -26853,7 +26897,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e5f5 <_sk_callback_sse41+0x3d639fe8>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e615 <_sk_callback_sse41+0x3d639fde>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -26862,13 +26906,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            57ee <.literal16+0x10fe>
+  .byte  114,28                              // jb            580e <.literal16+0x10fe>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         57f2 <_sk_callback_sse41+0x11e5>
+  .byte  62,114,28                           // jb,pt         5812 <_sk_callback_sse41+0x11db>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         57f6 <_sk_callback_sse41+0x11e9>
+  .byte  62,114,28                           // jb,pt         5816 <_sk_callback_sse41+0x11df>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         57fa <_sk_callback_sse41+0x11ed>
+  .byte  62,114,28                           // jb,pt         581a <_sk_callback_sse41+0x11e3>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -26938,7 +26982,7 @@ _sk_seed_shader_sse2:
   .byte  102,15,110,199                      // movd          %edi,%xmm0
   .byte  102,15,112,192,0                    // pshufd        $0x0,%xmm0,%xmm0
   .byte  15,91,200                           // cvtdq2ps      %xmm0,%xmm1
-  .byte  15,40,21,228,74,0,0                 // movaps        0x4ae4(%rip),%xmm2        # 4b60 <_sk_callback_sse2+0xdf>
+  .byte  15,40,21,4,75,0,0                   // movaps        0x4b04(%rip),%xmm2        # 4b80 <_sk_callback_sse2+0xd5>
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  15,16,2                             // movups        (%rdx),%xmm0
   .byte  15,88,193                           // addps         %xmm1,%xmm0
@@ -26947,7 +26991,7 @@ _sk_seed_shader_sse2:
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  15,88,202                           // addps         %xmm2,%xmm1
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,21,211,74,0,0                 // movaps        0x4ad3(%rip),%xmm2        # 4b70 <_sk_callback_sse2+0xef>
+  .byte  15,40,21,243,74,0,0                 // movaps        0x4af3(%rip),%xmm2        # 4b90 <_sk_callback_sse2+0xe5>
   .byte  15,87,219                           // xorps         %xmm3,%xmm3
   .byte  15,87,228                           // xorps         %xmm4,%xmm4
   .byte  15,87,237                           // xorps         %xmm5,%xmm5
@@ -26970,14 +27014,14 @@ _sk_dither_sse2:
   .byte  102,68,15,110,1                     // movd          (%rcx),%xmm8
   .byte  102,69,15,112,192,0                 // pshufd        $0x0,%xmm8,%xmm8
   .byte  102,69,15,239,193                   // pxor          %xmm9,%xmm8
-  .byte  102,68,15,111,21,152,74,0,0         // movdqa        0x4a98(%rip),%xmm10        # 4b80 <_sk_callback_sse2+0xff>
+  .byte  102,68,15,111,21,184,74,0,0         // movdqa        0x4ab8(%rip),%xmm10        # 4ba0 <_sk_callback_sse2+0xf5>
   .byte  102,69,15,111,216                   // movdqa        %xmm8,%xmm11
   .byte  102,69,15,219,218                   // pand          %xmm10,%xmm11
   .byte  102,65,15,114,243,5                 // pslld         $0x5,%xmm11
   .byte  102,69,15,219,209                   // pand          %xmm9,%xmm10
   .byte  102,65,15,114,242,4                 // pslld         $0x4,%xmm10
-  .byte  102,68,15,111,37,132,74,0,0         // movdqa        0x4a84(%rip),%xmm12        # 4b90 <_sk_callback_sse2+0x10f>
-  .byte  102,68,15,111,45,139,74,0,0         // movdqa        0x4a8b(%rip),%xmm13        # 4ba0 <_sk_callback_sse2+0x11f>
+  .byte  102,68,15,111,37,164,74,0,0         // movdqa        0x4aa4(%rip),%xmm12        # 4bb0 <_sk_callback_sse2+0x105>
+  .byte  102,68,15,111,45,171,74,0,0         // movdqa        0x4aab(%rip),%xmm13        # 4bc0 <_sk_callback_sse2+0x115>
   .byte  102,69,15,111,240                   // movdqa        %xmm8,%xmm14
   .byte  102,69,15,219,245                   // pand          %xmm13,%xmm14
   .byte  102,65,15,114,246,2                 // pslld         $0x2,%xmm14
@@ -26993,15 +27037,26 @@ _sk_dither_sse2:
   .byte  102,69,15,235,245                   // por           %xmm13,%xmm14
   .byte  102,69,15,235,240                   // por           %xmm8,%xmm14
   .byte  69,15,91,198                        // cvtdq2ps      %xmm14,%xmm8
-  .byte  68,15,89,5,70,74,0,0                // mulps         0x4a46(%rip),%xmm8        # 4bb0 <_sk_callback_sse2+0x12f>
-  .byte  68,15,88,5,78,74,0,0                // addps         0x4a4e(%rip),%xmm8        # 4bc0 <_sk_callback_sse2+0x13f>
-  .byte  243,68,15,16,72,8                   // movss         0x8(%rax),%xmm9
-  .byte  69,15,198,201,0                     // shufps        $0x0,%xmm9,%xmm9
-  .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
-  .byte  65,15,88,193                        // addps         %xmm9,%xmm0
-  .byte  65,15,88,201                        // addps         %xmm9,%xmm1
-  .byte  65,15,88,209                        // addps         %xmm9,%xmm2
+  .byte  68,15,89,5,102,74,0,0               // mulps         0x4a66(%rip),%xmm8        # 4bd0 <_sk_callback_sse2+0x125>
+  .byte  68,15,88,5,110,74,0,0               // addps         0x4a6e(%rip),%xmm8        # 4be0 <_sk_callback_sse2+0x135>
+  .byte  243,68,15,16,80,8                   // movss         0x8(%rax),%xmm10
+  .byte  69,15,198,210,0                     // shufps        $0x0,%xmm10,%xmm10
+  .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
+  .byte  65,15,88,194                        // addps         %xmm10,%xmm0
+  .byte  65,15,88,202                        // addps         %xmm10,%xmm1
+  .byte  68,15,88,210                        // addps         %xmm2,%xmm10
+  .byte  15,93,195                           // minps         %xmm3,%xmm0
+  .byte  15,87,210                           // xorps         %xmm2,%xmm2
+  .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
+  .byte  68,15,95,192                        // maxps         %xmm0,%xmm8
+  .byte  15,93,203                           // minps         %xmm3,%xmm1
+  .byte  102,69,15,239,201                   // pxor          %xmm9,%xmm9
+  .byte  68,15,95,201                        // maxps         %xmm1,%xmm9
+  .byte  68,15,93,211                        // minps         %xmm3,%xmm10
+  .byte  65,15,95,210                        // maxps         %xmm10,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
+  .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
+  .byte  65,15,40,201                        // movaps        %xmm9,%xmm1
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_constant_color_sse2
@@ -27060,7 +27115,7 @@ HIDDEN _sk_srcatop_sse2
 FUNCTION(_sk_srcatop_sse2)
 _sk_srcatop_sse2:
   .byte  15,89,199                           // mulps         %xmm7,%xmm0
-  .byte  68,15,40,5,209,73,0,0               // movaps        0x49d1(%rip),%xmm8        # 4bd0 <_sk_callback_sse2+0x14f>
+  .byte  68,15,40,5,199,73,0,0               // movaps        0x49c7(%rip),%xmm8        # 4bf0 <_sk_callback_sse2+0x145>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -27085,7 +27140,7 @@ FUNCTION(_sk_dstatop_sse2)
 _sk_dstatop_sse2:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
   .byte  68,15,89,196                        // mulps         %xmm4,%xmm8
-  .byte  68,15,40,13,148,73,0,0              // movaps        0x4994(%rip),%xmm9        # 4be0 <_sk_callback_sse2+0x15f>
+  .byte  68,15,40,13,138,73,0,0              // movaps        0x498a(%rip),%xmm9        # 4c00 <_sk_callback_sse2+0x155>
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
@@ -27132,7 +27187,7 @@ HIDDEN _sk_srcout_sse2
 .globl _sk_srcout_sse2
 FUNCTION(_sk_srcout_sse2)
 _sk_srcout_sse2:
-  .byte  68,15,40,5,56,73,0,0                // movaps        0x4938(%rip),%xmm8        # 4bf0 <_sk_callback_sse2+0x16f>
+  .byte  68,15,40,5,46,73,0,0                // movaps        0x492e(%rip),%xmm8        # 4c10 <_sk_callback_sse2+0x165>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
@@ -27145,7 +27200,7 @@ HIDDEN _sk_dstout_sse2
 .globl _sk_dstout_sse2
 FUNCTION(_sk_dstout_sse2)
 _sk_dstout_sse2:
-  .byte  68,15,40,5,40,73,0,0                // movaps        0x4928(%rip),%xmm8        # 4c00 <_sk_callback_sse2+0x17f>
+  .byte  68,15,40,5,30,73,0,0                // movaps        0x491e(%rip),%xmm8        # 4c20 <_sk_callback_sse2+0x175>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  15,89,196                           // mulps         %xmm4,%xmm0
@@ -27162,7 +27217,7 @@ HIDDEN _sk_srcover_sse2
 .globl _sk_srcover_sse2
 FUNCTION(_sk_srcover_sse2)
 _sk_srcover_sse2:
-  .byte  68,15,40,5,11,73,0,0                // movaps        0x490b(%rip),%xmm8        # 4c10 <_sk_callback_sse2+0x18f>
+  .byte  68,15,40,5,1,73,0,0                 // movaps        0x4901(%rip),%xmm8        # 4c30 <_sk_callback_sse2+0x185>
   .byte  68,15,92,195                        // subps         %xmm3,%xmm8
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
@@ -27182,7 +27237,7 @@ HIDDEN _sk_dstover_sse2
 .globl _sk_dstover_sse2
 FUNCTION(_sk_dstover_sse2)
 _sk_dstover_sse2:
-  .byte  68,15,40,5,223,72,0,0               // movaps        0x48df(%rip),%xmm8        # 4c20 <_sk_callback_sse2+0x19f>
+  .byte  68,15,40,5,213,72,0,0               // movaps        0x48d5(%rip),%xmm8        # 4c40 <_sk_callback_sse2+0x195>
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -27210,7 +27265,7 @@ HIDDEN _sk_multiply_sse2
 .globl _sk_multiply_sse2
 FUNCTION(_sk_multiply_sse2)
 _sk_multiply_sse2:
-  .byte  68,15,40,5,179,72,0,0               // movaps        0x48b3(%rip),%xmm8        # 4c30 <_sk_callback_sse2+0x1af>
+  .byte  68,15,40,5,169,72,0,0               // movaps        0x48a9(%rip),%xmm8        # 4c50 <_sk_callback_sse2+0x1a5>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
@@ -27286,7 +27341,7 @@ HIDDEN _sk_xor__sse2
 FUNCTION(_sk_xor__sse2)
 _sk_xor__sse2:
   .byte  68,15,40,195                        // movaps        %xmm3,%xmm8
-  .byte  15,40,29,228,71,0,0                 // movaps        0x47e4(%rip),%xmm3        # 4c40 <_sk_callback_sse2+0x1bf>
+  .byte  15,40,29,218,71,0,0                 // movaps        0x47da(%rip),%xmm3        # 4c60 <_sk_callback_sse2+0x1b5>
   .byte  68,15,40,203                        // movaps        %xmm3,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
@@ -27334,7 +27389,7 @@ _sk_darken_sse2:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,95,209                        // maxps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,79,71,0,0                  // movaps        0x474f(%rip),%xmm2        # 4c50 <_sk_callback_sse2+0x1cf>
+  .byte  15,40,21,69,71,0,0                  // movaps        0x4745(%rip),%xmm2        # 4c70 <_sk_callback_sse2+0x1c5>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -27368,7 +27423,7 @@ _sk_lighten_sse2:
   .byte  68,15,89,206                        // mulps         %xmm6,%xmm9
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,244,70,0,0                 // movaps        0x46f4(%rip),%xmm2        # 4c60 <_sk_callback_sse2+0x1df>
+  .byte  15,40,21,234,70,0,0                 // movaps        0x46ea(%rip),%xmm2        # 4c80 <_sk_callback_sse2+0x1d5>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -27405,7 +27460,7 @@ _sk_difference_sse2:
   .byte  65,15,93,209                        // minps         %xmm9,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,194                        // subps         %xmm2,%xmm8
-  .byte  15,40,21,142,70,0,0                 // movaps        0x468e(%rip),%xmm2        # 4c70 <_sk_callback_sse2+0x1ef>
+  .byte  15,40,21,132,70,0,0                 // movaps        0x4684(%rip),%xmm2        # 4c90 <_sk_callback_sse2+0x1e5>
   .byte  15,92,211                           // subps         %xmm3,%xmm2
   .byte  15,89,215                           // mulps         %xmm7,%xmm2
   .byte  15,88,218                           // addps         %xmm2,%xmm3
@@ -27432,7 +27487,7 @@ _sk_exclusion_sse2:
   .byte  15,89,214                           // mulps         %xmm6,%xmm2
   .byte  15,88,210                           // addps         %xmm2,%xmm2
   .byte  68,15,92,202                        // subps         %xmm2,%xmm9
-  .byte  15,40,13,79,70,0,0                  // movaps        0x464f(%rip),%xmm1        # 4c80 <_sk_callback_sse2+0x1ff>
+  .byte  15,40,13,69,70,0,0                  // movaps        0x4645(%rip),%xmm1        # 4ca0 <_sk_callback_sse2+0x1f5>
   .byte  15,92,203                           // subps         %xmm3,%xmm1
   .byte  15,89,207                           // mulps         %xmm7,%xmm1
   .byte  15,88,217                           // addps         %xmm1,%xmm3
@@ -27446,7 +27501,7 @@ HIDDEN _sk_colorburn_sse2
 FUNCTION(_sk_colorburn_sse2)
 _sk_colorburn_sse2:
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
-  .byte  68,15,40,21,62,70,0,0               // movaps        0x463e(%rip),%xmm10        # 4c90 <_sk_callback_sse2+0x20f>
+  .byte  68,15,40,21,52,70,0,0               // movaps        0x4634(%rip),%xmm10        # 4cb0 <_sk_callback_sse2+0x205>
   .byte  69,15,40,202                        // movaps        %xmm10,%xmm9
   .byte  68,15,92,207                        // subps         %xmm7,%xmm9
   .byte  69,15,40,217                        // movaps        %xmm9,%xmm11
@@ -27540,7 +27595,7 @@ HIDDEN _sk_colordodge_sse2
 FUNCTION(_sk_colordodge_sse2)
 _sk_colordodge_sse2:
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
-  .byte  68,15,40,21,244,68,0,0              // movaps        0x44f4(%rip),%xmm10        # 4ca0 <_sk_callback_sse2+0x21f>
+  .byte  68,15,40,21,234,68,0,0              // movaps        0x44ea(%rip),%xmm10        # 4cc0 <_sk_callback_sse2+0x215>
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
   .byte  68,15,92,223                        // subps         %xmm7,%xmm11
   .byte  69,15,40,227                        // movaps        %xmm11,%xmm12
@@ -27634,7 +27689,7 @@ _sk_hardlight_sse2:
   .byte  15,41,116,36,232                    // movaps        %xmm6,-0x18(%rsp)
   .byte  15,40,245                           // movaps        %xmm5,%xmm6
   .byte  15,40,236                           // movaps        %xmm4,%xmm5
-  .byte  68,15,40,29,169,67,0,0              // movaps        0x43a9(%rip),%xmm11        # 4cb0 <_sk_callback_sse2+0x22f>
+  .byte  68,15,40,29,159,67,0,0              // movaps        0x439f(%rip),%xmm11        # 4cd0 <_sk_callback_sse2+0x225>
   .byte  69,15,40,211                        // movaps        %xmm11,%xmm10
   .byte  68,15,92,215                        // subps         %xmm7,%xmm10
   .byte  69,15,40,194                        // movaps        %xmm10,%xmm8
@@ -27722,7 +27777,7 @@ FUNCTION(_sk_overlay_sse2)
 _sk_overlay_sse2:
   .byte  68,15,40,193                        // movaps        %xmm1,%xmm8
   .byte  68,15,40,232                        // movaps        %xmm0,%xmm13
-  .byte  68,15,40,13,119,66,0,0              // movaps        0x4277(%rip),%xmm9        # 4cc0 <_sk_callback_sse2+0x23f>
+  .byte  68,15,40,13,109,66,0,0              // movaps        0x426d(%rip),%xmm9        # 4ce0 <_sk_callback_sse2+0x235>
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
   .byte  68,15,92,215                        // subps         %xmm7,%xmm10
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
@@ -27813,7 +27868,7 @@ _sk_softlight_sse2:
   .byte  68,15,40,213                        // movaps        %xmm5,%xmm10
   .byte  68,15,94,215                        // divps         %xmm7,%xmm10
   .byte  69,15,84,212                        // andps         %xmm12,%xmm10
-  .byte  68,15,40,13,52,65,0,0               // movaps        0x4134(%rip),%xmm9        # 4cd0 <_sk_callback_sse2+0x24f>
+  .byte  68,15,40,13,42,65,0,0               // movaps        0x412a(%rip),%xmm9        # 4cf0 <_sk_callback_sse2+0x245>
   .byte  69,15,40,249                        // movaps        %xmm9,%xmm15
   .byte  69,15,92,250                        // subps         %xmm10,%xmm15
   .byte  69,15,40,218                        // movaps        %xmm10,%xmm11
@@ -27826,10 +27881,10 @@ _sk_softlight_sse2:
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  15,89,192                           // mulps         %xmm0,%xmm0
   .byte  65,15,88,194                        // addps         %xmm10,%xmm0
-  .byte  68,15,40,53,14,65,0,0               // movaps        0x410e(%rip),%xmm14        # 4ce0 <_sk_callback_sse2+0x25f>
+  .byte  68,15,40,53,4,65,0,0                // movaps        0x4104(%rip),%xmm14        # 4d00 <_sk_callback_sse2+0x255>
   .byte  69,15,88,222                        // addps         %xmm14,%xmm11
   .byte  68,15,89,216                        // mulps         %xmm0,%xmm11
-  .byte  68,15,40,21,14,65,0,0               // movaps        0x410e(%rip),%xmm10        # 4cf0 <_sk_callback_sse2+0x26f>
+  .byte  68,15,40,21,4,65,0,0                // movaps        0x4104(%rip),%xmm10        # 4d10 <_sk_callback_sse2+0x265>
   .byte  69,15,89,234                        // mulps         %xmm10,%xmm13
   .byte  69,15,88,235                        // addps         %xmm11,%xmm13
   .byte  15,88,228                           // addps         %xmm4,%xmm4
@@ -27974,7 +28029,7 @@ _sk_hue_sse2:
   .byte  68,15,40,209                        // movaps        %xmm1,%xmm10
   .byte  68,15,40,225                        // movaps        %xmm1,%xmm12
   .byte  68,15,89,211                        // mulps         %xmm3,%xmm10
-  .byte  68,15,40,5,81,63,0,0                // movaps        0x3f51(%rip),%xmm8        # 4d30 <_sk_callback_sse2+0x2af>
+  .byte  68,15,40,5,71,63,0,0                // movaps        0x3f47(%rip),%xmm8        # 4d50 <_sk_callback_sse2+0x2a5>
   .byte  69,15,40,216                        // movaps        %xmm8,%xmm11
   .byte  15,40,207                           // movaps        %xmm7,%xmm1
   .byte  68,15,92,217                        // subps         %xmm1,%xmm11
@@ -28020,12 +28075,12 @@ _sk_hue_sse2:
   .byte  69,15,84,206                        // andps         %xmm14,%xmm9
   .byte  69,15,84,214                        // andps         %xmm14,%xmm10
   .byte  65,15,84,214                        // andps         %xmm14,%xmm2
-  .byte  68,15,40,61,101,62,0,0              // movaps        0x3e65(%rip),%xmm15        # 4d00 <_sk_callback_sse2+0x27f>
+  .byte  68,15,40,61,91,62,0,0               // movaps        0x3e5b(%rip),%xmm15        # 4d20 <_sk_callback_sse2+0x275>
   .byte  65,15,89,231                        // mulps         %xmm15,%xmm4
-  .byte  15,40,5,106,62,0,0                  // movaps        0x3e6a(%rip),%xmm0        # 4d10 <_sk_callback_sse2+0x28f>
+  .byte  15,40,5,96,62,0,0                   // movaps        0x3e60(%rip),%xmm0        # 4d30 <_sk_callback_sse2+0x285>
   .byte  15,89,240                           // mulps         %xmm0,%xmm6
   .byte  15,88,244                           // addps         %xmm4,%xmm6
-  .byte  68,15,40,53,108,62,0,0              // movaps        0x3e6c(%rip),%xmm14        # 4d20 <_sk_callback_sse2+0x29f>
+  .byte  68,15,40,53,98,62,0,0               // movaps        0x3e62(%rip),%xmm14        # 4d40 <_sk_callback_sse2+0x295>
   .byte  68,15,40,239                        // movaps        %xmm7,%xmm13
   .byte  69,15,89,238                        // mulps         %xmm14,%xmm13
   .byte  68,15,88,238                        // addps         %xmm6,%xmm13
@@ -28202,14 +28257,14 @@ _sk_saturation_sse2:
   .byte  68,15,84,211                        // andps         %xmm3,%xmm10
   .byte  68,15,84,203                        // andps         %xmm3,%xmm9
   .byte  15,84,195                           // andps         %xmm3,%xmm0
-  .byte  68,15,40,5,1,60,0,0                 // movaps        0x3c01(%rip),%xmm8        # 4d40 <_sk_callback_sse2+0x2bf>
+  .byte  68,15,40,5,247,59,0,0               // movaps        0x3bf7(%rip),%xmm8        # 4d60 <_sk_callback_sse2+0x2b5>
   .byte  15,40,214                           // movaps        %xmm6,%xmm2
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
-  .byte  15,40,13,3,60,0,0                   // movaps        0x3c03(%rip),%xmm1        # 4d50 <_sk_callback_sse2+0x2cf>
+  .byte  15,40,13,249,59,0,0                 // movaps        0x3bf9(%rip),%xmm1        # 4d70 <_sk_callback_sse2+0x2c5>
   .byte  15,40,221                           // movaps        %xmm5,%xmm3
   .byte  15,89,217                           // mulps         %xmm1,%xmm3
   .byte  15,88,218                           // addps         %xmm2,%xmm3
-  .byte  68,15,40,37,2,60,0,0                // movaps        0x3c02(%rip),%xmm12        # 4d60 <_sk_callback_sse2+0x2df>
+  .byte  68,15,40,37,248,59,0,0              // movaps        0x3bf8(%rip),%xmm12        # 4d80 <_sk_callback_sse2+0x2d5>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
   .byte  68,15,88,235                        // addps         %xmm3,%xmm13
   .byte  65,15,40,210                        // movaps        %xmm10,%xmm2
@@ -28254,7 +28309,7 @@ _sk_saturation_sse2:
   .byte  15,40,223                           // movaps        %xmm7,%xmm3
   .byte  15,40,236                           // movaps        %xmm4,%xmm5
   .byte  15,89,221                           // mulps         %xmm5,%xmm3
-  .byte  68,15,40,5,103,59,0,0               // movaps        0x3b67(%rip),%xmm8        # 4d70 <_sk_callback_sse2+0x2ef>
+  .byte  68,15,40,5,93,59,0,0                // movaps        0x3b5d(%rip),%xmm8        # 4d90 <_sk_callback_sse2+0x2e5>
   .byte  65,15,40,224                        // movaps        %xmm8,%xmm4
   .byte  68,15,92,199                        // subps         %xmm7,%xmm8
   .byte  15,88,253                           // addps         %xmm5,%xmm7
@@ -28355,14 +28410,14 @@ _sk_color_sse2:
   .byte  68,15,40,213                        // movaps        %xmm5,%xmm10
   .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
   .byte  65,15,40,208                        // movaps        %xmm8,%xmm2
-  .byte  68,15,40,45,5,58,0,0                // movaps        0x3a05(%rip),%xmm13        # 4d80 <_sk_callback_sse2+0x2ff>
+  .byte  68,15,40,45,251,57,0,0              // movaps        0x39fb(%rip),%xmm13        # 4da0 <_sk_callback_sse2+0x2f5>
   .byte  68,15,40,198                        // movaps        %xmm6,%xmm8
   .byte  69,15,89,197                        // mulps         %xmm13,%xmm8
-  .byte  68,15,40,53,5,58,0,0                // movaps        0x3a05(%rip),%xmm14        # 4d90 <_sk_callback_sse2+0x30f>
+  .byte  68,15,40,53,251,57,0,0              // movaps        0x39fb(%rip),%xmm14        # 4db0 <_sk_callback_sse2+0x305>
   .byte  65,15,40,195                        // movaps        %xmm11,%xmm0
   .byte  65,15,89,198                        // mulps         %xmm14,%xmm0
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
-  .byte  68,15,40,29,1,58,0,0                // movaps        0x3a01(%rip),%xmm11        # 4da0 <_sk_callback_sse2+0x31f>
+  .byte  68,15,40,29,247,57,0,0              // movaps        0x39f7(%rip),%xmm11        # 4dc0 <_sk_callback_sse2+0x315>
   .byte  69,15,89,227                        // mulps         %xmm11,%xmm12
   .byte  68,15,88,224                        // addps         %xmm0,%xmm12
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
@@ -28370,7 +28425,7 @@ _sk_color_sse2:
   .byte  69,15,40,250                        // movaps        %xmm10,%xmm15
   .byte  69,15,89,254                        // mulps         %xmm14,%xmm15
   .byte  68,15,88,248                        // addps         %xmm0,%xmm15
-  .byte  68,15,40,5,237,57,0,0               // movaps        0x39ed(%rip),%xmm8        # 4db0 <_sk_callback_sse2+0x32f>
+  .byte  68,15,40,5,227,57,0,0               // movaps        0x39e3(%rip),%xmm8        # 4dd0 <_sk_callback_sse2+0x325>
   .byte  65,15,40,224                        // movaps        %xmm8,%xmm4
   .byte  15,92,226                           // subps         %xmm2,%xmm4
   .byte  15,89,252                           // mulps         %xmm4,%xmm7
@@ -28506,15 +28561,15 @@ _sk_luminosity_sse2:
   .byte  68,15,40,205                        // movaps        %xmm5,%xmm9
   .byte  68,15,89,204                        // mulps         %xmm4,%xmm9
   .byte  15,89,222                           // mulps         %xmm6,%xmm3
-  .byte  68,15,40,37,4,56,0,0                // movaps        0x3804(%rip),%xmm12        # 4dc0 <_sk_callback_sse2+0x33f>
+  .byte  68,15,40,37,250,55,0,0              // movaps        0x37fa(%rip),%xmm12        # 4de0 <_sk_callback_sse2+0x335>
   .byte  68,15,40,199                        // movaps        %xmm7,%xmm8
   .byte  69,15,89,196                        // mulps         %xmm12,%xmm8
-  .byte  68,15,40,45,4,56,0,0                // movaps        0x3804(%rip),%xmm13        # 4dd0 <_sk_callback_sse2+0x34f>
+  .byte  68,15,40,45,250,55,0,0              // movaps        0x37fa(%rip),%xmm13        # 4df0 <_sk_callback_sse2+0x345>
   .byte  68,15,40,241                        // movaps        %xmm1,%xmm14
   .byte  69,15,89,245                        // mulps         %xmm13,%xmm14
   .byte  69,15,88,240                        // addps         %xmm8,%xmm14
-  .byte  68,15,40,29,0,56,0,0                // movaps        0x3800(%rip),%xmm11        # 4de0 <_sk_callback_sse2+0x35f>
-  .byte  68,15,40,5,8,56,0,0                 // movaps        0x3808(%rip),%xmm8        # 4df0 <_sk_callback_sse2+0x36f>
+  .byte  68,15,40,29,246,55,0,0              // movaps        0x37f6(%rip),%xmm11        # 4e00 <_sk_callback_sse2+0x355>
+  .byte  68,15,40,5,254,55,0,0               // movaps        0x37fe(%rip),%xmm8        # 4e10 <_sk_callback_sse2+0x365>
   .byte  69,15,40,248                        // movaps        %xmm8,%xmm15
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  68,15,92,248                        // subps         %xmm0,%xmm15
@@ -28659,7 +28714,7 @@ HIDDEN _sk_clamp_1_sse2
 .globl _sk_clamp_1_sse2
 FUNCTION(_sk_clamp_1_sse2)
 _sk_clamp_1_sse2:
-  .byte  68,15,40,5,17,54,0,0                // movaps        0x3611(%rip),%xmm8        # 4e00 <_sk_callback_sse2+0x37f>
+  .byte  68,15,40,5,7,54,0,0                 // movaps        0x3607(%rip),%xmm8        # 4e20 <_sk_callback_sse2+0x375>
   .byte  65,15,93,192                        // minps         %xmm8,%xmm0
   .byte  65,15,93,200                        // minps         %xmm8,%xmm1
   .byte  65,15,93,208                        // minps         %xmm8,%xmm2
@@ -28671,7 +28726,7 @@ HIDDEN _sk_clamp_a_sse2
 .globl _sk_clamp_a_sse2
 FUNCTION(_sk_clamp_a_sse2)
 _sk_clamp_a_sse2:
-  .byte  15,93,29,6,54,0,0                   // minps         0x3606(%rip),%xmm3        # 4e10 <_sk_callback_sse2+0x38f>
+  .byte  15,93,29,252,53,0,0                 // minps         0x35fc(%rip),%xmm3        # 4e30 <_sk_callback_sse2+0x385>
   .byte  15,93,195                           // minps         %xmm3,%xmm0
   .byte  15,93,203                           // minps         %xmm3,%xmm1
   .byte  15,93,211                           // minps         %xmm3,%xmm2
@@ -28758,7 +28813,7 @@ HIDDEN _sk_unpremul_sse2
 FUNCTION(_sk_unpremul_sse2)
 _sk_unpremul_sse2:
   .byte  69,15,87,192                        // xorps         %xmm8,%xmm8
-  .byte  68,15,40,13,113,53,0,0              // movaps        0x3571(%rip),%xmm9        # 4e20 <_sk_callback_sse2+0x39f>
+  .byte  68,15,40,13,103,53,0,0              // movaps        0x3567(%rip),%xmm9        # 4e40 <_sk_callback_sse2+0x395>
   .byte  68,15,94,203                        // divps         %xmm3,%xmm9
   .byte  68,15,194,195,4                     // cmpneqps      %xmm3,%xmm8
   .byte  69,15,84,193                        // andps         %xmm9,%xmm8
@@ -28772,20 +28827,20 @@ HIDDEN _sk_from_srgb_sse2
 .globl _sk_from_srgb_sse2
 FUNCTION(_sk_from_srgb_sse2)
 _sk_from_srgb_sse2:
-  .byte  68,15,40,5,92,53,0,0                // movaps        0x355c(%rip),%xmm8        # 4e30 <_sk_callback_sse2+0x3af>
+  .byte  68,15,40,5,82,53,0,0                // movaps        0x3552(%rip),%xmm8        # 4e50 <_sk_callback_sse2+0x3a5>
   .byte  68,15,40,232                        // movaps        %xmm0,%xmm13
   .byte  69,15,89,232                        // mulps         %xmm8,%xmm13
   .byte  68,15,40,216                        // movaps        %xmm0,%xmm11
   .byte  69,15,89,219                        // mulps         %xmm11,%xmm11
-  .byte  68,15,40,13,84,53,0,0               // movaps        0x3554(%rip),%xmm9        # 4e40 <_sk_callback_sse2+0x3bf>
+  .byte  68,15,40,13,74,53,0,0               // movaps        0x354a(%rip),%xmm9        # 4e60 <_sk_callback_sse2+0x3b5>
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
   .byte  69,15,89,241                        // mulps         %xmm9,%xmm14
-  .byte  68,15,40,21,84,53,0,0               // movaps        0x3554(%rip),%xmm10        # 4e50 <_sk_callback_sse2+0x3cf>
+  .byte  68,15,40,21,74,53,0,0               // movaps        0x354a(%rip),%xmm10        # 4e70 <_sk_callback_sse2+0x3c5>
   .byte  69,15,88,242                        // addps         %xmm10,%xmm14
   .byte  69,15,89,243                        // mulps         %xmm11,%xmm14
-  .byte  68,15,40,29,84,53,0,0               // movaps        0x3554(%rip),%xmm11        # 4e60 <_sk_callback_sse2+0x3df>
+  .byte  68,15,40,29,74,53,0,0               // movaps        0x354a(%rip),%xmm11        # 4e80 <_sk_callback_sse2+0x3d5>
   .byte  69,15,88,243                        // addps         %xmm11,%xmm14
-  .byte  68,15,40,37,88,53,0,0               // movaps        0x3558(%rip),%xmm12        # 4e70 <_sk_callback_sse2+0x3ef>
+  .byte  68,15,40,37,78,53,0,0               // movaps        0x354e(%rip),%xmm12        # 4e90 <_sk_callback_sse2+0x3e5>
   .byte  65,15,194,196,1                     // cmpltps       %xmm12,%xmm0
   .byte  68,15,84,232                        // andps         %xmm0,%xmm13
   .byte  65,15,85,198                        // andnps        %xmm14,%xmm0
@@ -28824,20 +28879,20 @@ _sk_to_srgb_sse2:
   .byte  68,15,82,192                        // rsqrtps       %xmm0,%xmm8
   .byte  69,15,83,200                        // rcpps         %xmm8,%xmm9
   .byte  69,15,82,232                        // rsqrtps       %xmm8,%xmm13
-  .byte  68,15,40,5,221,52,0,0               // movaps        0x34dd(%rip),%xmm8        # 4e80 <_sk_callback_sse2+0x3ff>
+  .byte  68,15,40,5,211,52,0,0               // movaps        0x34d3(%rip),%xmm8        # 4ea0 <_sk_callback_sse2+0x3f5>
   .byte  68,15,40,240                        // movaps        %xmm0,%xmm14
   .byte  69,15,89,240                        // mulps         %xmm8,%xmm14
-  .byte  68,15,40,21,221,52,0,0              // movaps        0x34dd(%rip),%xmm10        # 4e90 <_sk_callback_sse2+0x40f>
+  .byte  68,15,40,21,211,52,0,0              // movaps        0x34d3(%rip),%xmm10        # 4eb0 <_sk_callback_sse2+0x405>
   .byte  69,15,89,202                        // mulps         %xmm10,%xmm9
-  .byte  68,15,40,29,225,52,0,0              // movaps        0x34e1(%rip),%xmm11        # 4ea0 <_sk_callback_sse2+0x41f>
+  .byte  68,15,40,29,215,52,0,0              // movaps        0x34d7(%rip),%xmm11        # 4ec0 <_sk_callback_sse2+0x415>
   .byte  69,15,88,203                        // addps         %xmm11,%xmm9
-  .byte  68,15,40,37,229,52,0,0              // movaps        0x34e5(%rip),%xmm12        # 4eb0 <_sk_callback_sse2+0x42f>
+  .byte  68,15,40,37,219,52,0,0              // movaps        0x34db(%rip),%xmm12        # 4ed0 <_sk_callback_sse2+0x425>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,40,13,229,52,0,0              // movaps        0x34e5(%rip),%xmm9        # 4ec0 <_sk_callback_sse2+0x43f>
+  .byte  68,15,40,13,219,52,0,0              // movaps        0x34db(%rip),%xmm9        # 4ee0 <_sk_callback_sse2+0x435>
   .byte  69,15,40,249                        // movaps        %xmm9,%xmm15
   .byte  69,15,93,253                        // minps         %xmm13,%xmm15
-  .byte  68,15,40,45,229,52,0,0              // movaps        0x34e5(%rip),%xmm13        # 4ed0 <_sk_callback_sse2+0x44f>
+  .byte  68,15,40,45,219,52,0,0              // movaps        0x34db(%rip),%xmm13        # 4ef0 <_sk_callback_sse2+0x445>
   .byte  65,15,194,197,1                     // cmpltps       %xmm13,%xmm0
   .byte  68,15,84,240                        // andps         %xmm0,%xmm14
   .byte  65,15,85,199                        // andnps        %xmm15,%xmm0
@@ -28887,7 +28942,7 @@ _sk_rgb_to_hsl_sse2:
   .byte  68,15,93,218                        // minps         %xmm2,%xmm11
   .byte  65,15,40,202                        // movaps        %xmm10,%xmm1
   .byte  65,15,92,203                        // subps         %xmm11,%xmm1
-  .byte  68,15,40,45,62,52,0,0               // movaps        0x343e(%rip),%xmm13        # 4ee0 <_sk_callback_sse2+0x45f>
+  .byte  68,15,40,45,52,52,0,0               // movaps        0x3434(%rip),%xmm13        # 4f00 <_sk_callback_sse2+0x455>
   .byte  68,15,94,233                        // divps         %xmm1,%xmm13
   .byte  65,15,40,194                        // movaps        %xmm10,%xmm0
   .byte  65,15,194,192,0                     // cmpeqps       %xmm8,%xmm0
@@ -28896,30 +28951,30 @@ _sk_rgb_to_hsl_sse2:
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,40,241                        // movaps        %xmm9,%xmm14
   .byte  68,15,194,242,1                     // cmpltps       %xmm2,%xmm14
-  .byte  68,15,84,53,36,52,0,0               // andps         0x3424(%rip),%xmm14        # 4ef0 <_sk_callback_sse2+0x46f>
+  .byte  68,15,84,53,26,52,0,0               // andps         0x341a(%rip),%xmm14        # 4f10 <_sk_callback_sse2+0x465>
   .byte  69,15,88,244                        // addps         %xmm12,%xmm14
   .byte  69,15,40,250                        // movaps        %xmm10,%xmm15
   .byte  69,15,194,249,0                     // cmpeqps       %xmm9,%xmm15
   .byte  65,15,92,208                        // subps         %xmm8,%xmm2
   .byte  65,15,89,213                        // mulps         %xmm13,%xmm2
-  .byte  68,15,40,37,23,52,0,0               // movaps        0x3417(%rip),%xmm12        # 4f00 <_sk_callback_sse2+0x47f>
+  .byte  68,15,40,37,13,52,0,0               // movaps        0x340d(%rip),%xmm12        # 4f20 <_sk_callback_sse2+0x475>
   .byte  65,15,88,212                        // addps         %xmm12,%xmm2
   .byte  69,15,92,193                        // subps         %xmm9,%xmm8
   .byte  69,15,89,197                        // mulps         %xmm13,%xmm8
-  .byte  68,15,88,5,19,52,0,0                // addps         0x3413(%rip),%xmm8        # 4f10 <_sk_callback_sse2+0x48f>
+  .byte  68,15,88,5,9,52,0,0                 // addps         0x3409(%rip),%xmm8        # 4f30 <_sk_callback_sse2+0x485>
   .byte  65,15,84,215                        // andps         %xmm15,%xmm2
   .byte  69,15,85,248                        // andnps        %xmm8,%xmm15
   .byte  68,15,86,250                        // orps          %xmm2,%xmm15
   .byte  68,15,84,240                        // andps         %xmm0,%xmm14
   .byte  65,15,85,199                        // andnps        %xmm15,%xmm0
   .byte  65,15,86,198                        // orps          %xmm14,%xmm0
-  .byte  15,89,5,4,52,0,0                    // mulps         0x3404(%rip),%xmm0        # 4f20 <_sk_callback_sse2+0x49f>
+  .byte  15,89,5,250,51,0,0                  // mulps         0x33fa(%rip),%xmm0        # 4f40 <_sk_callback_sse2+0x495>
   .byte  69,15,40,194                        // movaps        %xmm10,%xmm8
   .byte  69,15,194,195,4                     // cmpneqps      %xmm11,%xmm8
   .byte  65,15,84,192                        // andps         %xmm8,%xmm0
   .byte  69,15,92,226                        // subps         %xmm10,%xmm12
   .byte  69,15,88,211                        // addps         %xmm11,%xmm10
-  .byte  68,15,40,13,247,51,0,0              // movaps        0x33f7(%rip),%xmm9        # 4f30 <_sk_callback_sse2+0x4af>
+  .byte  68,15,40,13,237,51,0,0              // movaps        0x33ed(%rip),%xmm9        # 4f50 <_sk_callback_sse2+0x4a5>
   .byte  65,15,40,210                        // movaps        %xmm10,%xmm2
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
   .byte  68,15,194,202,1                     // cmpltps       %xmm2,%xmm9
@@ -28943,7 +28998,7 @@ _sk_hsl_to_rgb_sse2:
   .byte  15,41,92,36,168                     // movaps        %xmm3,-0x58(%rsp)
   .byte  68,15,40,218                        // movaps        %xmm2,%xmm11
   .byte  15,40,240                           // movaps        %xmm0,%xmm6
-  .byte  68,15,40,13,182,51,0,0              // movaps        0x33b6(%rip),%xmm9        # 4f40 <_sk_callback_sse2+0x4bf>
+  .byte  68,15,40,13,172,51,0,0              // movaps        0x33ac(%rip),%xmm9        # 4f60 <_sk_callback_sse2+0x4b5>
   .byte  69,15,40,209                        // movaps        %xmm9,%xmm10
   .byte  69,15,194,211,2                     // cmpleps       %xmm11,%xmm10
   .byte  15,40,193                           // movaps        %xmm1,%xmm0
@@ -28960,28 +29015,28 @@ _sk_hsl_to_rgb_sse2:
   .byte  69,15,88,211                        // addps         %xmm11,%xmm10
   .byte  69,15,88,219                        // addps         %xmm11,%xmm11
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
-  .byte  15,40,5,127,51,0,0                  // movaps        0x337f(%rip),%xmm0        # 4f50 <_sk_callback_sse2+0x4cf>
+  .byte  15,40,5,117,51,0,0                  // movaps        0x3375(%rip),%xmm0        # 4f70 <_sk_callback_sse2+0x4c5>
   .byte  15,88,198                           // addps         %xmm6,%xmm0
   .byte  243,15,91,200                       // cvttps2dq     %xmm0,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
   .byte  15,40,216                           // movaps        %xmm0,%xmm3
   .byte  15,194,217,1                        // cmpltps       %xmm1,%xmm3
-  .byte  15,84,29,119,51,0,0                 // andps         0x3377(%rip),%xmm3        # 4f60 <_sk_callback_sse2+0x4df>
+  .byte  15,84,29,109,51,0,0                 // andps         0x336d(%rip),%xmm3        # 4f80 <_sk_callback_sse2+0x4d5>
   .byte  15,92,203                           // subps         %xmm3,%xmm1
   .byte  15,92,193                           // subps         %xmm1,%xmm0
-  .byte  68,15,40,45,121,51,0,0              // movaps        0x3379(%rip),%xmm13        # 4f70 <_sk_callback_sse2+0x4ef>
+  .byte  68,15,40,45,111,51,0,0              // movaps        0x336f(%rip),%xmm13        # 4f90 <_sk_callback_sse2+0x4e5>
   .byte  69,15,40,197                        // movaps        %xmm13,%xmm8
   .byte  68,15,194,192,2                     // cmpleps       %xmm0,%xmm8
   .byte  69,15,40,242                        // movaps        %xmm10,%xmm14
   .byte  69,15,92,243                        // subps         %xmm11,%xmm14
   .byte  65,15,40,217                        // movaps        %xmm9,%xmm3
   .byte  15,194,216,2                        // cmpleps       %xmm0,%xmm3
-  .byte  15,40,21,137,51,0,0                 // movaps        0x3389(%rip),%xmm2        # 4fa0 <_sk_callback_sse2+0x51f>
+  .byte  15,40,21,127,51,0,0                 // movaps        0x337f(%rip),%xmm2        # 4fc0 <_sk_callback_sse2+0x515>
   .byte  68,15,40,250                        // movaps        %xmm2,%xmm15
   .byte  68,15,194,248,2                     // cmpleps       %xmm0,%xmm15
-  .byte  15,40,13,89,51,0,0                  // movaps        0x3359(%rip),%xmm1        # 4f80 <_sk_callback_sse2+0x4ff>
+  .byte  15,40,13,79,51,0,0                  // movaps        0x334f(%rip),%xmm1        # 4fa0 <_sk_callback_sse2+0x4f5>
   .byte  15,89,193                           // mulps         %xmm1,%xmm0
-  .byte  15,40,45,95,51,0,0                  // movaps        0x335f(%rip),%xmm5        # 4f90 <_sk_callback_sse2+0x50f>
+  .byte  15,40,45,85,51,0,0                  // movaps        0x3355(%rip),%xmm5        # 4fb0 <_sk_callback_sse2+0x505>
   .byte  15,40,229                           // movaps        %xmm5,%xmm4
   .byte  15,92,224                           // subps         %xmm0,%xmm4
   .byte  65,15,89,230                        // mulps         %xmm14,%xmm4
@@ -29004,7 +29059,7 @@ _sk_hsl_to_rgb_sse2:
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
   .byte  15,40,222                           // movaps        %xmm6,%xmm3
   .byte  15,194,216,1                        // cmpltps       %xmm0,%xmm3
-  .byte  15,84,29,212,50,0,0                 // andps         0x32d4(%rip),%xmm3        # 4f60 <_sk_callback_sse2+0x4df>
+  .byte  15,84,29,202,50,0,0                 // andps         0x32ca(%rip),%xmm3        # 4f80 <_sk_callback_sse2+0x4d5>
   .byte  15,92,195                           // subps         %xmm3,%xmm0
   .byte  68,15,40,230                        // movaps        %xmm6,%xmm12
   .byte  68,15,92,224                        // subps         %xmm0,%xmm12
@@ -29034,12 +29089,12 @@ _sk_hsl_to_rgb_sse2:
   .byte  15,40,124,36,136                    // movaps        -0x78(%rsp),%xmm7
   .byte  15,40,231                           // movaps        %xmm7,%xmm4
   .byte  15,85,227                           // andnps        %xmm3,%xmm4
-  .byte  15,88,53,172,50,0,0                 // addps         0x32ac(%rip),%xmm6        # 4fb0 <_sk_callback_sse2+0x52f>
+  .byte  15,88,53,162,50,0,0                 // addps         0x32a2(%rip),%xmm6        # 4fd0 <_sk_callback_sse2+0x525>
   .byte  243,15,91,198                       // cvttps2dq     %xmm6,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
   .byte  15,40,222                           // movaps        %xmm6,%xmm3
   .byte  15,194,216,1                        // cmpltps       %xmm0,%xmm3
-  .byte  15,84,29,71,50,0,0                  // andps         0x3247(%rip),%xmm3        # 4f60 <_sk_callback_sse2+0x4df>
+  .byte  15,84,29,61,50,0,0                  // andps         0x323d(%rip),%xmm3        # 4f80 <_sk_callback_sse2+0x4d5>
   .byte  15,92,195                           // subps         %xmm3,%xmm0
   .byte  15,92,240                           // subps         %xmm0,%xmm6
   .byte  15,89,206                           // mulps         %xmm6,%xmm1
@@ -29103,7 +29158,7 @@ _sk_scale_u8_sse2:
   .byte  102,69,15,96,193                    // punpcklbw     %xmm9,%xmm8
   .byte  102,69,15,97,193                    // punpcklwd     %xmm9,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,213,49,0,0               // mulps         0x31d5(%rip),%xmm8        # 4fc0 <_sk_callback_sse2+0x53f>
+  .byte  68,15,89,5,203,49,0,0               // mulps         0x31cb(%rip),%xmm8        # 4fe0 <_sk_callback_sse2+0x535>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
@@ -29144,7 +29199,7 @@ _sk_lerp_u8_sse2:
   .byte  102,69,15,96,193                    // punpcklbw     %xmm9,%xmm8
   .byte  102,69,15,97,193                    // punpcklwd     %xmm9,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,115,49,0,0               // mulps         0x3173(%rip),%xmm8        # 4fd0 <_sk_callback_sse2+0x54f>
+  .byte  68,15,89,5,105,49,0,0               // mulps         0x3169(%rip),%xmm8        # 4ff0 <_sk_callback_sse2+0x545>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -29169,17 +29224,17 @@ _sk_lerp_565_sse2:
   .byte  243,68,15,126,20,120                // movq          (%rax,%rdi,2),%xmm10
   .byte  102,69,15,239,192                   // pxor          %xmm8,%xmm8
   .byte  102,69,15,97,208                    // punpcklwd     %xmm8,%xmm10
-  .byte  102,68,15,111,5,57,49,0,0           // movdqa        0x3139(%rip),%xmm8        # 4fe0 <_sk_callback_sse2+0x55f>
+  .byte  102,68,15,111,5,47,49,0,0           // movdqa        0x312f(%rip),%xmm8        # 5000 <_sk_callback_sse2+0x555>
   .byte  102,69,15,219,194                   // pand          %xmm10,%xmm8
   .byte  69,15,91,192                        // cvtdq2ps      %xmm8,%xmm8
-  .byte  68,15,89,5,56,49,0,0                // mulps         0x3138(%rip),%xmm8        # 4ff0 <_sk_callback_sse2+0x56f>
-  .byte  102,68,15,111,13,63,49,0,0          // movdqa        0x313f(%rip),%xmm9        # 5000 <_sk_callback_sse2+0x57f>
+  .byte  68,15,89,5,46,49,0,0                // mulps         0x312e(%rip),%xmm8        # 5010 <_sk_callback_sse2+0x565>
+  .byte  102,68,15,111,13,53,49,0,0          // movdqa        0x3135(%rip),%xmm9        # 5020 <_sk_callback_sse2+0x575>
   .byte  102,69,15,219,202                   // pand          %xmm10,%xmm9
   .byte  69,15,91,201                        // cvtdq2ps      %xmm9,%xmm9
-  .byte  68,15,89,13,62,49,0,0               // mulps         0x313e(%rip),%xmm9        # 5010 <_sk_callback_sse2+0x58f>
-  .byte  102,68,15,219,21,69,49,0,0          // pand          0x3145(%rip),%xmm10        # 5020 <_sk_callback_sse2+0x59f>
+  .byte  68,15,89,13,52,49,0,0               // mulps         0x3134(%rip),%xmm9        # 5030 <_sk_callback_sse2+0x585>
+  .byte  102,68,15,219,21,59,49,0,0          // pand          0x313b(%rip),%xmm10        # 5040 <_sk_callback_sse2+0x595>
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
-  .byte  68,15,89,21,73,49,0,0               // mulps         0x3149(%rip),%xmm10        # 5030 <_sk_callback_sse2+0x5af>
+  .byte  68,15,89,21,63,49,0,0               // mulps         0x313f(%rip),%xmm10        # 5050 <_sk_callback_sse2+0x5a5>
   .byte  15,92,196                           // subps         %xmm4,%xmm0
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  15,88,196                           // addps         %xmm4,%xmm0
@@ -29210,7 +29265,7 @@ _sk_load_tables_sse2:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  76,139,72,8                         // mov           0x8(%rax),%r9
   .byte  243,69,15,111,12,184                // movdqu        (%r8,%rdi,4),%xmm9
-  .byte  102,68,15,111,5,249,48,0,0          // movdqa        0x30f9(%rip),%xmm8        # 5040 <_sk_callback_sse2+0x5bf>
+  .byte  102,68,15,111,5,239,48,0,0          // movdqa        0x30ef(%rip),%xmm8        # 5060 <_sk_callback_sse2+0x5b5>
   .byte  102,65,15,111,193                   // movdqa        %xmm9,%xmm0
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,15,112,200,78                   // pshufd        $0x4e,%xmm0,%xmm1
@@ -29265,7 +29320,7 @@ _sk_load_tables_sse2:
   .byte  65,15,20,208                        // unpcklps      %xmm8,%xmm2
   .byte  102,65,15,114,209,24                // psrld         $0x18,%xmm9
   .byte  65,15,91,217                        // cvtdq2ps      %xmm9,%xmm3
-  .byte  15,89,29,6,48,0,0                   // mulps         0x3006(%rip),%xmm3        # 5050 <_sk_callback_sse2+0x5cf>
+  .byte  15,89,29,252,47,0,0                 // mulps         0x2ffc(%rip),%xmm3        # 5070 <_sk_callback_sse2+0x5c5>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -29284,7 +29339,7 @@ _sk_load_tables_u16_be_sse2:
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,97,200                       // punpcklwd     %xmm0,%xmm1
   .byte  102,68,15,105,200                   // punpckhwd     %xmm0,%xmm9
-  .byte  102,68,15,111,21,217,47,0,0         // movdqa        0x2fd9(%rip),%xmm10        # 5060 <_sk_callback_sse2+0x5df>
+  .byte  102,68,15,111,21,207,47,0,0         // movdqa        0x2fcf(%rip),%xmm10        # 5080 <_sk_callback_sse2+0x5d5>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,194                   // pand          %xmm10,%xmm0
   .byte  102,69,15,239,192                   // pxor          %xmm8,%xmm8
@@ -29345,7 +29400,7 @@ _sk_load_tables_u16_be_sse2:
   .byte  102,65,15,235,217                   // por           %xmm9,%xmm3
   .byte  102,65,15,97,216                    // punpcklwd     %xmm8,%xmm3
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,200,46,0,0                 // mulps         0x2ec8(%rip),%xmm3        # 5070 <_sk_callback_sse2+0x5ef>
+  .byte  15,89,29,190,46,0,0                 // mulps         0x2ebe(%rip),%xmm3        # 5090 <_sk_callback_sse2+0x5e5>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -29367,7 +29422,7 @@ _sk_load_tables_rgb_u16_be_sse2:
   .byte  102,68,15,97,208                    // punpcklwd     %xmm0,%xmm10
   .byte  102,65,15,111,195                   // movdqa        %xmm11,%xmm0
   .byte  102,65,15,97,194                    // punpcklwd     %xmm10,%xmm0
-  .byte  102,68,15,111,5,136,46,0,0          // movdqa        0x2e88(%rip),%xmm8        # 5080 <_sk_callback_sse2+0x5ff>
+  .byte  102,68,15,111,5,126,46,0,0          // movdqa        0x2e7e(%rip),%xmm8        # 50a0 <_sk_callback_sse2+0x5f5>
   .byte  102,15,112,200,78                   // pshufd        $0x4e,%xmm0,%xmm1
   .byte  102,65,15,219,192                   // pand          %xmm8,%xmm0
   .byte  102,69,15,239,201                   // pxor          %xmm9,%xmm9
@@ -29422,7 +29477,7 @@ _sk_load_tables_rgb_u16_be_sse2:
   .byte  15,20,211                           // unpcklps      %xmm3,%xmm2
   .byte  65,15,20,208                        // unpcklps      %xmm8,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,151,45,0,0                 // movaps        0x2d97(%rip),%xmm3        # 5090 <_sk_callback_sse2+0x60f>
+  .byte  15,40,29,141,45,0,0                 // movaps        0x2d8d(%rip),%xmm3        # 50b0 <_sk_callback_sse2+0x605>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_byte_tables_sse2
@@ -29432,7 +29487,7 @@ _sk_byte_tables_sse2:
   .byte  65,86                               // push          %r14
   .byte  83                                  // push          %rbx
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,152,45,0,0               // movaps        0x2d98(%rip),%xmm8        # 50a0 <_sk_callback_sse2+0x61f>
+  .byte  68,15,40,5,142,45,0,0               // movaps        0x2d8e(%rip),%xmm8        # 50c0 <_sk_callback_sse2+0x615>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,91,192                       // cvtps2dq      %xmm0,%xmm0
   .byte  102,72,15,126,193                   // movq          %xmm0,%rcx
@@ -29459,7 +29514,7 @@ _sk_byte_tables_sse2:
   .byte  102,65,15,96,193                    // punpcklbw     %xmm9,%xmm0
   .byte  102,65,15,97,193                    // punpcklwd     %xmm9,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,21,53,45,0,0               // movaps        0x2d35(%rip),%xmm10        # 50b0 <_sk_callback_sse2+0x62f>
+  .byte  68,15,40,21,43,45,0,0               // movaps        0x2d2b(%rip),%xmm10        # 50d0 <_sk_callback_sse2+0x625>
   .byte  65,15,89,194                        // mulps         %xmm10,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -29575,7 +29630,7 @@ _sk_byte_tables_rgb_sse2:
   .byte  102,65,15,96,193                    // punpcklbw     %xmm9,%xmm0
   .byte  102,65,15,97,193                    // punpcklwd     %xmm9,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,21,136,43,0,0              // movaps        0x2b88(%rip),%xmm10        # 50c0 <_sk_callback_sse2+0x63f>
+  .byte  68,15,40,21,126,43,0,0              // movaps        0x2b7e(%rip),%xmm10        # 50e0 <_sk_callback_sse2+0x635>
   .byte  65,15,89,194                        // mulps         %xmm10,%xmm0
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
   .byte  102,15,91,201                       // cvtps2dq      %xmm1,%xmm1
@@ -29772,15 +29827,15 @@ _sk_parametric_r_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,199,40,0,0              // mulps         0x28c7(%rip),%xmm9        # 50d0 <_sk_callback_sse2+0x64f>
-  .byte  68,15,84,21,207,40,0,0              // andps         0x28cf(%rip),%xmm10        # 50e0 <_sk_callback_sse2+0x65f>
-  .byte  68,15,86,21,215,40,0,0              // orps          0x28d7(%rip),%xmm10        # 50f0 <_sk_callback_sse2+0x66f>
-  .byte  68,15,88,13,223,40,0,0              // addps         0x28df(%rip),%xmm9        # 5100 <_sk_callback_sse2+0x67f>
-  .byte  68,15,40,37,231,40,0,0              // movaps        0x28e7(%rip),%xmm12        # 5110 <_sk_callback_sse2+0x68f>
+  .byte  68,15,89,13,189,40,0,0              // mulps         0x28bd(%rip),%xmm9        # 50f0 <_sk_callback_sse2+0x645>
+  .byte  68,15,84,21,197,40,0,0              // andps         0x28c5(%rip),%xmm10        # 5100 <_sk_callback_sse2+0x655>
+  .byte  68,15,86,21,205,40,0,0              // orps          0x28cd(%rip),%xmm10        # 5110 <_sk_callback_sse2+0x665>
+  .byte  68,15,88,13,213,40,0,0              // addps         0x28d5(%rip),%xmm9        # 5120 <_sk_callback_sse2+0x675>
+  .byte  68,15,40,37,221,40,0,0              // movaps        0x28dd(%rip),%xmm12        # 5130 <_sk_callback_sse2+0x685>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,231,40,0,0              // addps         0x28e7(%rip),%xmm10        # 5120 <_sk_callback_sse2+0x69f>
-  .byte  68,15,40,37,239,40,0,0              // movaps        0x28ef(%rip),%xmm12        # 5130 <_sk_callback_sse2+0x6af>
+  .byte  68,15,88,21,221,40,0,0              // addps         0x28dd(%rip),%xmm10        # 5140 <_sk_callback_sse2+0x695>
+  .byte  68,15,40,37,229,40,0,0              // movaps        0x28e5(%rip),%xmm12        # 5150 <_sk_callback_sse2+0x6a5>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -29788,22 +29843,22 @@ _sk_parametric_r_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,217,40,0,0              // movaps        0x28d9(%rip),%xmm10        # 5140 <_sk_callback_sse2+0x6bf>
+  .byte  68,15,40,21,207,40,0,0              // movaps        0x28cf(%rip),%xmm10        # 5160 <_sk_callback_sse2+0x6b5>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,205,40,0,0              // addps         0x28cd(%rip),%xmm9        # 5150 <_sk_callback_sse2+0x6cf>
-  .byte  68,15,40,37,213,40,0,0              // movaps        0x28d5(%rip),%xmm12        # 5160 <_sk_callback_sse2+0x6df>
+  .byte  68,15,88,13,195,40,0,0              // addps         0x28c3(%rip),%xmm9        # 5170 <_sk_callback_sse2+0x6c5>
+  .byte  68,15,40,37,203,40,0,0              // movaps        0x28cb(%rip),%xmm12        # 5180 <_sk_callback_sse2+0x6d5>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,213,40,0,0              // movaps        0x28d5(%rip),%xmm12        # 5170 <_sk_callback_sse2+0x6ef>
+  .byte  68,15,40,37,203,40,0,0              // movaps        0x28cb(%rip),%xmm12        # 5190 <_sk_callback_sse2+0x6e5>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,217,40,0,0              // movaps        0x28d9(%rip),%xmm13        # 5180 <_sk_callback_sse2+0x6ff>
+  .byte  68,15,40,45,207,40,0,0              // movaps        0x28cf(%rip),%xmm13        # 51a0 <_sk_callback_sse2+0x6f5>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,217,40,0,0              // mulps         0x28d9(%rip),%xmm13        # 5190 <_sk_callback_sse2+0x70f>
+  .byte  68,15,89,45,207,40,0,0              // mulps         0x28cf(%rip),%xmm13        # 51b0 <_sk_callback_sse2+0x705>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -29839,15 +29894,15 @@ _sk_parametric_g_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,89,40,0,0               // mulps         0x2859(%rip),%xmm9        # 51a0 <_sk_callback_sse2+0x71f>
-  .byte  68,15,84,21,97,40,0,0               // andps         0x2861(%rip),%xmm10        # 51b0 <_sk_callback_sse2+0x72f>
-  .byte  68,15,86,21,105,40,0,0              // orps          0x2869(%rip),%xmm10        # 51c0 <_sk_callback_sse2+0x73f>
-  .byte  68,15,88,13,113,40,0,0              // addps         0x2871(%rip),%xmm9        # 51d0 <_sk_callback_sse2+0x74f>
-  .byte  68,15,40,37,121,40,0,0              // movaps        0x2879(%rip),%xmm12        # 51e0 <_sk_callback_sse2+0x75f>
+  .byte  68,15,89,13,79,40,0,0               // mulps         0x284f(%rip),%xmm9        # 51c0 <_sk_callback_sse2+0x715>
+  .byte  68,15,84,21,87,40,0,0               // andps         0x2857(%rip),%xmm10        # 51d0 <_sk_callback_sse2+0x725>
+  .byte  68,15,86,21,95,40,0,0               // orps          0x285f(%rip),%xmm10        # 51e0 <_sk_callback_sse2+0x735>
+  .byte  68,15,88,13,103,40,0,0              // addps         0x2867(%rip),%xmm9        # 51f0 <_sk_callback_sse2+0x745>
+  .byte  68,15,40,37,111,40,0,0              // movaps        0x286f(%rip),%xmm12        # 5200 <_sk_callback_sse2+0x755>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,121,40,0,0              // addps         0x2879(%rip),%xmm10        # 51f0 <_sk_callback_sse2+0x76f>
-  .byte  68,15,40,37,129,40,0,0              // movaps        0x2881(%rip),%xmm12        # 5200 <_sk_callback_sse2+0x77f>
+  .byte  68,15,88,21,111,40,0,0              // addps         0x286f(%rip),%xmm10        # 5210 <_sk_callback_sse2+0x765>
+  .byte  68,15,40,37,119,40,0,0              // movaps        0x2877(%rip),%xmm12        # 5220 <_sk_callback_sse2+0x775>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -29855,22 +29910,22 @@ _sk_parametric_g_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,107,40,0,0              // movaps        0x286b(%rip),%xmm10        # 5210 <_sk_callback_sse2+0x78f>
+  .byte  68,15,40,21,97,40,0,0               // movaps        0x2861(%rip),%xmm10        # 5230 <_sk_callback_sse2+0x785>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,95,40,0,0               // addps         0x285f(%rip),%xmm9        # 5220 <_sk_callback_sse2+0x79f>
-  .byte  68,15,40,37,103,40,0,0              // movaps        0x2867(%rip),%xmm12        # 5230 <_sk_callback_sse2+0x7af>
+  .byte  68,15,88,13,85,40,0,0               // addps         0x2855(%rip),%xmm9        # 5240 <_sk_callback_sse2+0x795>
+  .byte  68,15,40,37,93,40,0,0               // movaps        0x285d(%rip),%xmm12        # 5250 <_sk_callback_sse2+0x7a5>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,103,40,0,0              // movaps        0x2867(%rip),%xmm12        # 5240 <_sk_callback_sse2+0x7bf>
+  .byte  68,15,40,37,93,40,0,0               // movaps        0x285d(%rip),%xmm12        # 5260 <_sk_callback_sse2+0x7b5>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,107,40,0,0              // movaps        0x286b(%rip),%xmm13        # 5250 <_sk_callback_sse2+0x7cf>
+  .byte  68,15,40,45,97,40,0,0               // movaps        0x2861(%rip),%xmm13        # 5270 <_sk_callback_sse2+0x7c5>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,107,40,0,0              // mulps         0x286b(%rip),%xmm13        # 5260 <_sk_callback_sse2+0x7df>
+  .byte  68,15,89,45,97,40,0,0               // mulps         0x2861(%rip),%xmm13        # 5280 <_sk_callback_sse2+0x7d5>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -29906,15 +29961,15 @@ _sk_parametric_b_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,235,39,0,0              // mulps         0x27eb(%rip),%xmm9        # 5270 <_sk_callback_sse2+0x7ef>
-  .byte  68,15,84,21,243,39,0,0              // andps         0x27f3(%rip),%xmm10        # 5280 <_sk_callback_sse2+0x7ff>
-  .byte  68,15,86,21,251,39,0,0              // orps          0x27fb(%rip),%xmm10        # 5290 <_sk_callback_sse2+0x80f>
-  .byte  68,15,88,13,3,40,0,0                // addps         0x2803(%rip),%xmm9        # 52a0 <_sk_callback_sse2+0x81f>
-  .byte  68,15,40,37,11,40,0,0               // movaps        0x280b(%rip),%xmm12        # 52b0 <_sk_callback_sse2+0x82f>
+  .byte  68,15,89,13,225,39,0,0              // mulps         0x27e1(%rip),%xmm9        # 5290 <_sk_callback_sse2+0x7e5>
+  .byte  68,15,84,21,233,39,0,0              // andps         0x27e9(%rip),%xmm10        # 52a0 <_sk_callback_sse2+0x7f5>
+  .byte  68,15,86,21,241,39,0,0              // orps          0x27f1(%rip),%xmm10        # 52b0 <_sk_callback_sse2+0x805>
+  .byte  68,15,88,13,249,39,0,0              // addps         0x27f9(%rip),%xmm9        # 52c0 <_sk_callback_sse2+0x815>
+  .byte  68,15,40,37,1,40,0,0                // movaps        0x2801(%rip),%xmm12        # 52d0 <_sk_callback_sse2+0x825>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,11,40,0,0               // addps         0x280b(%rip),%xmm10        # 52c0 <_sk_callback_sse2+0x83f>
-  .byte  68,15,40,37,19,40,0,0               // movaps        0x2813(%rip),%xmm12        # 52d0 <_sk_callback_sse2+0x84f>
+  .byte  68,15,88,21,1,40,0,0                // addps         0x2801(%rip),%xmm10        # 52e0 <_sk_callback_sse2+0x835>
+  .byte  68,15,40,37,9,40,0,0                // movaps        0x2809(%rip),%xmm12        # 52f0 <_sk_callback_sse2+0x845>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -29922,22 +29977,22 @@ _sk_parametric_b_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,253,39,0,0              // movaps        0x27fd(%rip),%xmm10        # 52e0 <_sk_callback_sse2+0x85f>
+  .byte  68,15,40,21,243,39,0,0              // movaps        0x27f3(%rip),%xmm10        # 5300 <_sk_callback_sse2+0x855>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,241,39,0,0              // addps         0x27f1(%rip),%xmm9        # 52f0 <_sk_callback_sse2+0x86f>
-  .byte  68,15,40,37,249,39,0,0              // movaps        0x27f9(%rip),%xmm12        # 5300 <_sk_callback_sse2+0x87f>
+  .byte  68,15,88,13,231,39,0,0              // addps         0x27e7(%rip),%xmm9        # 5310 <_sk_callback_sse2+0x865>
+  .byte  68,15,40,37,239,39,0,0              // movaps        0x27ef(%rip),%xmm12        # 5320 <_sk_callback_sse2+0x875>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,249,39,0,0              // movaps        0x27f9(%rip),%xmm12        # 5310 <_sk_callback_sse2+0x88f>
+  .byte  68,15,40,37,239,39,0,0              // movaps        0x27ef(%rip),%xmm12        # 5330 <_sk_callback_sse2+0x885>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,253,39,0,0              // movaps        0x27fd(%rip),%xmm13        # 5320 <_sk_callback_sse2+0x89f>
+  .byte  68,15,40,45,243,39,0,0              // movaps        0x27f3(%rip),%xmm13        # 5340 <_sk_callback_sse2+0x895>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,253,39,0,0              // mulps         0x27fd(%rip),%xmm13        # 5330 <_sk_callback_sse2+0x8af>
+  .byte  68,15,89,45,243,39,0,0              // mulps         0x27f3(%rip),%xmm13        # 5350 <_sk_callback_sse2+0x8a5>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -29973,15 +30028,15 @@ _sk_parametric_a_sse2:
   .byte  69,15,88,209                        // addps         %xmm9,%xmm10
   .byte  69,15,198,219,0                     // shufps        $0x0,%xmm11,%xmm11
   .byte  69,15,91,202                        // cvtdq2ps      %xmm10,%xmm9
-  .byte  68,15,89,13,125,39,0,0              // mulps         0x277d(%rip),%xmm9        # 5340 <_sk_callback_sse2+0x8bf>
-  .byte  68,15,84,21,133,39,0,0              // andps         0x2785(%rip),%xmm10        # 5350 <_sk_callback_sse2+0x8cf>
-  .byte  68,15,86,21,141,39,0,0              // orps          0x278d(%rip),%xmm10        # 5360 <_sk_callback_sse2+0x8df>
-  .byte  68,15,88,13,149,39,0,0              // addps         0x2795(%rip),%xmm9        # 5370 <_sk_callback_sse2+0x8ef>
-  .byte  68,15,40,37,157,39,0,0              // movaps        0x279d(%rip),%xmm12        # 5380 <_sk_callback_sse2+0x8ff>
+  .byte  68,15,89,13,115,39,0,0              // mulps         0x2773(%rip),%xmm9        # 5360 <_sk_callback_sse2+0x8b5>
+  .byte  68,15,84,21,123,39,0,0              // andps         0x277b(%rip),%xmm10        # 5370 <_sk_callback_sse2+0x8c5>
+  .byte  68,15,86,21,131,39,0,0              // orps          0x2783(%rip),%xmm10        # 5380 <_sk_callback_sse2+0x8d5>
+  .byte  68,15,88,13,139,39,0,0              // addps         0x278b(%rip),%xmm9        # 5390 <_sk_callback_sse2+0x8e5>
+  .byte  68,15,40,37,147,39,0,0              // movaps        0x2793(%rip),%xmm12        # 53a0 <_sk_callback_sse2+0x8f5>
   .byte  69,15,89,226                        // mulps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,88,21,157,39,0,0              // addps         0x279d(%rip),%xmm10        # 5390 <_sk_callback_sse2+0x90f>
-  .byte  68,15,40,37,165,39,0,0              // movaps        0x27a5(%rip),%xmm12        # 53a0 <_sk_callback_sse2+0x91f>
+  .byte  68,15,88,21,147,39,0,0              // addps         0x2793(%rip),%xmm10        # 53b0 <_sk_callback_sse2+0x905>
+  .byte  68,15,40,37,155,39,0,0              // movaps        0x279b(%rip),%xmm12        # 53c0 <_sk_callback_sse2+0x915>
   .byte  69,15,94,226                        // divps         %xmm10,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
   .byte  69,15,89,203                        // mulps         %xmm11,%xmm9
@@ -29989,22 +30044,22 @@ _sk_parametric_a_sse2:
   .byte  69,15,91,226                        // cvtdq2ps      %xmm10,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,194,236,1                     // cmpltps       %xmm12,%xmm13
-  .byte  68,15,40,21,143,39,0,0              // movaps        0x278f(%rip),%xmm10        # 53b0 <_sk_callback_sse2+0x92f>
+  .byte  68,15,40,21,133,39,0,0              // movaps        0x2785(%rip),%xmm10        # 53d0 <_sk_callback_sse2+0x925>
   .byte  69,15,84,234                        // andps         %xmm10,%xmm13
   .byte  69,15,87,219                        // xorps         %xmm11,%xmm11
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
   .byte  69,15,40,233                        // movaps        %xmm9,%xmm13
   .byte  69,15,92,236                        // subps         %xmm12,%xmm13
-  .byte  68,15,88,13,131,39,0,0              // addps         0x2783(%rip),%xmm9        # 53c0 <_sk_callback_sse2+0x93f>
-  .byte  68,15,40,37,139,39,0,0              // movaps        0x278b(%rip),%xmm12        # 53d0 <_sk_callback_sse2+0x94f>
+  .byte  68,15,88,13,121,39,0,0              // addps         0x2779(%rip),%xmm9        # 53e0 <_sk_callback_sse2+0x935>
+  .byte  68,15,40,37,129,39,0,0              // movaps        0x2781(%rip),%xmm12        # 53f0 <_sk_callback_sse2+0x945>
   .byte  69,15,89,229                        // mulps         %xmm13,%xmm12
   .byte  69,15,92,204                        // subps         %xmm12,%xmm9
-  .byte  68,15,40,37,139,39,0,0              // movaps        0x278b(%rip),%xmm12        # 53e0 <_sk_callback_sse2+0x95f>
+  .byte  68,15,40,37,129,39,0,0              // movaps        0x2781(%rip),%xmm12        # 5400 <_sk_callback_sse2+0x955>
   .byte  69,15,92,229                        // subps         %xmm13,%xmm12
-  .byte  68,15,40,45,143,39,0,0              // movaps        0x278f(%rip),%xmm13        # 53f0 <_sk_callback_sse2+0x96f>
+  .byte  68,15,40,45,133,39,0,0              // movaps        0x2785(%rip),%xmm13        # 5410 <_sk_callback_sse2+0x965>
   .byte  69,15,94,236                        // divps         %xmm12,%xmm13
   .byte  69,15,88,233                        // addps         %xmm9,%xmm13
-  .byte  68,15,89,45,143,39,0,0              // mulps         0x278f(%rip),%xmm13        # 5400 <_sk_callback_sse2+0x97f>
+  .byte  68,15,89,45,133,39,0,0              // mulps         0x2785(%rip),%xmm13        # 5420 <_sk_callback_sse2+0x975>
   .byte  102,69,15,91,205                    // cvtps2dq      %xmm13,%xmm9
   .byte  243,68,15,16,96,20                  // movss         0x14(%rax),%xmm12
   .byte  69,15,198,228,0                     // shufps        $0x0,%xmm12,%xmm12
@@ -30021,29 +30076,29 @@ HIDDEN _sk_lab_to_xyz_sse2
 .globl _sk_lab_to_xyz_sse2
 FUNCTION(_sk_lab_to_xyz_sse2)
 _sk_lab_to_xyz_sse2:
-  .byte  15,89,5,108,39,0,0                  // mulps         0x276c(%rip),%xmm0        # 5410 <_sk_callback_sse2+0x98f>
-  .byte  68,15,40,5,116,39,0,0               // movaps        0x2774(%rip),%xmm8        # 5420 <_sk_callback_sse2+0x99f>
+  .byte  15,89,5,98,39,0,0                   // mulps         0x2762(%rip),%xmm0        # 5430 <_sk_callback_sse2+0x985>
+  .byte  68,15,40,5,106,39,0,0               // movaps        0x276a(%rip),%xmm8        # 5440 <_sk_callback_sse2+0x995>
   .byte  65,15,89,200                        // mulps         %xmm8,%xmm1
-  .byte  68,15,40,13,120,39,0,0              // movaps        0x2778(%rip),%xmm9        # 5430 <_sk_callback_sse2+0x9af>
+  .byte  68,15,40,13,110,39,0,0              // movaps        0x276e(%rip),%xmm9        # 5450 <_sk_callback_sse2+0x9a5>
   .byte  65,15,88,201                        // addps         %xmm9,%xmm1
   .byte  65,15,89,208                        // mulps         %xmm8,%xmm2
   .byte  65,15,88,209                        // addps         %xmm9,%xmm2
-  .byte  15,88,5,117,39,0,0                  // addps         0x2775(%rip),%xmm0        # 5440 <_sk_callback_sse2+0x9bf>
-  .byte  15,89,5,126,39,0,0                  // mulps         0x277e(%rip),%xmm0        # 5450 <_sk_callback_sse2+0x9cf>
-  .byte  15,89,13,135,39,0,0                 // mulps         0x2787(%rip),%xmm1        # 5460 <_sk_callback_sse2+0x9df>
+  .byte  15,88,5,107,39,0,0                  // addps         0x276b(%rip),%xmm0        # 5460 <_sk_callback_sse2+0x9b5>
+  .byte  15,89,5,116,39,0,0                  // mulps         0x2774(%rip),%xmm0        # 5470 <_sk_callback_sse2+0x9c5>
+  .byte  15,89,13,125,39,0,0                 // mulps         0x277d(%rip),%xmm1        # 5480 <_sk_callback_sse2+0x9d5>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  15,89,21,141,39,0,0                 // mulps         0x278d(%rip),%xmm2        # 5470 <_sk_callback_sse2+0x9ef>
+  .byte  15,89,21,131,39,0,0                 // mulps         0x2783(%rip),%xmm2        # 5490 <_sk_callback_sse2+0x9e5>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  68,15,92,202                        // subps         %xmm2,%xmm9
   .byte  68,15,40,225                        // movaps        %xmm1,%xmm12
   .byte  69,15,89,228                        // mulps         %xmm12,%xmm12
   .byte  68,15,89,225                        // mulps         %xmm1,%xmm12
-  .byte  15,40,21,130,39,0,0                 // movaps        0x2782(%rip),%xmm2        # 5480 <_sk_callback_sse2+0x9ff>
+  .byte  15,40,21,120,39,0,0                 // movaps        0x2778(%rip),%xmm2        # 54a0 <_sk_callback_sse2+0x9f5>
   .byte  68,15,40,194                        // movaps        %xmm2,%xmm8
   .byte  69,15,194,196,1                     // cmpltps       %xmm12,%xmm8
-  .byte  68,15,40,21,129,39,0,0              // movaps        0x2781(%rip),%xmm10        # 5490 <_sk_callback_sse2+0xa0f>
+  .byte  68,15,40,21,119,39,0,0              // movaps        0x2777(%rip),%xmm10        # 54b0 <_sk_callback_sse2+0xa05>
   .byte  65,15,88,202                        // addps         %xmm10,%xmm1
-  .byte  68,15,40,29,133,39,0,0              // movaps        0x2785(%rip),%xmm11        # 54a0 <_sk_callback_sse2+0xa1f>
+  .byte  68,15,40,29,123,39,0,0              // movaps        0x277b(%rip),%xmm11        # 54c0 <_sk_callback_sse2+0xa15>
   .byte  65,15,89,203                        // mulps         %xmm11,%xmm1
   .byte  69,15,84,224                        // andps         %xmm8,%xmm12
   .byte  68,15,85,193                        // andnps        %xmm1,%xmm8
@@ -30067,8 +30122,8 @@ _sk_lab_to_xyz_sse2:
   .byte  15,84,194                           // andps         %xmm2,%xmm0
   .byte  65,15,85,209                        // andnps        %xmm9,%xmm2
   .byte  15,86,208                           // orps          %xmm0,%xmm2
-  .byte  68,15,89,5,53,39,0,0                // mulps         0x2735(%rip),%xmm8        # 54b0 <_sk_callback_sse2+0xa2f>
-  .byte  15,89,21,62,39,0,0                  // mulps         0x273e(%rip),%xmm2        # 54c0 <_sk_callback_sse2+0xa3f>
+  .byte  68,15,89,5,43,39,0,0                // mulps         0x272b(%rip),%xmm8        # 54d0 <_sk_callback_sse2+0xa25>
+  .byte  15,89,21,52,39,0,0                  // mulps         0x2734(%rip),%xmm2        # 54e0 <_sk_callback_sse2+0xa35>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  65,15,40,192                        // movaps        %xmm8,%xmm0
   .byte  255,224                             // jmpq          *%rax
@@ -30084,7 +30139,7 @@ _sk_load_a8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,38,39,0,0                  // mulps         0x2726(%rip),%xmm3        # 54d0 <_sk_callback_sse2+0xa4f>
+  .byte  15,89,29,28,39,0,0                  // mulps         0x271c(%rip),%xmm3        # 54f0 <_sk_callback_sse2+0xa45>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
@@ -30129,7 +30184,7 @@ _sk_gather_a8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,216                           // cvtdq2ps      %xmm0,%xmm3
-  .byte  15,89,29,149,38,0,0                 // mulps         0x2695(%rip),%xmm3        # 54e0 <_sk_callback_sse2+0xa5f>
+  .byte  15,89,29,139,38,0,0                 // mulps         0x268b(%rip),%xmm3        # 5500 <_sk_callback_sse2+0xa55>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
@@ -30142,7 +30197,7 @@ FUNCTION(_sk_store_a8_sse2)
 _sk_store_a8_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,137,38,0,0               // movaps        0x2689(%rip),%xmm8        # 54f0 <_sk_callback_sse2+0xa6f>
+  .byte  68,15,40,5,127,38,0,0               // movaps        0x267f(%rip),%xmm8        # 5510 <_sk_callback_sse2+0xa65>
   .byte  68,15,89,195                        // mulps         %xmm3,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
   .byte  102,65,15,114,240,16                // pslld         $0x10,%xmm8
@@ -30164,9 +30219,9 @@ _sk_load_g8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,80,38,0,0                   // mulps         0x2650(%rip),%xmm0        # 5500 <_sk_callback_sse2+0xa7f>
+  .byte  15,89,5,70,38,0,0                   // mulps         0x2646(%rip),%xmm0        # 5520 <_sk_callback_sse2+0xa75>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,87,38,0,0                  // movaps        0x2657(%rip),%xmm3        # 5510 <_sk_callback_sse2+0xa8f>
+  .byte  15,40,29,77,38,0,0                  // movaps        0x264d(%rip),%xmm3        # 5530 <_sk_callback_sse2+0xa85>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -30209,9 +30264,9 @@ _sk_gather_g8_sse2:
   .byte  102,15,96,193                       // punpcklbw     %xmm1,%xmm0
   .byte  102,15,97,193                       // punpcklwd     %xmm1,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,204,37,0,0                  // mulps         0x25cc(%rip),%xmm0        # 5520 <_sk_callback_sse2+0xa9f>
+  .byte  15,89,5,194,37,0,0                  // mulps         0x25c2(%rip),%xmm0        # 5540 <_sk_callback_sse2+0xa95>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,211,37,0,0                 // movaps        0x25d3(%rip),%xmm3        # 5530 <_sk_callback_sse2+0xaaf>
+  .byte  15,40,29,201,37,0,0                 // movaps        0x25c9(%rip),%xmm3        # 5550 <_sk_callback_sse2+0xaa5>
   .byte  15,40,200                           // movaps        %xmm0,%xmm1
   .byte  15,40,208                           // movaps        %xmm0,%xmm2
   .byte  255,224                             // jmpq          *%rax
@@ -30223,9 +30278,9 @@ _sk_gather_i8_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  73,137,192                          // mov           %rax,%r8
   .byte  77,133,192                          // test          %r8,%r8
-  .byte  116,5                               // je            2f74 <_sk_gather_i8_sse2+0xf>
+  .byte  116,5                               // je            2f9e <_sk_gather_i8_sse2+0xf>
   .byte  76,137,192                          // mov           %r8,%rax
-  .byte  235,2                               // jmp           2f76 <_sk_gather_i8_sse2+0x11>
+  .byte  235,2                               // jmp           2fa0 <_sk_gather_i8_sse2+0x11>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  76,139,16                           // mov           (%rax),%r10
   .byte  243,15,91,201                       // cvttps2dq     %xmm1,%xmm1
@@ -30274,11 +30329,11 @@ _sk_gather_i8_sse2:
   .byte  102,67,15,110,12,136                // movd          (%r8,%r9,4),%xmm1
   .byte  102,68,15,98,201                    // punpckldq     %xmm1,%xmm9
   .byte  102,68,15,98,200                    // punpckldq     %xmm0,%xmm9
-  .byte  102,15,111,21,242,36,0,0            // movdqa        0x24f2(%rip),%xmm2        # 5540 <_sk_callback_sse2+0xabf>
+  .byte  102,15,111,21,232,36,0,0            // movdqa        0x24e8(%rip),%xmm2        # 5560 <_sk_callback_sse2+0xab5>
   .byte  102,65,15,111,193                   // movdqa        %xmm9,%xmm0
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,238,36,0,0               // movaps        0x24ee(%rip),%xmm8        # 5550 <_sk_callback_sse2+0xacf>
+  .byte  68,15,40,5,228,36,0,0               // movaps        0x24e4(%rip),%xmm8        # 5570 <_sk_callback_sse2+0xac5>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,114,209,8                    // psrld         $0x8,%xmm1
@@ -30305,19 +30360,19 @@ _sk_load_565_sse2:
   .byte  243,15,126,20,120                   // movq          (%rax,%rdi,2),%xmm2
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,208                       // punpcklwd     %xmm0,%xmm2
-  .byte  102,15,111,5,164,36,0,0             // movdqa        0x24a4(%rip),%xmm0        # 5560 <_sk_callback_sse2+0xadf>
+  .byte  102,15,111,5,154,36,0,0             // movdqa        0x249a(%rip),%xmm0        # 5580 <_sk_callback_sse2+0xad5>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,166,36,0,0                  // mulps         0x24a6(%rip),%xmm0        # 5570 <_sk_callback_sse2+0xaef>
-  .byte  102,15,111,13,174,36,0,0            // movdqa        0x24ae(%rip),%xmm1        # 5580 <_sk_callback_sse2+0xaff>
+  .byte  15,89,5,156,36,0,0                  // mulps         0x249c(%rip),%xmm0        # 5590 <_sk_callback_sse2+0xae5>
+  .byte  102,15,111,13,164,36,0,0            // movdqa        0x24a4(%rip),%xmm1        # 55a0 <_sk_callback_sse2+0xaf5>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,176,36,0,0                 // mulps         0x24b0(%rip),%xmm1        # 5590 <_sk_callback_sse2+0xb0f>
-  .byte  102,15,219,21,184,36,0,0            // pand          0x24b8(%rip),%xmm2        # 55a0 <_sk_callback_sse2+0xb1f>
+  .byte  15,89,13,166,36,0,0                 // mulps         0x24a6(%rip),%xmm1        # 55b0 <_sk_callback_sse2+0xb05>
+  .byte  102,15,219,21,174,36,0,0            // pand          0x24ae(%rip),%xmm2        # 55c0 <_sk_callback_sse2+0xb15>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,190,36,0,0                 // mulps         0x24be(%rip),%xmm2        # 55b0 <_sk_callback_sse2+0xb2f>
+  .byte  15,89,21,180,36,0,0                 // mulps         0x24b4(%rip),%xmm2        # 55d0 <_sk_callback_sse2+0xb25>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,197,36,0,0                 // movaps        0x24c5(%rip),%xmm3        # 55c0 <_sk_callback_sse2+0xb3f>
+  .byte  15,40,29,187,36,0,0                 // movaps        0x24bb(%rip),%xmm3        # 55e0 <_sk_callback_sse2+0xb35>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_gather_565_sse2
@@ -30352,19 +30407,19 @@ _sk_gather_565_sse2:
   .byte  102,15,196,208,3                    // pinsrw        $0x3,%eax,%xmm2
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,208                       // punpcklwd     %xmm0,%xmm2
-  .byte  102,15,111,5,78,36,0,0              // movdqa        0x244e(%rip),%xmm0        # 55d0 <_sk_callback_sse2+0xb4f>
+  .byte  102,15,111,5,68,36,0,0              // movdqa        0x2444(%rip),%xmm0        # 55f0 <_sk_callback_sse2+0xb45>
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,80,36,0,0                   // mulps         0x2450(%rip),%xmm0        # 55e0 <_sk_callback_sse2+0xb5f>
-  .byte  102,15,111,13,88,36,0,0             // movdqa        0x2458(%rip),%xmm1        # 55f0 <_sk_callback_sse2+0xb6f>
+  .byte  15,89,5,70,36,0,0                   // mulps         0x2446(%rip),%xmm0        # 5600 <_sk_callback_sse2+0xb55>
+  .byte  102,15,111,13,78,36,0,0             // movdqa        0x244e(%rip),%xmm1        # 5610 <_sk_callback_sse2+0xb65>
   .byte  102,15,219,202                      // pand          %xmm2,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,90,36,0,0                  // mulps         0x245a(%rip),%xmm1        # 5600 <_sk_callback_sse2+0xb7f>
-  .byte  102,15,219,21,98,36,0,0             // pand          0x2462(%rip),%xmm2        # 5610 <_sk_callback_sse2+0xb8f>
+  .byte  15,89,13,80,36,0,0                  // mulps         0x2450(%rip),%xmm1        # 5620 <_sk_callback_sse2+0xb75>
+  .byte  102,15,219,21,88,36,0,0             // pand          0x2458(%rip),%xmm2        # 5630 <_sk_callback_sse2+0xb85>
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,104,36,0,0                 // mulps         0x2468(%rip),%xmm2        # 5620 <_sk_callback_sse2+0xb9f>
+  .byte  15,89,21,94,36,0,0                  // mulps         0x245e(%rip),%xmm2        # 5640 <_sk_callback_sse2+0xb95>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,111,36,0,0                 // movaps        0x246f(%rip),%xmm3        # 5630 <_sk_callback_sse2+0xbaf>
+  .byte  15,40,29,101,36,0,0                 // movaps        0x2465(%rip),%xmm3        # 5650 <_sk_callback_sse2+0xba5>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_565_sse2
@@ -30373,12 +30428,12 @@ FUNCTION(_sk_store_565_sse2)
 _sk_store_565_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,112,36,0,0               // movaps        0x2470(%rip),%xmm8        # 5640 <_sk_callback_sse2+0xbbf>
+  .byte  68,15,40,5,102,36,0,0               // movaps        0x2466(%rip),%xmm8        # 5660 <_sk_callback_sse2+0xbb5>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
   .byte  102,65,15,114,241,11                // pslld         $0xb,%xmm9
-  .byte  68,15,40,21,101,36,0,0              // movaps        0x2465(%rip),%xmm10        # 5650 <_sk_callback_sse2+0xbcf>
+  .byte  68,15,40,21,91,36,0,0               // movaps        0x245b(%rip),%xmm10        # 5670 <_sk_callback_sse2+0xbc5>
   .byte  68,15,89,209                        // mulps         %xmm1,%xmm10
   .byte  102,69,15,91,210                    // cvtps2dq      %xmm10,%xmm10
   .byte  102,65,15,114,242,5                 // pslld         $0x5,%xmm10
@@ -30402,21 +30457,21 @@ _sk_load_4444_sse2:
   .byte  243,15,126,28,120                   // movq          (%rax,%rdi,2),%xmm3
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,216                       // punpcklwd     %xmm0,%xmm3
-  .byte  102,15,111,5,30,36,0,0              // movdqa        0x241e(%rip),%xmm0        # 5660 <_sk_callback_sse2+0xbdf>
+  .byte  102,15,111,5,20,36,0,0              // movdqa        0x2414(%rip),%xmm0        # 5680 <_sk_callback_sse2+0xbd5>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,32,36,0,0                   // mulps         0x2420(%rip),%xmm0        # 5670 <_sk_callback_sse2+0xbef>
-  .byte  102,15,111,13,40,36,0,0             // movdqa        0x2428(%rip),%xmm1        # 5680 <_sk_callback_sse2+0xbff>
+  .byte  15,89,5,22,36,0,0                   // mulps         0x2416(%rip),%xmm0        # 5690 <_sk_callback_sse2+0xbe5>
+  .byte  102,15,111,13,30,36,0,0             // movdqa        0x241e(%rip),%xmm1        # 56a0 <_sk_callback_sse2+0xbf5>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,42,36,0,0                  // mulps         0x242a(%rip),%xmm1        # 5690 <_sk_callback_sse2+0xc0f>
-  .byte  102,15,111,21,50,36,0,0             // movdqa        0x2432(%rip),%xmm2        # 56a0 <_sk_callback_sse2+0xc1f>
+  .byte  15,89,13,32,36,0,0                  // mulps         0x2420(%rip),%xmm1        # 56b0 <_sk_callback_sse2+0xc05>
+  .byte  102,15,111,21,40,36,0,0             // movdqa        0x2428(%rip),%xmm2        # 56c0 <_sk_callback_sse2+0xc15>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,52,36,0,0                  // mulps         0x2434(%rip),%xmm2        # 56b0 <_sk_callback_sse2+0xc2f>
-  .byte  102,15,219,29,60,36,0,0             // pand          0x243c(%rip),%xmm3        # 56c0 <_sk_callback_sse2+0xc3f>
+  .byte  15,89,21,42,36,0,0                  // mulps         0x242a(%rip),%xmm2        # 56d0 <_sk_callback_sse2+0xc25>
+  .byte  102,15,219,29,50,36,0,0             // pand          0x2432(%rip),%xmm3        # 56e0 <_sk_callback_sse2+0xc35>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,66,36,0,0                  // mulps         0x2442(%rip),%xmm3        # 56d0 <_sk_callback_sse2+0xc4f>
+  .byte  15,89,29,56,36,0,0                  // mulps         0x2438(%rip),%xmm3        # 56f0 <_sk_callback_sse2+0xc45>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -30452,21 +30507,21 @@ _sk_gather_4444_sse2:
   .byte  102,15,196,216,3                    // pinsrw        $0x3,%eax,%xmm3
   .byte  102,15,239,192                      // pxor          %xmm0,%xmm0
   .byte  102,15,97,216                       // punpcklwd     %xmm0,%xmm3
-  .byte  102,15,111,5,201,35,0,0             // movdqa        0x23c9(%rip),%xmm0        # 56e0 <_sk_callback_sse2+0xc5f>
+  .byte  102,15,111,5,191,35,0,0             // movdqa        0x23bf(%rip),%xmm0        # 5700 <_sk_callback_sse2+0xc55>
   .byte  102,15,219,195                      // pand          %xmm3,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  15,89,5,203,35,0,0                  // mulps         0x23cb(%rip),%xmm0        # 56f0 <_sk_callback_sse2+0xc6f>
-  .byte  102,15,111,13,211,35,0,0            // movdqa        0x23d3(%rip),%xmm1        # 5700 <_sk_callback_sse2+0xc7f>
+  .byte  15,89,5,193,35,0,0                  // mulps         0x23c1(%rip),%xmm0        # 5710 <_sk_callback_sse2+0xc65>
+  .byte  102,15,111,13,201,35,0,0            // movdqa        0x23c9(%rip),%xmm1        # 5720 <_sk_callback_sse2+0xc75>
   .byte  102,15,219,203                      // pand          %xmm3,%xmm1
   .byte  15,91,201                           // cvtdq2ps      %xmm1,%xmm1
-  .byte  15,89,13,213,35,0,0                 // mulps         0x23d5(%rip),%xmm1        # 5710 <_sk_callback_sse2+0xc8f>
-  .byte  102,15,111,21,221,35,0,0            // movdqa        0x23dd(%rip),%xmm2        # 5720 <_sk_callback_sse2+0xc9f>
+  .byte  15,89,13,203,35,0,0                 // mulps         0x23cb(%rip),%xmm1        # 5730 <_sk_callback_sse2+0xc85>
+  .byte  102,15,111,21,211,35,0,0            // movdqa        0x23d3(%rip),%xmm2        # 5740 <_sk_callback_sse2+0xc95>
   .byte  102,15,219,211                      // pand          %xmm3,%xmm2
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
-  .byte  15,89,21,223,35,0,0                 // mulps         0x23df(%rip),%xmm2        # 5730 <_sk_callback_sse2+0xcaf>
-  .byte  102,15,219,29,231,35,0,0            // pand          0x23e7(%rip),%xmm3        # 5740 <_sk_callback_sse2+0xcbf>
+  .byte  15,89,21,213,35,0,0                 // mulps         0x23d5(%rip),%xmm2        # 5750 <_sk_callback_sse2+0xca5>
+  .byte  102,15,219,29,221,35,0,0            // pand          0x23dd(%rip),%xmm3        # 5760 <_sk_callback_sse2+0xcb5>
   .byte  15,91,219                           // cvtdq2ps      %xmm3,%xmm3
-  .byte  15,89,29,237,35,0,0                 // mulps         0x23ed(%rip),%xmm3        # 5750 <_sk_callback_sse2+0xccf>
+  .byte  15,89,29,227,35,0,0                 // mulps         0x23e3(%rip),%xmm3        # 5770 <_sk_callback_sse2+0xcc5>
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
 
@@ -30476,7 +30531,7 @@ FUNCTION(_sk_store_4444_sse2)
 _sk_store_4444_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,236,35,0,0               // movaps        0x23ec(%rip),%xmm8        # 5760 <_sk_callback_sse2+0xcdf>
+  .byte  68,15,40,5,226,35,0,0               // movaps        0x23e2(%rip),%xmm8        # 5780 <_sk_callback_sse2+0xcd5>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -30508,11 +30563,11 @@ _sk_load_8888_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
   .byte  68,15,16,12,184                     // movups        (%rax,%rdi,4),%xmm9
-  .byte  15,40,21,127,35,0,0                 // movaps        0x237f(%rip),%xmm2        # 5770 <_sk_callback_sse2+0xcef>
+  .byte  15,40,21,117,35,0,0                 // movaps        0x2375(%rip),%xmm2        # 5790 <_sk_callback_sse2+0xce5>
   .byte  65,15,40,193                        // movaps        %xmm9,%xmm0
   .byte  15,84,194                           // andps         %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,125,35,0,0               // movaps        0x237d(%rip),%xmm8        # 5780 <_sk_callback_sse2+0xcff>
+  .byte  68,15,40,5,115,35,0,0               // movaps        0x2373(%rip),%xmm8        # 57a0 <_sk_callback_sse2+0xcf5>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  65,15,40,201                        // movaps        %xmm9,%xmm1
   .byte  102,15,114,209,8                    // psrld         $0x8,%xmm1
@@ -30561,11 +30616,11 @@ _sk_gather_8888_sse2:
   .byte  102,67,15,110,12,129                // movd          (%r9,%r8,4),%xmm1
   .byte  102,68,15,98,201                    // punpckldq     %xmm1,%xmm9
   .byte  102,68,15,98,200                    // punpckldq     %xmm0,%xmm9
-  .byte  102,15,111,21,206,34,0,0            // movdqa        0x22ce(%rip),%xmm2        # 5790 <_sk_callback_sse2+0xd0f>
+  .byte  102,15,111,21,196,34,0,0            // movdqa        0x22c4(%rip),%xmm2        # 57b0 <_sk_callback_sse2+0xd05>
   .byte  102,65,15,111,193                   // movdqa        %xmm9,%xmm0
   .byte  102,15,219,194                      // pand          %xmm2,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,5,202,34,0,0               // movaps        0x22ca(%rip),%xmm8        # 57a0 <_sk_callback_sse2+0xd1f>
+  .byte  68,15,40,5,192,34,0,0               // movaps        0x22c0(%rip),%xmm8        # 57c0 <_sk_callback_sse2+0xd15>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,65,15,111,201                   // movdqa        %xmm9,%xmm1
   .byte  102,15,114,209,8                    // psrld         $0x8,%xmm1
@@ -30589,7 +30644,7 @@ FUNCTION(_sk_store_8888_sse2)
 _sk_store_8888_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,5,141,34,0,0               // movaps        0x228d(%rip),%xmm8        # 57b0 <_sk_callback_sse2+0xd2f>
+  .byte  68,15,40,5,131,34,0,0               // movaps        0x2283(%rip),%xmm8        # 57d0 <_sk_callback_sse2+0xd25>
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  102,69,15,91,201                    // cvtps2dq      %xmm9,%xmm9
@@ -30628,7 +30683,7 @@ _sk_load_f16_sse2:
   .byte  102,69,15,239,210                   // pxor          %xmm10,%xmm10
   .byte  102,65,15,111,206                   // movdqa        %xmm14,%xmm1
   .byte  102,65,15,97,202                    // punpcklwd     %xmm10,%xmm1
-  .byte  102,68,15,111,13,253,33,0,0         // movdqa        0x21fd(%rip),%xmm9        # 57c0 <_sk_callback_sse2+0xd3f>
+  .byte  102,68,15,111,13,243,33,0,0         // movdqa        0x21f3(%rip),%xmm9        # 57e0 <_sk_callback_sse2+0xd35>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,193                   // pand          %xmm9,%xmm0
   .byte  102,15,239,200                      // pxor          %xmm0,%xmm1
@@ -30636,11 +30691,11 @@ _sk_load_f16_sse2:
   .byte  102,68,15,111,233                   // movdqa        %xmm1,%xmm13
   .byte  102,65,15,114,245,13                // pslld         $0xd,%xmm13
   .byte  102,68,15,235,232                   // por           %xmm0,%xmm13
-  .byte  102,68,15,111,29,226,33,0,0         // movdqa        0x21e2(%rip),%xmm11        # 57d0 <_sk_callback_sse2+0xd4f>
+  .byte  102,68,15,111,29,216,33,0,0         // movdqa        0x21d8(%rip),%xmm11        # 57f0 <_sk_callback_sse2+0xd45>
   .byte  102,69,15,254,235                   // paddd         %xmm11,%xmm13
-  .byte  102,68,15,111,37,228,33,0,0         // movdqa        0x21e4(%rip),%xmm12        # 57e0 <_sk_callback_sse2+0xd5f>
+  .byte  102,68,15,111,37,218,33,0,0         // movdqa        0x21da(%rip),%xmm12        # 5800 <_sk_callback_sse2+0xd55>
   .byte  102,65,15,239,204                   // pxor          %xmm12,%xmm1
-  .byte  102,15,111,29,231,33,0,0            // movdqa        0x21e7(%rip),%xmm3        # 57f0 <_sk_callback_sse2+0xd6f>
+  .byte  102,15,111,29,221,33,0,0            // movdqa        0x21dd(%rip),%xmm3        # 5810 <_sk_callback_sse2+0xd65>
   .byte  102,15,111,195                      // movdqa        %xmm3,%xmm0
   .byte  102,15,102,193                      // pcmpgtd       %xmm1,%xmm0
   .byte  102,65,15,223,197                   // pandn         %xmm13,%xmm0
@@ -30726,7 +30781,7 @@ _sk_gather_f16_sse2:
   .byte  102,69,15,239,210                   // pxor          %xmm10,%xmm10
   .byte  102,65,15,111,206                   // movdqa        %xmm14,%xmm1
   .byte  102,65,15,97,202                    // punpcklwd     %xmm10,%xmm1
-  .byte  102,68,15,111,13,117,32,0,0         // movdqa        0x2075(%rip),%xmm9        # 5800 <_sk_callback_sse2+0xd7f>
+  .byte  102,68,15,111,13,107,32,0,0         // movdqa        0x206b(%rip),%xmm9        # 5820 <_sk_callback_sse2+0xd75>
   .byte  102,15,111,193                      // movdqa        %xmm1,%xmm0
   .byte  102,65,15,219,193                   // pand          %xmm9,%xmm0
   .byte  102,15,239,200                      // pxor          %xmm0,%xmm1
@@ -30734,11 +30789,11 @@ _sk_gather_f16_sse2:
   .byte  102,68,15,111,233                   // movdqa        %xmm1,%xmm13
   .byte  102,65,15,114,245,13                // pslld         $0xd,%xmm13
   .byte  102,68,15,235,232                   // por           %xmm0,%xmm13
-  .byte  102,68,15,111,29,90,32,0,0          // movdqa        0x205a(%rip),%xmm11        # 5810 <_sk_callback_sse2+0xd8f>
+  .byte  102,68,15,111,29,80,32,0,0          // movdqa        0x2050(%rip),%xmm11        # 5830 <_sk_callback_sse2+0xd85>
   .byte  102,69,15,254,235                   // paddd         %xmm11,%xmm13
-  .byte  102,68,15,111,37,92,32,0,0          // movdqa        0x205c(%rip),%xmm12        # 5820 <_sk_callback_sse2+0xd9f>
+  .byte  102,68,15,111,37,82,32,0,0          // movdqa        0x2052(%rip),%xmm12        # 5840 <_sk_callback_sse2+0xd95>
   .byte  102,65,15,239,204                   // pxor          %xmm12,%xmm1
-  .byte  102,15,111,29,95,32,0,0             // movdqa        0x205f(%rip),%xmm3        # 5830 <_sk_callback_sse2+0xdaf>
+  .byte  102,15,111,29,85,32,0,0             // movdqa        0x2055(%rip),%xmm3        # 5850 <_sk_callback_sse2+0xda5>
   .byte  102,15,111,195                      // movdqa        %xmm3,%xmm0
   .byte  102,15,102,193                      // pcmpgtd       %xmm1,%xmm0
   .byte  102,65,15,223,197                   // pandn         %xmm13,%xmm0
@@ -30791,17 +30846,17 @@ FUNCTION(_sk_store_f16_sse2)
 _sk_store_f16_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  102,68,15,111,21,135,31,0,0         // movdqa        0x1f87(%rip),%xmm10        # 5840 <_sk_callback_sse2+0xdbf>
+  .byte  102,68,15,111,21,125,31,0,0         // movdqa        0x1f7d(%rip),%xmm10        # 5860 <_sk_callback_sse2+0xdb5>
   .byte  102,68,15,111,224                   // movdqa        %xmm0,%xmm12
   .byte  102,68,15,111,232                   // movdqa        %xmm0,%xmm13
   .byte  102,69,15,219,234                   // pand          %xmm10,%xmm13
   .byte  102,69,15,239,229                   // pxor          %xmm13,%xmm12
-  .byte  102,68,15,111,13,122,31,0,0         // movdqa        0x1f7a(%rip),%xmm9        # 5850 <_sk_callback_sse2+0xdcf>
+  .byte  102,68,15,111,13,112,31,0,0         // movdqa        0x1f70(%rip),%xmm9        # 5870 <_sk_callback_sse2+0xdc5>
   .byte  102,65,15,114,213,16                // psrld         $0x10,%xmm13
   .byte  102,69,15,111,193                   // movdqa        %xmm9,%xmm8
   .byte  102,69,15,102,196                   // pcmpgtd       %xmm12,%xmm8
   .byte  102,65,15,114,212,13                // psrld         $0xd,%xmm12
-  .byte  102,68,15,111,29,107,31,0,0         // movdqa        0x1f6b(%rip),%xmm11        # 5860 <_sk_callback_sse2+0xddf>
+  .byte  102,68,15,111,29,97,31,0,0          // movdqa        0x1f61(%rip),%xmm11        # 5880 <_sk_callback_sse2+0xdd5>
   .byte  102,69,15,235,235                   // por           %xmm11,%xmm13
   .byte  102,69,15,254,236                   // paddd         %xmm12,%xmm13
   .byte  102,65,15,114,245,16                // pslld         $0x10,%xmm13
@@ -30880,7 +30935,7 @@ _sk_load_u16_be_sse2:
   .byte  102,69,15,239,201                   // pxor          %xmm9,%xmm9
   .byte  102,65,15,97,201                    // punpcklwd     %xmm9,%xmm1
   .byte  15,91,193                           // cvtdq2ps      %xmm1,%xmm0
-  .byte  68,15,40,5,9,30,0,0                 // movaps        0x1e09(%rip),%xmm8        # 5870 <_sk_callback_sse2+0xdef>
+  .byte  68,15,40,5,255,29,0,0               // movaps        0x1dff(%rip),%xmm8        # 5890 <_sk_callback_sse2+0xde5>
   .byte  65,15,89,192                        // mulps         %xmm8,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -30933,7 +30988,7 @@ _sk_load_rgb_u16_be_sse2:
   .byte  102,69,15,239,192                   // pxor          %xmm8,%xmm8
   .byte  102,65,15,97,192                    // punpcklwd     %xmm8,%xmm0
   .byte  15,91,192                           // cvtdq2ps      %xmm0,%xmm0
-  .byte  68,15,40,13,69,29,0,0               // movaps        0x1d45(%rip),%xmm9        # 5880 <_sk_callback_sse2+0xdff>
+  .byte  68,15,40,13,59,29,0,0               // movaps        0x1d3b(%rip),%xmm9        # 58a0 <_sk_callback_sse2+0xdf5>
   .byte  65,15,89,193                        // mulps         %xmm9,%xmm0
   .byte  102,15,111,203                      // movdqa        %xmm3,%xmm1
   .byte  102,15,113,241,8                    // psllw         $0x8,%xmm1
@@ -30950,7 +31005,7 @@ _sk_load_rgb_u16_be_sse2:
   .byte  15,91,210                           // cvtdq2ps      %xmm2,%xmm2
   .byte  65,15,89,209                        // mulps         %xmm9,%xmm2
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  15,40,29,12,29,0,0                  // movaps        0x1d0c(%rip),%xmm3        # 5890 <_sk_callback_sse2+0xe0f>
+  .byte  15,40,29,2,29,0,0                   // movaps        0x1d02(%rip),%xmm3        # 58b0 <_sk_callback_sse2+0xe05>
   .byte  255,224                             // jmpq          *%rax
 
 HIDDEN _sk_store_u16_be_sse2
@@ -30959,7 +31014,7 @@ FUNCTION(_sk_store_u16_be_sse2)
 _sk_store_u16_be_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  72,139,0                            // mov           (%rax),%rax
-  .byte  68,15,40,13,13,29,0,0               // movaps        0x1d0d(%rip),%xmm9        # 58a0 <_sk_callback_sse2+0xe1f>
+  .byte  68,15,40,13,3,29,0,0                // movaps        0x1d03(%rip),%xmm9        # 58c0 <_sk_callback_sse2+0xe15>
   .byte  68,15,40,192                        // movaps        %xmm0,%xmm8
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  102,69,15,91,192                    // cvtps2dq      %xmm8,%xmm8
@@ -31105,7 +31160,7 @@ _sk_repeat_x_sse2:
   .byte  243,69,15,91,209                    // cvttps2dq     %xmm9,%xmm10
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
   .byte  69,15,194,202,1                     // cmpltps       %xmm10,%xmm9
-  .byte  68,15,84,13,13,27,0,0               // andps         0x1b0d(%rip),%xmm9        # 58b0 <_sk_callback_sse2+0xe2f>
+  .byte  68,15,84,13,3,27,0,0                // andps         0x1b03(%rip),%xmm9        # 58d0 <_sk_callback_sse2+0xe25>
   .byte  69,15,92,209                        // subps         %xmm9,%xmm10
   .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
   .byte  65,15,92,194                        // subps         %xmm10,%xmm0
@@ -31125,7 +31180,7 @@ _sk_repeat_y_sse2:
   .byte  243,69,15,91,209                    // cvttps2dq     %xmm9,%xmm10
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
   .byte  69,15,194,202,1                     // cmpltps       %xmm10,%xmm9
-  .byte  68,15,84,13,223,26,0,0              // andps         0x1adf(%rip),%xmm9        # 58c0 <_sk_callback_sse2+0xe3f>
+  .byte  68,15,84,13,213,26,0,0              // andps         0x1ad5(%rip),%xmm9        # 58e0 <_sk_callback_sse2+0xe35>
   .byte  69,15,92,209                        // subps         %xmm9,%xmm10
   .byte  69,15,89,208                        // mulps         %xmm8,%xmm10
   .byte  65,15,92,202                        // subps         %xmm10,%xmm1
@@ -31149,7 +31204,7 @@ _sk_mirror_x_sse2:
   .byte  243,69,15,91,218                    // cvttps2dq     %xmm10,%xmm11
   .byte  69,15,91,219                        // cvtdq2ps      %xmm11,%xmm11
   .byte  69,15,194,211,1                     // cmpltps       %xmm11,%xmm10
-  .byte  68,15,84,21,159,26,0,0              // andps         0x1a9f(%rip),%xmm10        # 58d0 <_sk_callback_sse2+0xe4f>
+  .byte  68,15,84,21,149,26,0,0              // andps         0x1a95(%rip),%xmm10        # 58f0 <_sk_callback_sse2+0xe45>
   .byte  69,15,87,228                        // xorps         %xmm12,%xmm12
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  69,15,89,216                        // mulps         %xmm8,%xmm11
@@ -31177,7 +31232,7 @@ _sk_mirror_y_sse2:
   .byte  243,69,15,91,218                    // cvttps2dq     %xmm10,%xmm11
   .byte  69,15,91,219                        // cvtdq2ps      %xmm11,%xmm11
   .byte  69,15,194,211,1                     // cmpltps       %xmm11,%xmm10
-  .byte  68,15,84,21,79,26,0,0               // andps         0x1a4f(%rip),%xmm10        # 58e0 <_sk_callback_sse2+0xe5f>
+  .byte  68,15,84,21,69,26,0,0               // andps         0x1a45(%rip),%xmm10        # 5900 <_sk_callback_sse2+0xe55>
   .byte  69,15,87,228                        // xorps         %xmm12,%xmm12
   .byte  69,15,92,218                        // subps         %xmm10,%xmm11
   .byte  69,15,89,216                        // mulps         %xmm8,%xmm11
@@ -31194,10 +31249,10 @@ HIDDEN _sk_luminance_to_alpha_sse2
 FUNCTION(_sk_luminance_to_alpha_sse2)
 _sk_luminance_to_alpha_sse2:
   .byte  15,40,218                           // movaps        %xmm2,%xmm3
-  .byte  15,89,5,49,26,0,0                   // mulps         0x1a31(%rip),%xmm0        # 58f0 <_sk_callback_sse2+0xe6f>
-  .byte  15,89,13,58,26,0,0                  // mulps         0x1a3a(%rip),%xmm1        # 5900 <_sk_callback_sse2+0xe7f>
+  .byte  15,89,5,39,26,0,0                   // mulps         0x1a27(%rip),%xmm0        # 5910 <_sk_callback_sse2+0xe65>
+  .byte  15,89,13,48,26,0,0                  // mulps         0x1a30(%rip),%xmm1        # 5920 <_sk_callback_sse2+0xe75>
   .byte  15,88,200                           // addps         %xmm0,%xmm1
-  .byte  15,89,29,64,26,0,0                  // mulps         0x1a40(%rip),%xmm3        # 5910 <_sk_callback_sse2+0xe8f>
+  .byte  15,89,29,54,26,0,0                  // mulps         0x1a36(%rip),%xmm3        # 5930 <_sk_callback_sse2+0xe85>
   .byte  15,88,217                           // addps         %xmm1,%xmm3
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,87,192                           // xorps         %xmm0,%xmm0
@@ -31423,9 +31478,9 @@ _sk_evenly_spaced_gradient_sse2:
   .byte  72,139,8                            // mov           (%rax),%rcx
   .byte  76,139,88,8                         // mov           0x8(%rax),%r11
   .byte  72,255,201                          // dec           %rcx
-  .byte  120,7                               // js            424f <_sk_evenly_spaced_gradient_sse2+0x15>
+  .byte  120,7                               // js            4279 <_sk_evenly_spaced_gradient_sse2+0x15>
   .byte  243,72,15,42,201                    // cvtsi2ss      %rcx,%xmm1
-  .byte  235,21                              // jmp           4264 <_sk_evenly_spaced_gradient_sse2+0x2a>
+  .byte  235,21                              // jmp           428e <_sk_evenly_spaced_gradient_sse2+0x2a>
   .byte  73,137,200                          // mov           %rcx,%r8
   .byte  73,209,232                          // shr           %r8
   .byte  131,225,1                           // and           $0x1,%ecx
@@ -31525,12 +31580,12 @@ _sk_gradient_sse2:
   .byte  76,139,0                            // mov           (%rax),%r8
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
   .byte  73,131,248,2                        // cmp           $0x2,%r8
-  .byte  114,50                              // jb            4427 <_sk_gradient_sse2+0x41>
+  .byte  114,50                              // jb            4451 <_sk_gradient_sse2+0x41>
   .byte  72,139,72,72                        // mov           0x48(%rax),%rcx
   .byte  73,255,200                          // dec           %r8
   .byte  72,131,193,4                        // add           $0x4,%rcx
   .byte  102,15,239,201                      // pxor          %xmm1,%xmm1
-  .byte  15,40,21,21,21,0,0                  // movaps        0x1515(%rip),%xmm2        # 5920 <_sk_callback_sse2+0xe9f>
+  .byte  15,40,21,11,21,0,0                  // movaps        0x150b(%rip),%xmm2        # 5940 <_sk_callback_sse2+0xe95>
   .byte  243,15,16,25                        // movss         (%rcx),%xmm3
   .byte  15,198,219,0                        // shufps        $0x0,%xmm3,%xmm3
   .byte  15,194,216,2                        // cmpleps       %xmm0,%xmm3
@@ -31538,7 +31593,7 @@ _sk_gradient_sse2:
   .byte  102,15,254,203                      // paddd         %xmm3,%xmm1
   .byte  72,131,193,4                        // add           $0x4,%rcx
   .byte  73,255,200                          // dec           %r8
-  .byte  117,228                             // jne           440b <_sk_gradient_sse2+0x25>
+  .byte  117,228                             // jne           4435 <_sk_gradient_sse2+0x25>
   .byte  65,86                               // push          %r14
   .byte  83                                  // push          %rbx
   .byte  102,15,112,209,78                   // pshufd        $0x4e,%xmm1,%xmm2
@@ -31678,29 +31733,29 @@ _sk_xy_to_unit_angle_sse2:
   .byte  69,15,94,220                        // divps         %xmm12,%xmm11
   .byte  69,15,40,227                        // movaps        %xmm11,%xmm12
   .byte  69,15,89,228                        // mulps         %xmm12,%xmm12
-  .byte  68,15,40,45,215,18,0,0              // movaps        0x12d7(%rip),%xmm13        # 5930 <_sk_callback_sse2+0xeaf>
+  .byte  68,15,40,45,205,18,0,0              // movaps        0x12cd(%rip),%xmm13        # 5950 <_sk_callback_sse2+0xea5>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
-  .byte  68,15,88,45,219,18,0,0              // addps         0x12db(%rip),%xmm13        # 5940 <_sk_callback_sse2+0xebf>
+  .byte  68,15,88,45,209,18,0,0              // addps         0x12d1(%rip),%xmm13        # 5960 <_sk_callback_sse2+0xeb5>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
-  .byte  68,15,88,45,223,18,0,0              // addps         0x12df(%rip),%xmm13        # 5950 <_sk_callback_sse2+0xecf>
+  .byte  68,15,88,45,213,18,0,0              // addps         0x12d5(%rip),%xmm13        # 5970 <_sk_callback_sse2+0xec5>
   .byte  69,15,89,236                        // mulps         %xmm12,%xmm13
-  .byte  68,15,88,45,227,18,0,0              // addps         0x12e3(%rip),%xmm13        # 5960 <_sk_callback_sse2+0xedf>
+  .byte  68,15,88,45,217,18,0,0              // addps         0x12d9(%rip),%xmm13        # 5980 <_sk_callback_sse2+0xed5>
   .byte  69,15,89,235                        // mulps         %xmm11,%xmm13
   .byte  69,15,194,202,1                     // cmpltps       %xmm10,%xmm9
-  .byte  68,15,40,21,226,18,0,0              // movaps        0x12e2(%rip),%xmm10        # 5970 <_sk_callback_sse2+0xeef>
+  .byte  68,15,40,21,216,18,0,0              // movaps        0x12d8(%rip),%xmm10        # 5990 <_sk_callback_sse2+0xee5>
   .byte  69,15,92,213                        // subps         %xmm13,%xmm10
   .byte  69,15,84,209                        // andps         %xmm9,%xmm10
   .byte  69,15,85,205                        // andnps        %xmm13,%xmm9
   .byte  69,15,86,202                        // orps          %xmm10,%xmm9
   .byte  68,15,194,192,1                     // cmpltps       %xmm0,%xmm8
-  .byte  68,15,40,21,213,18,0,0              // movaps        0x12d5(%rip),%xmm10        # 5980 <_sk_callback_sse2+0xeff>
+  .byte  68,15,40,21,203,18,0,0              // movaps        0x12cb(%rip),%xmm10        # 59a0 <_sk_callback_sse2+0xef5>
   .byte  69,15,92,209                        // subps         %xmm9,%xmm10
   .byte  69,15,84,208                        // andps         %xmm8,%xmm10
   .byte  69,15,85,193                        // andnps        %xmm9,%xmm8
   .byte  69,15,86,194                        // orps          %xmm10,%xmm8
   .byte  68,15,40,201                        // movaps        %xmm1,%xmm9
   .byte  68,15,194,200,1                     // cmpltps       %xmm0,%xmm9
-  .byte  68,15,40,21,196,18,0,0              // movaps        0x12c4(%rip),%xmm10        # 5990 <_sk_callback_sse2+0xf0f>
+  .byte  68,15,40,21,186,18,0,0              // movaps        0x12ba(%rip),%xmm10        # 59b0 <_sk_callback_sse2+0xf05>
   .byte  69,15,92,208                        // subps         %xmm8,%xmm10
   .byte  69,15,84,209                        // andps         %xmm9,%xmm10
   .byte  69,15,85,200                        // andnps        %xmm8,%xmm9
@@ -31727,7 +31782,7 @@ HIDDEN _sk_save_xy_sse2
 FUNCTION(_sk_save_xy_sse2)
 _sk_save_xy_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,150,18,0,0               // movaps        0x1296(%rip),%xmm8        # 59a0 <_sk_callback_sse2+0xf1f>
+  .byte  68,15,40,5,140,18,0,0               // movaps        0x128c(%rip),%xmm8        # 59c0 <_sk_callback_sse2+0xf15>
   .byte  15,17,0                             // movups        %xmm0,(%rax)
   .byte  68,15,40,200                        // movaps        %xmm0,%xmm9
   .byte  69,15,88,200                        // addps         %xmm8,%xmm9
@@ -31735,7 +31790,7 @@ _sk_save_xy_sse2:
   .byte  69,15,91,210                        // cvtdq2ps      %xmm10,%xmm10
   .byte  69,15,40,217                        // movaps        %xmm9,%xmm11
   .byte  69,15,194,218,1                     // cmpltps       %xmm10,%xmm11
-  .byte  68,15,40,37,129,18,0,0              // movaps        0x1281(%rip),%xmm12        # 59b0 <_sk_callback_sse2+0xf2f>
+  .byte  68,15,40,37,119,18,0,0              // movaps        0x1277(%rip),%xmm12        # 59d0 <_sk_callback_sse2+0xf25>
   .byte  69,15,84,220                        // andps         %xmm12,%xmm11
   .byte  69,15,92,211                        // subps         %xmm11,%xmm10
   .byte  69,15,92,202                        // subps         %xmm10,%xmm9
@@ -31782,8 +31837,8 @@ _sk_bilinear_nx_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,250,17,0,0                  // addps         0x11fa(%rip),%xmm0        # 59c0 <_sk_callback_sse2+0xf3f>
-  .byte  68,15,40,13,2,18,0,0                // movaps        0x1202(%rip),%xmm9        # 59d0 <_sk_callback_sse2+0xf4f>
+  .byte  15,88,5,240,17,0,0                  // addps         0x11f0(%rip),%xmm0        # 59e0 <_sk_callback_sse2+0xf35>
+  .byte  68,15,40,13,248,17,0,0              // movaps        0x11f8(%rip),%xmm9        # 59f0 <_sk_callback_sse2+0xf45>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -31796,7 +31851,7 @@ _sk_bilinear_px_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,241,17,0,0                  // addps         0x11f1(%rip),%xmm0        # 59e0 <_sk_callback_sse2+0xf5f>
+  .byte  15,88,5,231,17,0,0                  // addps         0x11e7(%rip),%xmm0        # 5a00 <_sk_callback_sse2+0xf55>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -31808,8 +31863,8 @@ _sk_bilinear_ny_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,227,17,0,0                 // addps         0x11e3(%rip),%xmm1        # 59f0 <_sk_callback_sse2+0xf6f>
-  .byte  68,15,40,13,235,17,0,0              // movaps        0x11eb(%rip),%xmm9        # 5a00 <_sk_callback_sse2+0xf7f>
+  .byte  15,88,13,217,17,0,0                 // addps         0x11d9(%rip),%xmm1        # 5a10 <_sk_callback_sse2+0xf65>
+  .byte  68,15,40,13,225,17,0,0              // movaps        0x11e1(%rip),%xmm9        # 5a20 <_sk_callback_sse2+0xf75>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -31822,7 +31877,7 @@ _sk_bilinear_py_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,217,17,0,0                 // addps         0x11d9(%rip),%xmm1        # 5a10 <_sk_callback_sse2+0xf8f>
+  .byte  15,88,13,207,17,0,0                 // addps         0x11cf(%rip),%xmm1        # 5a30 <_sk_callback_sse2+0xf85>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -31834,13 +31889,13 @@ _sk_bicubic_n3x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,204,17,0,0                  // addps         0x11cc(%rip),%xmm0        # 5a20 <_sk_callback_sse2+0xf9f>
-  .byte  68,15,40,13,212,17,0,0              // movaps        0x11d4(%rip),%xmm9        # 5a30 <_sk_callback_sse2+0xfaf>
+  .byte  15,88,5,194,17,0,0                  // addps         0x11c2(%rip),%xmm0        # 5a40 <_sk_callback_sse2+0xf95>
+  .byte  68,15,40,13,202,17,0,0              // movaps        0x11ca(%rip),%xmm9        # 5a50 <_sk_callback_sse2+0xfa5>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,208,17,0,0              // mulps         0x11d0(%rip),%xmm9        # 5a40 <_sk_callback_sse2+0xfbf>
-  .byte  68,15,88,13,216,17,0,0              // addps         0x11d8(%rip),%xmm9        # 5a50 <_sk_callback_sse2+0xfcf>
+  .byte  68,15,89,13,198,17,0,0              // mulps         0x11c6(%rip),%xmm9        # 5a60 <_sk_callback_sse2+0xfb5>
+  .byte  68,15,88,13,206,17,0,0              // addps         0x11ce(%rip),%xmm9        # 5a70 <_sk_callback_sse2+0xfc5>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,128,0,0,0              // movups        %xmm9,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -31853,16 +31908,16 @@ _sk_bicubic_n1x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,199,17,0,0                  // addps         0x11c7(%rip),%xmm0        # 5a60 <_sk_callback_sse2+0xfdf>
-  .byte  68,15,40,13,207,17,0,0              // movaps        0x11cf(%rip),%xmm9        # 5a70 <_sk_callback_sse2+0xfef>
+  .byte  15,88,5,189,17,0,0                  // addps         0x11bd(%rip),%xmm0        # 5a80 <_sk_callback_sse2+0xfd5>
+  .byte  68,15,40,13,197,17,0,0              // movaps        0x11c5(%rip),%xmm9        # 5a90 <_sk_callback_sse2+0xfe5>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,211,17,0,0               // movaps        0x11d3(%rip),%xmm8        # 5a80 <_sk_callback_sse2+0xfff>
+  .byte  68,15,40,5,201,17,0,0               // movaps        0x11c9(%rip),%xmm8        # 5aa0 <_sk_callback_sse2+0xff5>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,215,17,0,0               // addps         0x11d7(%rip),%xmm8        # 5a90 <_sk_callback_sse2+0x100f>
+  .byte  68,15,88,5,205,17,0,0               // addps         0x11cd(%rip),%xmm8        # 5ab0 <_sk_callback_sse2+0x1005>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,219,17,0,0               // addps         0x11db(%rip),%xmm8        # 5aa0 <_sk_callback_sse2+0x101f>
+  .byte  68,15,88,5,209,17,0,0               // addps         0x11d1(%rip),%xmm8        # 5ac0 <_sk_callback_sse2+0x1015>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,223,17,0,0               // addps         0x11df(%rip),%xmm8        # 5ab0 <_sk_callback_sse2+0x102f>
+  .byte  68,15,88,5,213,17,0,0               // addps         0x11d5(%rip),%xmm8        # 5ad0 <_sk_callback_sse2+0x1025>
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -31872,17 +31927,17 @@ HIDDEN _sk_bicubic_p1x_sse2
 FUNCTION(_sk_bicubic_p1x_sse2)
 _sk_bicubic_p1x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,217,17,0,0               // movaps        0x11d9(%rip),%xmm8        # 5ac0 <_sk_callback_sse2+0x103f>
+  .byte  68,15,40,5,207,17,0,0               // movaps        0x11cf(%rip),%xmm8        # 5ae0 <_sk_callback_sse2+0x1035>
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,72,64                      // movups        0x40(%rax),%xmm9
   .byte  65,15,88,192                        // addps         %xmm8,%xmm0
-  .byte  68,15,40,21,213,17,0,0              // movaps        0x11d5(%rip),%xmm10        # 5ad0 <_sk_callback_sse2+0x104f>
+  .byte  68,15,40,21,203,17,0,0              // movaps        0x11cb(%rip),%xmm10        # 5af0 <_sk_callback_sse2+0x1045>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,217,17,0,0              // addps         0x11d9(%rip),%xmm10        # 5ae0 <_sk_callback_sse2+0x105f>
+  .byte  68,15,88,21,207,17,0,0              // addps         0x11cf(%rip),%xmm10        # 5b00 <_sk_callback_sse2+0x1055>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,213,17,0,0              // addps         0x11d5(%rip),%xmm10        # 5af0 <_sk_callback_sse2+0x106f>
+  .byte  68,15,88,21,203,17,0,0              // addps         0x11cb(%rip),%xmm10        # 5b10 <_sk_callback_sse2+0x1065>
   .byte  68,15,17,144,128,0,0,0              // movups        %xmm10,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -31894,11 +31949,11 @@ _sk_bicubic_p3x_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,0                             // movups        (%rax),%xmm0
   .byte  68,15,16,64,64                      // movups        0x40(%rax),%xmm8
-  .byte  15,88,5,200,17,0,0                  // addps         0x11c8(%rip),%xmm0        # 5b00 <_sk_callback_sse2+0x107f>
+  .byte  15,88,5,190,17,0,0                  // addps         0x11be(%rip),%xmm0        # 5b20 <_sk_callback_sse2+0x1075>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,200,17,0,0               // mulps         0x11c8(%rip),%xmm8        # 5b10 <_sk_callback_sse2+0x108f>
-  .byte  68,15,88,5,208,17,0,0               // addps         0x11d0(%rip),%xmm8        # 5b20 <_sk_callback_sse2+0x109f>
+  .byte  68,15,89,5,190,17,0,0               // mulps         0x11be(%rip),%xmm8        # 5b30 <_sk_callback_sse2+0x1085>
+  .byte  68,15,88,5,198,17,0,0               // addps         0x11c6(%rip),%xmm8        # 5b40 <_sk_callback_sse2+0x1095>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,128,0,0,0              // movups        %xmm8,0x80(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -31911,13 +31966,13 @@ _sk_bicubic_n3y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,190,17,0,0                 // addps         0x11be(%rip),%xmm1        # 5b30 <_sk_callback_sse2+0x10af>
-  .byte  68,15,40,13,198,17,0,0              // movaps        0x11c6(%rip),%xmm9        # 5b40 <_sk_callback_sse2+0x10bf>
+  .byte  15,88,13,180,17,0,0                 // addps         0x11b4(%rip),%xmm1        # 5b50 <_sk_callback_sse2+0x10a5>
+  .byte  68,15,40,13,188,17,0,0              // movaps        0x11bc(%rip),%xmm9        # 5b60 <_sk_callback_sse2+0x10b5>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
   .byte  69,15,40,193                        // movaps        %xmm9,%xmm8
   .byte  69,15,89,192                        // mulps         %xmm8,%xmm8
-  .byte  68,15,89,13,194,17,0,0              // mulps         0x11c2(%rip),%xmm9        # 5b50 <_sk_callback_sse2+0x10cf>
-  .byte  68,15,88,13,202,17,0,0              // addps         0x11ca(%rip),%xmm9        # 5b60 <_sk_callback_sse2+0x10df>
+  .byte  68,15,89,13,184,17,0,0              // mulps         0x11b8(%rip),%xmm9        # 5b70 <_sk_callback_sse2+0x10c5>
+  .byte  68,15,88,13,192,17,0,0              // addps         0x11c0(%rip),%xmm9        # 5b80 <_sk_callback_sse2+0x10d5>
   .byte  69,15,89,200                        // mulps         %xmm8,%xmm9
   .byte  68,15,17,136,160,0,0,0              // movups        %xmm9,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -31930,16 +31985,16 @@ _sk_bicubic_n1y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,184,17,0,0                 // addps         0x11b8(%rip),%xmm1        # 5b70 <_sk_callback_sse2+0x10ef>
-  .byte  68,15,40,13,192,17,0,0              // movaps        0x11c0(%rip),%xmm9        # 5b80 <_sk_callback_sse2+0x10ff>
+  .byte  15,88,13,174,17,0,0                 // addps         0x11ae(%rip),%xmm1        # 5b90 <_sk_callback_sse2+0x10e5>
+  .byte  68,15,40,13,182,17,0,0              // movaps        0x11b6(%rip),%xmm9        # 5ba0 <_sk_callback_sse2+0x10f5>
   .byte  69,15,92,200                        // subps         %xmm8,%xmm9
-  .byte  68,15,40,5,196,17,0,0               // movaps        0x11c4(%rip),%xmm8        # 5b90 <_sk_callback_sse2+0x110f>
+  .byte  68,15,40,5,186,17,0,0               // movaps        0x11ba(%rip),%xmm8        # 5bb0 <_sk_callback_sse2+0x1105>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,200,17,0,0               // addps         0x11c8(%rip),%xmm8        # 5ba0 <_sk_callback_sse2+0x111f>
+  .byte  68,15,88,5,190,17,0,0               // addps         0x11be(%rip),%xmm8        # 5bc0 <_sk_callback_sse2+0x1115>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,204,17,0,0               // addps         0x11cc(%rip),%xmm8        # 5bb0 <_sk_callback_sse2+0x112f>
+  .byte  68,15,88,5,194,17,0,0               // addps         0x11c2(%rip),%xmm8        # 5bd0 <_sk_callback_sse2+0x1125>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
-  .byte  68,15,88,5,208,17,0,0               // addps         0x11d0(%rip),%xmm8        # 5bc0 <_sk_callback_sse2+0x113f>
+  .byte  68,15,88,5,198,17,0,0               // addps         0x11c6(%rip),%xmm8        # 5be0 <_sk_callback_sse2+0x1135>
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -31949,17 +32004,17 @@ HIDDEN _sk_bicubic_p1y_sse2
 FUNCTION(_sk_bicubic_p1y_sse2)
 _sk_bicubic_p1y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
-  .byte  68,15,40,5,202,17,0,0               // movaps        0x11ca(%rip),%xmm8        # 5bd0 <_sk_callback_sse2+0x114f>
+  .byte  68,15,40,5,192,17,0,0               // movaps        0x11c0(%rip),%xmm8        # 5bf0 <_sk_callback_sse2+0x1145>
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,72,96                      // movups        0x60(%rax),%xmm9
   .byte  65,15,88,200                        // addps         %xmm8,%xmm1
-  .byte  68,15,40,21,197,17,0,0              // movaps        0x11c5(%rip),%xmm10        # 5be0 <_sk_callback_sse2+0x115f>
+  .byte  68,15,40,21,187,17,0,0              // movaps        0x11bb(%rip),%xmm10        # 5c00 <_sk_callback_sse2+0x1155>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,201,17,0,0              // addps         0x11c9(%rip),%xmm10        # 5bf0 <_sk_callback_sse2+0x116f>
+  .byte  68,15,88,21,191,17,0,0              // addps         0x11bf(%rip),%xmm10        # 5c10 <_sk_callback_sse2+0x1165>
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
   .byte  69,15,88,208                        // addps         %xmm8,%xmm10
   .byte  69,15,89,209                        // mulps         %xmm9,%xmm10
-  .byte  68,15,88,21,197,17,0,0              // addps         0x11c5(%rip),%xmm10        # 5c00 <_sk_callback_sse2+0x117f>
+  .byte  68,15,88,21,187,17,0,0              // addps         0x11bb(%rip),%xmm10        # 5c20 <_sk_callback_sse2+0x1175>
   .byte  68,15,17,144,160,0,0,0              // movups        %xmm10,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  255,224                             // jmpq          *%rax
@@ -31971,11 +32026,11 @@ _sk_bicubic_p3y_sse2:
   .byte  72,173                              // lods          %ds:(%rsi),%rax
   .byte  15,16,72,32                         // movups        0x20(%rax),%xmm1
   .byte  68,15,16,64,96                      // movups        0x60(%rax),%xmm8
-  .byte  15,88,13,183,17,0,0                 // addps         0x11b7(%rip),%xmm1        # 5c10 <_sk_callback_sse2+0x118f>
+  .byte  15,88,13,173,17,0,0                 // addps         0x11ad(%rip),%xmm1        # 5c30 <_sk_callback_sse2+0x1185>
   .byte  69,15,40,200                        // movaps        %xmm8,%xmm9
   .byte  69,15,89,201                        // mulps         %xmm9,%xmm9
-  .byte  68,15,89,5,183,17,0,0               // mulps         0x11b7(%rip),%xmm8        # 5c20 <_sk_callback_sse2+0x119f>
-  .byte  68,15,88,5,191,17,0,0               // addps         0x11bf(%rip),%xmm8        # 5c30 <_sk_callback_sse2+0x11af>
+  .byte  68,15,89,5,173,17,0,0               // mulps         0x11ad(%rip),%xmm8        # 5c40 <_sk_callback_sse2+0x1195>
+  .byte  68,15,88,5,181,17,0,0               // addps         0x11b5(%rip),%xmm8        # 5c50 <_sk_callback_sse2+0x11a5>
   .byte  69,15,89,193                        // mulps         %xmm9,%xmm8
   .byte  68,15,17,128,160,0,0,0              // movups        %xmm8,0xa0(%rax)
   .byte  72,173                              // lods          %ds:(%rsi),%rax
@@ -32194,11 +32249,11 @@ BALIGN16
   .byte  128,191,0,0,128,191,0               // cmpb          $0x0,-0x40800000(%rdi)
   .byte  0,224                               // add           %ah,%al
   .byte  64,0,0                              // add           %al,(%rax)
-  .byte  224,64                              // loopne        4d38 <.literal16+0x1d8>
+  .byte  224,64                              // loopne        4d58 <.literal16+0x1d8>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        4d3c <.literal16+0x1dc>
+  .byte  224,64                              // loopne        4d5c <.literal16+0x1dc>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,64                              // loopne        4d40 <.literal16+0x1e0>
+  .byte  224,64                              // loopne        4d60 <.literal16+0x1e0>
   .byte  154                                 // (bad)
   .byte  153                                 // cltd
   .byte  153                                 // cltd
@@ -32218,13 +32273,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4d61 <.literal16+0x201>
+  .byte  71,225,61                           // rex.RXB       loope 4d81 <.literal16+0x201>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4d65 <.literal16+0x205>
+  .byte  71,225,61                           // rex.RXB       loope 4d85 <.literal16+0x205>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4d69 <.literal16+0x209>
+  .byte  71,225,61                           // rex.RXB       loope 4d89 <.literal16+0x209>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4d6d <.literal16+0x20d>
+  .byte  71,225,61                           // rex.RXB       loope 4d8d <.literal16+0x20d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -32249,13 +32304,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4da1 <.literal16+0x241>
+  .byte  71,225,61                           // rex.RXB       loope 4dc1 <.literal16+0x241>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4da5 <.literal16+0x245>
+  .byte  71,225,61                           // rex.RXB       loope 4dc5 <.literal16+0x245>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4da9 <.literal16+0x249>
+  .byte  71,225,61                           // rex.RXB       loope 4dc9 <.literal16+0x249>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4dad <.literal16+0x24d>
+  .byte  71,225,61                           // rex.RXB       loope 4dcd <.literal16+0x24d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -32280,13 +32335,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4de1 <.literal16+0x281>
+  .byte  71,225,61                           // rex.RXB       loope 4e01 <.literal16+0x281>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4de5 <.literal16+0x285>
+  .byte  71,225,61                           // rex.RXB       loope 4e05 <.literal16+0x285>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4de9 <.literal16+0x289>
+  .byte  71,225,61                           // rex.RXB       loope 4e09 <.literal16+0x289>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4ded <.literal16+0x28d>
+  .byte  71,225,61                           // rex.RXB       loope 4e0d <.literal16+0x28d>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -32311,13 +32366,13 @@ BALIGN16
   .byte  10,23                               // or            (%rdi),%dl
   .byte  63                                  // (bad)
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4e21 <.literal16+0x2c1>
+  .byte  71,225,61                           // rex.RXB       loope 4e41 <.literal16+0x2c1>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4e25 <.literal16+0x2c5>
+  .byte  71,225,61                           // rex.RXB       loope 4e45 <.literal16+0x2c5>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4e29 <.literal16+0x2c9>
+  .byte  71,225,61                           // rex.RXB       loope 4e49 <.literal16+0x2c9>
   .byte  174                                 // scas          %es:(%rdi),%al
-  .byte  71,225,61                           // rex.RXB       loope 4e2d <.literal16+0x2cd>
+  .byte  71,225,61                           // rex.RXB       loope 4e4d <.literal16+0x2cd>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -32546,13 +32601,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5009 <.literal16+0x4a9>
+  .byte  224,7                               // loopne        5029 <.literal16+0x4a9>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        500d <.literal16+0x4ad>
+  .byte  224,7                               // loopne        502d <.literal16+0x4ad>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5011 <.literal16+0x4b1>
+  .byte  224,7                               // loopne        5031 <.literal16+0x4b1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5015 <.literal16+0x4b5>
+  .byte  224,7                               // loopne        5035 <.literal16+0x4b5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -32617,11 +32672,11 @@ BALIGN16
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            50eb <.literal16+0x58b>
+  .byte  127,67                              // jg            510b <.literal16+0x58b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            50ef <.literal16+0x58f>
+  .byte  127,67                              // jg            510f <.literal16+0x58f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            50f3 <.literal16+0x593>
+  .byte  127,67                              // jg            5113 <.literal16+0x593>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,129,128,128,59           // addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -32636,16 +32691,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            50e4 <.literal16+0x584>
+  .byte  127,0                               // jg            5104 <.literal16+0x584>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            50e8 <.literal16+0x588>
+  .byte  127,0                               // jg            5108 <.literal16+0x588>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            50ec <.literal16+0x58c>
+  .byte  127,0                               // jg            510c <.literal16+0x58c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            50f0 <.literal16+0x590>
+  .byte  127,0                               // jg            5110 <.literal16+0x590>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -32654,7 +32709,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5175 <.literal16+0x615>
+  .byte  119,115                             // ja            5195 <.literal16+0x615>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -32665,7 +32720,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           50d9 <.literal16+0x579>
+  .byte  117,191                             // jne           50f9 <.literal16+0x579>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -32677,7 +32732,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3911a <_sk_callback_sse2+0xffffffffe9a34699>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3913a <_sk_callback_sse2+0xffffffffe9a3468f>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -32731,16 +32786,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            51b4 <.literal16+0x654>
+  .byte  127,0                               // jg            51d4 <.literal16+0x654>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            51b8 <.literal16+0x658>
+  .byte  127,0                               // jg            51d8 <.literal16+0x658>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            51bc <.literal16+0x65c>
+  .byte  127,0                               // jg            51dc <.literal16+0x65c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            51c0 <.literal16+0x660>
+  .byte  127,0                               // jg            51e0 <.literal16+0x660>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -32749,7 +32804,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5245 <.literal16+0x6e5>
+  .byte  119,115                             // ja            5265 <.literal16+0x6e5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -32760,7 +32815,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           51a9 <.literal16+0x649>
+  .byte  117,191                             // jne           51c9 <.literal16+0x649>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -32772,7 +32827,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a391ea <_sk_callback_sse2+0xffffffffe9a34769>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3920a <_sk_callback_sse2+0xffffffffe9a3475f>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -32826,16 +32881,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5284 <.literal16+0x724>
+  .byte  127,0                               // jg            52a4 <.literal16+0x724>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5288 <.literal16+0x728>
+  .byte  127,0                               // jg            52a8 <.literal16+0x728>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            528c <.literal16+0x72c>
+  .byte  127,0                               // jg            52ac <.literal16+0x72c>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5290 <.literal16+0x730>
+  .byte  127,0                               // jg            52b0 <.literal16+0x730>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -32844,7 +32899,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            5315 <.literal16+0x7b5>
+  .byte  119,115                             // ja            5335 <.literal16+0x7b5>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -32855,7 +32910,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           5279 <.literal16+0x719>
+  .byte  117,191                             // jne           5299 <.literal16+0x719>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -32867,7 +32922,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a392ba <_sk_callback_sse2+0xffffffffe9a34839>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a392da <_sk_callback_sse2+0xffffffffe9a3482f>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -32921,16 +32976,16 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  52,255                              // xor           $0xff,%al
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5354 <.literal16+0x7f4>
+  .byte  127,0                               // jg            5374 <.literal16+0x7f4>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5358 <.literal16+0x7f8>
+  .byte  127,0                               // jg            5378 <.literal16+0x7f8>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            535c <.literal16+0x7fc>
+  .byte  127,0                               // jg            537c <.literal16+0x7fc>
   .byte  255                                 // (bad)
   .byte  255                                 // (bad)
-  .byte  127,0                               // jg            5360 <.literal16+0x800>
+  .byte  127,0                               // jg            5380 <.literal16+0x800>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -32939,7 +32994,7 @@ BALIGN16
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
-  .byte  119,115                             // ja            53e5 <.literal16+0x885>
+  .byte  119,115                             // ja            5405 <.literal16+0x885>
   .byte  248                                 // clc
   .byte  194,119,115                         // retq          $0x7377
   .byte  248                                 // clc
@@ -32950,7 +33005,7 @@ BALIGN16
   .byte  194,117,191                         // retq          $0xbf75
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
-  .byte  117,191                             // jne           5349 <.literal16+0x7e9>
+  .byte  117,191                             // jne           5369 <.literal16+0x7e9>
   .byte  191,63,117,191,191                  // mov           $0xbfbf753f,%edi
   .byte  63                                  // (bad)
   .byte  249                                 // stc
@@ -32962,7 +33017,7 @@ BALIGN16
   .byte  249                                 // stc
   .byte  68,180,62                           // rex.R         mov $0x3e,%spl
   .byte  163,233,220,63,163,233,220,63,163   // movabs        %eax,0xa33fdce9a33fdce9
-  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a3938a <_sk_callback_sse2+0xffffffffe9a34909>
+  .byte  233,220,63,163,233                  // jmpq          ffffffffe9a393aa <_sk_callback_sse2+0xffffffffe9a348ff>
   .byte  220,63                              // fdivrl        (%rdi)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
@@ -33012,13 +33067,13 @@ BALIGN16
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
   .byte  200,66,0,0                          // enterq        $0x42,$0x0
-  .byte  127,67                              // jg            5467 <.literal16+0x907>
+  .byte  127,67                              // jg            5487 <.literal16+0x907>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            546b <.literal16+0x90b>
+  .byte  127,67                              // jg            548b <.literal16+0x90b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            546f <.literal16+0x90f>
+  .byte  127,67                              // jg            548f <.literal16+0x90f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5473 <.literal16+0x913>
+  .byte  127,67                              // jg            5493 <.literal16+0x913>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,195                               // add           %al,%bl
   .byte  0,0                                 // add           %al,(%rax)
@@ -33065,16 +33120,16 @@ BALIGN16
   .byte  128,3,62                            // addb          $0x3e,(%rbx)
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           54f3 <.literal16+0x993>
+  .byte  118,63                              // jbe           5513 <.literal16+0x993>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           54f7 <.literal16+0x997>
+  .byte  118,63                              // jbe           5517 <.literal16+0x997>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           54fb <.literal16+0x99b>
+  .byte  118,63                              // jbe           551b <.literal16+0x99b>
   .byte  31                                  // (bad)
   .byte  215                                 // xlat          %ds:(%rbx)
-  .byte  118,63                              // jbe           54ff <.literal16+0x99f>
+  .byte  118,63                              // jbe           551f <.literal16+0x99f>
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
   .byte  246,64,83,63                        // testb         $0x3f,0x53(%rax)
@@ -33086,11 +33141,11 @@ BALIGN16
   .byte  128,59,0                            // cmpb          $0x0,(%rbx)
   .byte  0,127,67                            // add           %bh,0x43(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            553b <.literal16+0x9db>
+  .byte  127,67                              // jg            555b <.literal16+0x9db>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            553f <.literal16+0x9df>
+  .byte  127,67                              // jg            555f <.literal16+0x9df>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5543 <.literal16+0x9e3>
+  .byte  127,67                              // jg            5563 <.literal16+0x9e3>
   .byte  129,128,128,59,129,128,128,59,129,128// addl          $0x80813b80,-0x7f7ec480(%rax)
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,0,0,128,63               // addb          $0x3f,-0x7fffffc5(%rax)
@@ -33130,13 +33185,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        5589 <.literal16+0xa29>
+  .byte  224,7                               // loopne        55a9 <.literal16+0xa29>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        558d <.literal16+0xa2d>
+  .byte  224,7                               // loopne        55ad <.literal16+0xa2d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5591 <.literal16+0xa31>
+  .byte  224,7                               // loopne        55b1 <.literal16+0xa31>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5595 <.literal16+0xa35>
+  .byte  224,7                               // loopne        55b5 <.literal16+0xa35>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -33182,13 +33237,13 @@ BALIGN16
   .byte  132,55                              // test          %dh,(%rdi)
   .byte  8,33                                // or            %ah,(%rcx)
   .byte  132,55                              // test          %dh,(%rdi)
-  .byte  224,7                               // loopne        55f9 <.literal16+0xa99>
+  .byte  224,7                               // loopne        5619 <.literal16+0xa99>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        55fd <.literal16+0xa9d>
+  .byte  224,7                               // loopne        561d <.literal16+0xa9d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5601 <.literal16+0xaa1>
+  .byte  224,7                               // loopne        5621 <.literal16+0xaa1>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  224,7                               // loopne        5605 <.literal16+0xaa5>
+  .byte  224,7                               // loopne        5625 <.literal16+0xaa5>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  33,8                                // and           %ecx,(%rax)
   .byte  2,58                                // add           (%rdx),%bh
@@ -33226,13 +33281,13 @@ BALIGN16
   .byte  65,0,0                              // add           %al,(%r8)
   .byte  248                                 // clc
   .byte  65,0,0                              // add           %al,(%r8)
-  .byte  124,66                              // jl            5696 <.literal16+0xb36>
+  .byte  124,66                              // jl            56b6 <.literal16+0xb36>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            569a <.literal16+0xb3a>
+  .byte  124,66                              // jl            56ba <.literal16+0xb3a>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            569e <.literal16+0xb3e>
+  .byte  124,66                              // jl            56be <.literal16+0xb3e>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  124,66                              // jl            56a2 <.literal16+0xb42>
+  .byte  124,66                              // jl            56c2 <.literal16+0xb42>
   .byte  0,240                               // add           %dh,%al
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,240                               // add           %dh,%al
@@ -33322,13 +33377,13 @@ BALIGN16
   .byte  136,136,61,137,136,136              // mov           %cl,-0x777776c3(%rax)
   .byte  61,137,136,136,61                   // cmp           $0x3d888889,%eax
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            57a5 <.literal16+0xc45>
+  .byte  112,65                              // jo            57c5 <.literal16+0xc45>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            57a9 <.literal16+0xc49>
+  .byte  112,65                              // jo            57c9 <.literal16+0xc49>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            57ad <.literal16+0xc4d>
+  .byte  112,65                              // jo            57cd <.literal16+0xc4d>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  112,65                              // jo            57b1 <.literal16+0xc51>
+  .byte  112,65                              // jo            57d1 <.literal16+0xc51>
   .byte  255,0                               // incl          (%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  255,0                               // incl          (%rax)
@@ -33350,11 +33405,11 @@ BALIGN16
   .byte  128,59,129                          // cmpb          $0x81,(%rbx)
   .byte  128,128,59,0,0,127,67               // addb          $0x43,0x7f00003b(%rax)
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            57fb <.literal16+0xc9b>
+  .byte  127,67                              // jg            581b <.literal16+0xc9b>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            57ff <.literal16+0xc9f>
+  .byte  127,67                              // jg            581f <.literal16+0xc9f>
   .byte  0,0                                 // add           %al,(%rax)
-  .byte  127,67                              // jg            5803 <.literal16+0xca3>
+  .byte  127,67                              // jg            5823 <.literal16+0xca3>
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,128,0,0,0,128                     // add           %al,-0x80000000(%rax)
@@ -33430,13 +33485,13 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  255                                 // (bad)
-  .byte  127,71                              // jg            58eb <.literal16+0xd8b>
+  .byte  127,71                              // jg            590b <.literal16+0xd8b>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            58ef <.literal16+0xd8f>
+  .byte  127,71                              // jg            590f <.literal16+0xd8f>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            58f3 <.literal16+0xd93>
+  .byte  127,71                              // jg            5913 <.literal16+0xd93>
   .byte  0,255                               // add           %bh,%bh
-  .byte  127,71                              // jg            58f7 <.literal16+0xd97>
+  .byte  127,71                              // jg            5917 <.literal16+0xd97>
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,0                            // cmpb          $0x0,(%rdi)
   .byte  0,128,63,0,0,128                    // add           %al,-0x7fffffc1(%rax)
@@ -33597,11 +33652,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5a62 <.literal16+0xf02>
+  .byte  62,114,28                           // jb,pt         5a82 <.literal16+0xf02>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5a66 <.literal16+0xf06>
+  .byte  62,114,28                           // jb,pt         5a86 <.literal16+0xf06>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5a6a <.literal16+0xf0a>
+  .byte  62,114,28                           // jb,pt         5a8a <.literal16+0xf0a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -33645,7 +33700,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e8f5 <_sk_callback_sse2+0x3d639e74>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e915 <_sk_callback_sse2+0x3d639e6a>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -33671,7 +33726,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e935 <_sk_callback_sse2+0x3d639eb4>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63e955 <_sk_callback_sse2+0x3d639eaa>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -33680,13 +33735,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            5b2e <.literal16+0xfce>
+  .byte  114,28                              // jb            5b4e <.literal16+0xfce>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5b32 <.literal16+0xfd2>
+  .byte  62,114,28                           // jb,pt         5b52 <.literal16+0xfd2>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5b36 <.literal16+0xfd6>
+  .byte  62,114,28                           // jb,pt         5b56 <.literal16+0xfd6>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5b3a <.literal16+0xfda>
+  .byte  62,114,28                           // jb,pt         5b5a <.literal16+0xfda>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -33707,11 +33762,11 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  128,63,114                          // cmpb          $0x72,(%rdi)
   .byte  28,199                              // sbb           $0xc7,%al
-  .byte  62,114,28                           // jb,pt         5b72 <.literal16+0x1012>
+  .byte  62,114,28                           // jb,pt         5b92 <.literal16+0x1012>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5b76 <.literal16+0x1016>
+  .byte  62,114,28                           // jb,pt         5b96 <.literal16+0x1016>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5b7a <.literal16+0x101a>
+  .byte  62,114,28                           // jb,pt         5b9a <.literal16+0x101a>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
@@ -33755,7 +33810,7 @@ BALIGN16
   .byte  0,0                                 // add           %al,(%rax)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63ea05 <_sk_callback_sse2+0x3d639f84>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63ea25 <_sk_callback_sse2+0x3d639f7a>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  0,63                                // add           %bh,(%rdi)
   .byte  0,0                                 // add           %al,(%rax)
@@ -33781,7 +33836,7 @@ BALIGN16
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
   .byte  57,142,99,61,57,142                 // cmp           %ecx,-0x71c6c29d(%rsi)
-  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63ea45 <_sk_callback_sse2+0x3d639fc4>
+  .byte  99,61,57,142,99,61                  // movslq        0x3d638e39(%rip),%edi        # 3d63ea65 <_sk_callback_sse2+0x3d639fba>
   .byte  57,142,99,61,0,0                    // cmp           %ecx,0x3d63(%rsi)
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
@@ -33790,13 +33845,13 @@ BALIGN16
   .byte  192,63,0                            // sarb          $0x0,(%rdi)
   .byte  0,192                               // add           %al,%al
   .byte  63                                  // (bad)
-  .byte  114,28                              // jb            5c3e <.literal16+0x10de>
+  .byte  114,28                              // jb            5c5e <.literal16+0x10de>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5c42 <_sk_callback_sse2+0x11c1>
+  .byte  62,114,28                           // jb,pt         5c62 <_sk_callback_sse2+0x11b7>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5c46 <_sk_callback_sse2+0x11c5>
+  .byte  62,114,28                           // jb,pt         5c66 <_sk_callback_sse2+0x11bb>
   .byte  199                                 // (bad)
-  .byte  62,114,28                           // jb,pt         5c4a <_sk_callback_sse2+0x11c9>
+  .byte  62,114,28                           // jb,pt         5c6a <_sk_callback_sse2+0x11bf>
   .byte  199                                 // (bad)
   .byte  62,171                              // ds            stos %eax,%es:(%rdi)
   .byte  170                                 // stos          %al,%es:(%rdi)
index f759020..d670b65 100644 (file)
@@ -106,14 +106,14 @@ _sk_seed_shader_hsw LABEL PROC
   DB  197,249,110,199                     ; vmovd         %edi,%xmm0
   DB  196,226,125,88,192                  ; vpbroadcastd  %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,214,70,0,0        ; vbroadcastss  0x46d6(%rip),%ymm1        # 4830 <_sk_callback_hsw+0x11c>
+  DB  196,226,125,24,13,242,70,0,0        ; vbroadcastss  0x46f2(%rip),%ymm1        # 484c <_sk_callback_hsw+0x11c>
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
   DB  197,252,88,2                        ; vaddps        (%rdx),%ymm0,%ymm0
   DB  196,226,125,24,16                   ; vbroadcastss  (%rax),%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  197,236,88,201                      ; vaddps        %ymm1,%ymm2,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,21,186,70,0,0        ; vbroadcastss  0x46ba(%rip),%ymm2        # 4834 <_sk_callback_hsw+0x120>
+  DB  196,226,125,24,21,214,70,0,0        ; vbroadcastss  0x46d6(%rip),%ymm2        # 4850 <_sk_callback_hsw+0x120>
   DB  197,228,87,219                      ; vxorps        %ymm3,%ymm3,%ymm3
   DB  197,220,87,228                      ; vxorps        %ymm4,%ymm4,%ymm4
   DB  197,212,87,237                      ; vxorps        %ymm5,%ymm5,%ymm5
@@ -132,13 +132,13 @@ _sk_dither_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  196,66,125,88,8                     ; vpbroadcastd  (%r8),%ymm9
   DB  196,65,61,239,201                   ; vpxor         %ymm9,%ymm8,%ymm9
-  DB  196,98,125,88,21,121,70,0,0         ; vpbroadcastd  0x4679(%rip),%ymm10        # 4838 <_sk_callback_hsw+0x124>
+  DB  196,98,125,88,21,149,70,0,0         ; vpbroadcastd  0x4695(%rip),%ymm10        # 4854 <_sk_callback_hsw+0x124>
   DB  196,65,53,219,218                   ; vpand         %ymm10,%ymm9,%ymm11
   DB  196,193,37,114,243,5                ; vpslld        $0x5,%ymm11,%ymm11
   DB  196,65,61,219,210                   ; vpand         %ymm10,%ymm8,%ymm10
   DB  196,193,45,114,242,4                ; vpslld        $0x4,%ymm10,%ymm10
-  DB  196,98,125,88,37,94,70,0,0          ; vpbroadcastd  0x465e(%rip),%ymm12        # 483c <_sk_callback_hsw+0x128>
-  DB  196,98,125,88,45,89,70,0,0          ; vpbroadcastd  0x4659(%rip),%ymm13        # 4840 <_sk_callback_hsw+0x12c>
+  DB  196,98,125,88,37,122,70,0,0         ; vpbroadcastd  0x467a(%rip),%ymm12        # 4858 <_sk_callback_hsw+0x128>
+  DB  196,98,125,88,45,117,70,0,0         ; vpbroadcastd  0x4675(%rip),%ymm13        # 485c <_sk_callback_hsw+0x12c>
   DB  196,65,53,219,245                   ; vpand         %ymm13,%ymm9,%ymm14
   DB  196,193,13,114,246,2                ; vpslld        $0x2,%ymm14,%ymm14
   DB  196,65,61,219,237                   ; vpand         %ymm13,%ymm8,%ymm13
@@ -153,14 +153,21 @@ _sk_dither_hsw LABEL PROC
   DB  196,65,61,235,194                   ; vpor          %ymm10,%ymm8,%ymm8
   DB  196,65,61,235,193                   ; vpor          %ymm9,%ymm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,11,70,0,0          ; vbroadcastss  0x460b(%rip),%ymm9        # 4844 <_sk_callback_hsw+0x130>
-  DB  196,98,125,24,21,6,70,0,0           ; vbroadcastss  0x4606(%rip),%ymm10        # 4848 <_sk_callback_hsw+0x134>
+  DB  196,98,125,24,13,39,70,0,0          ; vbroadcastss  0x4627(%rip),%ymm9        # 4860 <_sk_callback_hsw+0x130>
+  DB  196,98,125,24,21,34,70,0,0          ; vbroadcastss  0x4622(%rip),%ymm10        # 4864 <_sk_callback_hsw+0x134>
   DB  196,66,61,184,209                   ; vfmadd231ps   %ymm9,%ymm8,%ymm10
   DB  196,98,125,24,64,8                  ; vbroadcastss  0x8(%rax),%ymm8
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
   DB  197,188,88,192                      ; vaddps        %ymm0,%ymm8,%ymm0
   DB  197,188,88,201                      ; vaddps        %ymm1,%ymm8,%ymm1
   DB  197,188,88,210                      ; vaddps        %ymm2,%ymm8,%ymm2
+  DB  197,252,93,195                      ; vminps        %ymm3,%ymm0,%ymm0
+  DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
+  DB  197,188,95,192                      ; vmaxps        %ymm0,%ymm8,%ymm0
+  DB  197,244,93,203                      ; vminps        %ymm3,%ymm1,%ymm1
+  DB  197,188,95,201                      ; vmaxps        %ymm1,%ymm8,%ymm1
+  DB  197,236,93,211                      ; vminps        %ymm3,%ymm2,%ymm2
+  DB  197,188,95,210                      ; vmaxps        %ymm2,%ymm8,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -206,7 +213,7 @@ _sk_clear_hsw LABEL PROC
 PUBLIC _sk_srcatop_hsw
 _sk_srcatop_hsw LABEL PROC
   DB  197,252,89,199                      ; vmulps        %ymm7,%ymm0,%ymm0
-  DB  196,98,125,24,5,122,69,0,0          ; vbroadcastss  0x457a(%rip),%ymm8        # 484c <_sk_callback_hsw+0x138>
+  DB  196,98,125,24,5,121,69,0,0          ; vbroadcastss  0x4579(%rip),%ymm8        # 4868 <_sk_callback_hsw+0x138>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,226,61,184,196                  ; vfmadd231ps   %ymm4,%ymm8,%ymm0
   DB  197,244,89,207                      ; vmulps        %ymm7,%ymm1,%ymm1
@@ -220,7 +227,7 @@ _sk_srcatop_hsw LABEL PROC
 
 PUBLIC _sk_dstatop_hsw
 _sk_dstatop_hsw LABEL PROC
-  DB  196,98,125,24,5,77,69,0,0           ; vbroadcastss  0x454d(%rip),%ymm8        # 4850 <_sk_callback_hsw+0x13c>
+  DB  196,98,125,24,5,76,69,0,0           ; vbroadcastss  0x454c(%rip),%ymm8        # 486c <_sk_callback_hsw+0x13c>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  196,226,101,184,196                 ; vfmadd231ps   %ymm4,%ymm3,%ymm0
@@ -253,7 +260,7 @@ _sk_dstin_hsw LABEL PROC
 
 PUBLIC _sk_srcout_hsw
 _sk_srcout_hsw LABEL PROC
-  DB  196,98,125,24,5,244,68,0,0          ; vbroadcastss  0x44f4(%rip),%ymm8        # 4854 <_sk_callback_hsw+0x140>
+  DB  196,98,125,24,5,243,68,0,0          ; vbroadcastss  0x44f3(%rip),%ymm8        # 4870 <_sk_callback_hsw+0x140>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -264,7 +271,7 @@ _sk_srcout_hsw LABEL PROC
 
 PUBLIC _sk_dstout_hsw
 _sk_dstout_hsw LABEL PROC
-  DB  196,226,125,24,5,215,68,0,0         ; vbroadcastss  0x44d7(%rip),%ymm0        # 4858 <_sk_callback_hsw+0x144>
+  DB  196,226,125,24,5,214,68,0,0         ; vbroadcastss  0x44d6(%rip),%ymm0        # 4874 <_sk_callback_hsw+0x144>
   DB  197,252,92,219                      ; vsubps        %ymm3,%ymm0,%ymm3
   DB  197,228,89,196                      ; vmulps        %ymm4,%ymm3,%ymm0
   DB  197,228,89,205                      ; vmulps        %ymm5,%ymm3,%ymm1
@@ -275,7 +282,7 @@ _sk_dstout_hsw LABEL PROC
 
 PUBLIC _sk_srcover_hsw
 _sk_srcover_hsw LABEL PROC
-  DB  196,98,125,24,5,186,68,0,0          ; vbroadcastss  0x44ba(%rip),%ymm8        # 485c <_sk_callback_hsw+0x148>
+  DB  196,98,125,24,5,185,68,0,0          ; vbroadcastss  0x44b9(%rip),%ymm8        # 4878 <_sk_callback_hsw+0x148>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,93,184,192                  ; vfmadd231ps   %ymm8,%ymm4,%ymm0
   DB  196,194,85,184,200                  ; vfmadd231ps   %ymm8,%ymm5,%ymm1
@@ -286,7 +293,7 @@ _sk_srcover_hsw LABEL PROC
 
 PUBLIC _sk_dstover_hsw
 _sk_dstover_hsw LABEL PROC
-  DB  196,98,125,24,5,153,68,0,0          ; vbroadcastss  0x4499(%rip),%ymm8        # 4860 <_sk_callback_hsw+0x14c>
+  DB  196,98,125,24,5,152,68,0,0          ; vbroadcastss  0x4498(%rip),%ymm8        # 487c <_sk_callback_hsw+0x14c>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  196,226,61,168,196                  ; vfmadd213ps   %ymm4,%ymm8,%ymm0
   DB  196,226,61,168,205                  ; vfmadd213ps   %ymm5,%ymm8,%ymm1
@@ -306,7 +313,7 @@ _sk_modulate_hsw LABEL PROC
 
 PUBLIC _sk_multiply_hsw
 _sk_multiply_hsw LABEL PROC
-  DB  196,98,125,24,5,100,68,0,0          ; vbroadcastss  0x4464(%rip),%ymm8        # 4864 <_sk_callback_hsw+0x150>
+  DB  196,98,125,24,5,99,68,0,0           ; vbroadcastss  0x4463(%rip),%ymm8        # 4880 <_sk_callback_hsw+0x150>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,208                       ; vmulps        %ymm0,%ymm9,%ymm10
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -348,7 +355,7 @@ _sk_screen_hsw LABEL PROC
 
 PUBLIC _sk_xor__hsw
 _sk_xor__hsw LABEL PROC
-  DB  196,98,125,24,5,223,67,0,0          ; vbroadcastss  0x43df(%rip),%ymm8        # 4868 <_sk_callback_hsw+0x154>
+  DB  196,98,125,24,5,222,67,0,0          ; vbroadcastss  0x43de(%rip),%ymm8        # 4884 <_sk_callback_hsw+0x154>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -380,7 +387,7 @@ _sk_darken_hsw LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,95,209                  ; vmaxps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,103,67,0,0          ; vbroadcastss  0x4367(%rip),%ymm8        # 486c <_sk_callback_hsw+0x158>
+  DB  196,98,125,24,5,102,67,0,0          ; vbroadcastss  0x4366(%rip),%ymm8        # 4888 <_sk_callback_hsw+0x158>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -403,7 +410,7 @@ _sk_lighten_hsw LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,22,67,0,0           ; vbroadcastss  0x4316(%rip),%ymm8        # 4870 <_sk_callback_hsw+0x15c>
+  DB  196,98,125,24,5,21,67,0,0           ; vbroadcastss  0x4315(%rip),%ymm8        # 488c <_sk_callback_hsw+0x15c>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -429,7 +436,7 @@ _sk_difference_hsw LABEL PROC
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,185,66,0,0          ; vbroadcastss  0x42b9(%rip),%ymm8        # 4874 <_sk_callback_hsw+0x160>
+  DB  196,98,125,24,5,184,66,0,0          ; vbroadcastss  0x42b8(%rip),%ymm8        # 4890 <_sk_callback_hsw+0x160>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -449,7 +456,7 @@ _sk_exclusion_hsw LABEL PROC
   DB  197,236,89,214                      ; vmulps        %ymm6,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,119,66,0,0          ; vbroadcastss  0x4277(%rip),%ymm8        # 4878 <_sk_callback_hsw+0x164>
+  DB  196,98,125,24,5,118,66,0,0          ; vbroadcastss  0x4276(%rip),%ymm8        # 4894 <_sk_callback_hsw+0x164>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  196,194,69,184,216                  ; vfmadd231ps   %ymm8,%ymm7,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -457,7 +464,7 @@ _sk_exclusion_hsw LABEL PROC
 
 PUBLIC _sk_colorburn_hsw
 _sk_colorburn_hsw LABEL PROC
-  DB  196,98,125,24,5,101,66,0,0          ; vbroadcastss  0x4265(%rip),%ymm8        # 487c <_sk_callback_hsw+0x168>
+  DB  196,98,125,24,5,100,66,0,0          ; vbroadcastss  0x4264(%rip),%ymm8        # 4898 <_sk_callback_hsw+0x168>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,216                       ; vmulps        %ymm0,%ymm9,%ymm11
   DB  196,65,44,87,210                    ; vxorps        %ymm10,%ymm10,%ymm10
@@ -513,7 +520,7 @@ _sk_colorburn_hsw LABEL PROC
 PUBLIC _sk_colordodge_hsw
 _sk_colordodge_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
-  DB  196,98,125,24,13,112,65,0,0         ; vbroadcastss  0x4170(%rip),%ymm9        # 4880 <_sk_callback_hsw+0x16c>
+  DB  196,98,125,24,13,111,65,0,0         ; vbroadcastss  0x416f(%rip),%ymm9        # 489c <_sk_callback_hsw+0x16c>
   DB  197,52,92,215                       ; vsubps        %ymm7,%ymm9,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,52,92,203                       ; vsubps        %ymm3,%ymm9,%ymm9
@@ -564,7 +571,7 @@ _sk_colordodge_hsw LABEL PROC
 
 PUBLIC _sk_hardlight_hsw
 _sk_hardlight_hsw LABEL PROC
-  DB  196,98,125,24,5,145,64,0,0          ; vbroadcastss  0x4091(%rip),%ymm8        # 4884 <_sk_callback_hsw+0x170>
+  DB  196,98,125,24,5,144,64,0,0          ; vbroadcastss  0x4090(%rip),%ymm8        # 48a0 <_sk_callback_hsw+0x170>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -613,7 +620,7 @@ _sk_hardlight_hsw LABEL PROC
 
 PUBLIC _sk_overlay_hsw
 _sk_overlay_hsw LABEL PROC
-  DB  196,98,125,24,5,201,63,0,0          ; vbroadcastss  0x3fc9(%rip),%ymm8        # 4888 <_sk_callback_hsw+0x174>
+  DB  196,98,125,24,5,200,63,0,0          ; vbroadcastss  0x3fc8(%rip),%ymm8        # 48a4 <_sk_callback_hsw+0x174>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -673,10 +680,10 @@ _sk_softlight_hsw LABEL PROC
   DB  196,65,20,88,197                    ; vaddps        %ymm13,%ymm13,%ymm8
   DB  196,65,60,88,192                    ; vaddps        %ymm8,%ymm8,%ymm8
   DB  196,66,61,168,192                   ; vfmadd213ps   %ymm8,%ymm8,%ymm8
-  DB  196,98,125,24,29,208,62,0,0         ; vbroadcastss  0x3ed0(%rip),%ymm11        # 4890 <_sk_callback_hsw+0x17c>
+  DB  196,98,125,24,29,207,62,0,0         ; vbroadcastss  0x3ecf(%rip),%ymm11        # 48ac <_sk_callback_hsw+0x17c>
   DB  196,65,20,88,227                    ; vaddps        %ymm11,%ymm13,%ymm12
   DB  196,65,28,89,192                    ; vmulps        %ymm8,%ymm12,%ymm8
-  DB  196,98,125,24,37,193,62,0,0         ; vbroadcastss  0x3ec1(%rip),%ymm12        # 4894 <_sk_callback_hsw+0x180>
+  DB  196,98,125,24,37,192,62,0,0         ; vbroadcastss  0x3ec0(%rip),%ymm12        # 48b0 <_sk_callback_hsw+0x180>
   DB  196,66,21,184,196                   ; vfmadd231ps   %ymm12,%ymm13,%ymm8
   DB  196,65,124,82,245                   ; vrsqrtps      %ymm13,%ymm14
   DB  196,65,124,83,246                   ; vrcpps        %ymm14,%ymm14
@@ -686,7 +693,7 @@ _sk_softlight_hsw LABEL PROC
   DB  197,4,194,255,2                     ; vcmpleps      %ymm7,%ymm15,%ymm15
   DB  196,67,13,74,240,240                ; vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   DB  197,116,88,249                      ; vaddps        %ymm1,%ymm1,%ymm15
-  DB  196,98,125,24,5,132,62,0,0          ; vbroadcastss  0x3e84(%rip),%ymm8        # 488c <_sk_callback_hsw+0x178>
+  DB  196,98,125,24,5,131,62,0,0          ; vbroadcastss  0x3e83(%rip),%ymm8        # 48a8 <_sk_callback_hsw+0x178>
   DB  196,65,60,92,237                    ; vsubps        %ymm13,%ymm8,%ymm13
   DB  197,132,92,195                      ; vsubps        %ymm3,%ymm15,%ymm0
   DB  196,98,125,168,235                  ; vfmadd213ps   %ymm3,%ymm0,%ymm13
@@ -799,11 +806,11 @@ _sk_hue_hsw LABEL PROC
   DB  196,65,28,89,210                    ; vmulps        %ymm10,%ymm12,%ymm10
   DB  196,65,44,94,214                    ; vdivps        %ymm14,%ymm10,%ymm10
   DB  196,67,45,74,224,240                ; vblendvps     %ymm15,%ymm8,%ymm10,%ymm12
-  DB  196,98,125,24,53,131,60,0,0         ; vbroadcastss  0x3c83(%rip),%ymm14        # 4898 <_sk_callback_hsw+0x184>
-  DB  196,98,125,24,61,126,60,0,0         ; vbroadcastss  0x3c7e(%rip),%ymm15        # 489c <_sk_callback_hsw+0x188>
+  DB  196,98,125,24,53,130,60,0,0         ; vbroadcastss  0x3c82(%rip),%ymm14        # 48b4 <_sk_callback_hsw+0x184>
+  DB  196,98,125,24,61,125,60,0,0         ; vbroadcastss  0x3c7d(%rip),%ymm15        # 48b8 <_sk_callback_hsw+0x188>
   DB  196,65,84,89,239                    ; vmulps        %ymm15,%ymm5,%ymm13
   DB  196,66,93,184,238                   ; vfmadd231ps   %ymm14,%ymm4,%ymm13
-  DB  196,226,125,24,5,111,60,0,0         ; vbroadcastss  0x3c6f(%rip),%ymm0        # 48a0 <_sk_callback_hsw+0x18c>
+  DB  196,226,125,24,5,110,60,0,0         ; vbroadcastss  0x3c6e(%rip),%ymm0        # 48bc <_sk_callback_hsw+0x18c>
   DB  196,98,77,184,232                   ; vfmadd231ps   %ymm0,%ymm6,%ymm13
   DB  196,65,116,89,215                   ; vmulps        %ymm15,%ymm1,%ymm10
   DB  196,66,53,184,214                   ; vfmadd231ps   %ymm14,%ymm9,%ymm10
@@ -858,7 +865,7 @@ _sk_hue_hsw LABEL PROC
   DB  196,193,124,95,192                  ; vmaxps        %ymm8,%ymm0,%ymm0
   DB  196,65,36,95,200                    ; vmaxps        %ymm8,%ymm11,%ymm9
   DB  196,65,116,95,192                   ; vmaxps        %ymm8,%ymm1,%ymm8
-  DB  196,226,125,24,13,92,59,0,0         ; vbroadcastss  0x3b5c(%rip),%ymm1        # 48a4 <_sk_callback_hsw+0x190>
+  DB  196,226,125,24,13,91,59,0,0         ; vbroadcastss  0x3b5b(%rip),%ymm1        # 48c0 <_sk_callback_hsw+0x190>
   DB  197,116,92,215                      ; vsubps        %ymm7,%ymm1,%ymm10
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  197,116,92,219                      ; vsubps        %ymm3,%ymm1,%ymm11
@@ -912,11 +919,11 @@ _sk_saturation_hsw LABEL PROC
   DB  196,65,28,89,210                    ; vmulps        %ymm10,%ymm12,%ymm10
   DB  196,65,44,94,214                    ; vdivps        %ymm14,%ymm10,%ymm10
   DB  196,67,45,74,224,240                ; vblendvps     %ymm15,%ymm8,%ymm10,%ymm12
-  DB  196,98,125,24,53,109,58,0,0         ; vbroadcastss  0x3a6d(%rip),%ymm14        # 48a8 <_sk_callback_hsw+0x194>
-  DB  196,98,125,24,61,104,58,0,0         ; vbroadcastss  0x3a68(%rip),%ymm15        # 48ac <_sk_callback_hsw+0x198>
+  DB  196,98,125,24,53,108,58,0,0         ; vbroadcastss  0x3a6c(%rip),%ymm14        # 48c4 <_sk_callback_hsw+0x194>
+  DB  196,98,125,24,61,103,58,0,0         ; vbroadcastss  0x3a67(%rip),%ymm15        # 48c8 <_sk_callback_hsw+0x198>
   DB  196,65,84,89,239                    ; vmulps        %ymm15,%ymm5,%ymm13
   DB  196,66,93,184,238                   ; vfmadd231ps   %ymm14,%ymm4,%ymm13
-  DB  196,226,125,24,5,89,58,0,0          ; vbroadcastss  0x3a59(%rip),%ymm0        # 48b0 <_sk_callback_hsw+0x19c>
+  DB  196,226,125,24,5,88,58,0,0          ; vbroadcastss  0x3a58(%rip),%ymm0        # 48cc <_sk_callback_hsw+0x19c>
   DB  196,98,77,184,232                   ; vfmadd231ps   %ymm0,%ymm6,%ymm13
   DB  196,65,116,89,215                   ; vmulps        %ymm15,%ymm1,%ymm10
   DB  196,66,53,184,214                   ; vfmadd231ps   %ymm14,%ymm9,%ymm10
@@ -971,7 +978,7 @@ _sk_saturation_hsw LABEL PROC
   DB  196,193,124,95,192                  ; vmaxps        %ymm8,%ymm0,%ymm0
   DB  196,65,36,95,200                    ; vmaxps        %ymm8,%ymm11,%ymm9
   DB  196,65,116,95,192                   ; vmaxps        %ymm8,%ymm1,%ymm8
-  DB  196,226,125,24,13,70,57,0,0         ; vbroadcastss  0x3946(%rip),%ymm1        # 48b4 <_sk_callback_hsw+0x1a0>
+  DB  196,226,125,24,13,69,57,0,0         ; vbroadcastss  0x3945(%rip),%ymm1        # 48d0 <_sk_callback_hsw+0x1a0>
   DB  197,116,92,215                      ; vsubps        %ymm7,%ymm1,%ymm10
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  197,116,92,219                      ; vsubps        %ymm3,%ymm1,%ymm11
@@ -999,11 +1006,11 @@ _sk_color_hsw LABEL PROC
   DB  197,108,89,199                      ; vmulps        %ymm7,%ymm2,%ymm8
   DB  197,116,89,215                      ; vmulps        %ymm7,%ymm1,%ymm10
   DB  197,52,89,223                       ; vmulps        %ymm7,%ymm9,%ymm11
-  DB  196,98,125,24,45,217,56,0,0         ; vbroadcastss  0x38d9(%rip),%ymm13        # 48b8 <_sk_callback_hsw+0x1a4>
-  DB  196,98,125,24,53,212,56,0,0         ; vbroadcastss  0x38d4(%rip),%ymm14        # 48bc <_sk_callback_hsw+0x1a8>
+  DB  196,98,125,24,45,216,56,0,0         ; vbroadcastss  0x38d8(%rip),%ymm13        # 48d4 <_sk_callback_hsw+0x1a4>
+  DB  196,98,125,24,53,211,56,0,0         ; vbroadcastss  0x38d3(%rip),%ymm14        # 48d8 <_sk_callback_hsw+0x1a8>
   DB  196,65,84,89,230                    ; vmulps        %ymm14,%ymm5,%ymm12
   DB  196,66,93,184,229                   ; vfmadd231ps   %ymm13,%ymm4,%ymm12
-  DB  196,98,125,24,61,197,56,0,0         ; vbroadcastss  0x38c5(%rip),%ymm15        # 48c0 <_sk_callback_hsw+0x1ac>
+  DB  196,98,125,24,61,196,56,0,0         ; vbroadcastss  0x38c4(%rip),%ymm15        # 48dc <_sk_callback_hsw+0x1ac>
   DB  196,66,77,184,231                   ; vfmadd231ps   %ymm15,%ymm6,%ymm12
   DB  196,65,44,89,206                    ; vmulps        %ymm14,%ymm10,%ymm9
   DB  196,66,61,184,205                   ; vfmadd231ps   %ymm13,%ymm8,%ymm9
@@ -1059,7 +1066,7 @@ _sk_color_hsw LABEL PROC
   DB  196,193,116,95,206                  ; vmaxps        %ymm14,%ymm1,%ymm1
   DB  196,65,44,95,198                    ; vmaxps        %ymm14,%ymm10,%ymm8
   DB  196,65,124,95,206                   ; vmaxps        %ymm14,%ymm0,%ymm9
-  DB  196,226,125,24,5,167,55,0,0         ; vbroadcastss  0x37a7(%rip),%ymm0        # 48c4 <_sk_callback_hsw+0x1b0>
+  DB  196,226,125,24,5,166,55,0,0         ; vbroadcastss  0x37a6(%rip),%ymm0        # 48e0 <_sk_callback_hsw+0x1b0>
   DB  197,124,92,215                      ; vsubps        %ymm7,%ymm0,%ymm10
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  197,124,92,219                      ; vsubps        %ymm3,%ymm0,%ymm11
@@ -1087,11 +1094,11 @@ _sk_luminosity_hsw LABEL PROC
   DB  197,100,89,196                      ; vmulps        %ymm4,%ymm3,%ymm8
   DB  197,100,89,213                      ; vmulps        %ymm5,%ymm3,%ymm10
   DB  197,100,89,222                      ; vmulps        %ymm6,%ymm3,%ymm11
-  DB  196,98,125,24,45,58,55,0,0          ; vbroadcastss  0x373a(%rip),%ymm13        # 48c8 <_sk_callback_hsw+0x1b4>
-  DB  196,98,125,24,53,53,55,0,0          ; vbroadcastss  0x3735(%rip),%ymm14        # 48cc <_sk_callback_hsw+0x1b8>
+  DB  196,98,125,24,45,57,55,0,0          ; vbroadcastss  0x3739(%rip),%ymm13        # 48e4 <_sk_callback_hsw+0x1b4>
+  DB  196,98,125,24,53,52,55,0,0          ; vbroadcastss  0x3734(%rip),%ymm14        # 48e8 <_sk_callback_hsw+0x1b8>
   DB  196,65,116,89,230                   ; vmulps        %ymm14,%ymm1,%ymm12
   DB  196,66,109,184,229                  ; vfmadd231ps   %ymm13,%ymm2,%ymm12
-  DB  196,98,125,24,61,38,55,0,0          ; vbroadcastss  0x3726(%rip),%ymm15        # 48d0 <_sk_callback_hsw+0x1bc>
+  DB  196,98,125,24,61,37,55,0,0          ; vbroadcastss  0x3725(%rip),%ymm15        # 48ec <_sk_callback_hsw+0x1bc>
   DB  196,66,53,184,231                   ; vfmadd231ps   %ymm15,%ymm9,%ymm12
   DB  196,65,44,89,206                    ; vmulps        %ymm14,%ymm10,%ymm9
   DB  196,66,61,184,205                   ; vfmadd231ps   %ymm13,%ymm8,%ymm9
@@ -1147,7 +1154,7 @@ _sk_luminosity_hsw LABEL PROC
   DB  196,193,116,95,206                  ; vmaxps        %ymm14,%ymm1,%ymm1
   DB  196,65,44,95,198                    ; vmaxps        %ymm14,%ymm10,%ymm8
   DB  196,65,124,95,206                   ; vmaxps        %ymm14,%ymm0,%ymm9
-  DB  196,226,125,24,5,8,54,0,0           ; vbroadcastss  0x3608(%rip),%ymm0        # 48d4 <_sk_callback_hsw+0x1c0>
+  DB  196,226,125,24,5,7,54,0,0           ; vbroadcastss  0x3607(%rip),%ymm0        # 48f0 <_sk_callback_hsw+0x1c0>
   DB  197,124,92,215                      ; vsubps        %ymm7,%ymm0,%ymm10
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  197,124,92,219                      ; vsubps        %ymm3,%ymm0,%ymm11
@@ -1177,7 +1184,7 @@ _sk_clamp_0_hsw LABEL PROC
 
 PUBLIC _sk_clamp_1_hsw
 _sk_clamp_1_hsw LABEL PROC
-  DB  196,98,125,24,5,161,53,0,0          ; vbroadcastss  0x35a1(%rip),%ymm8        # 48d8 <_sk_callback_hsw+0x1c4>
+  DB  196,98,125,24,5,160,53,0,0          ; vbroadcastss  0x35a0(%rip),%ymm8        # 48f4 <_sk_callback_hsw+0x1c4>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
@@ -1187,7 +1194,7 @@ _sk_clamp_1_hsw LABEL PROC
 
 PUBLIC _sk_clamp_a_hsw
 _sk_clamp_a_hsw LABEL PROC
-  DB  196,98,125,24,5,132,53,0,0          ; vbroadcastss  0x3584(%rip),%ymm8        # 48dc <_sk_callback_hsw+0x1c8>
+  DB  196,98,125,24,5,131,53,0,0          ; vbroadcastss  0x3583(%rip),%ymm8        # 48f8 <_sk_callback_hsw+0x1c8>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  197,252,93,195                      ; vminps        %ymm3,%ymm0,%ymm0
   DB  197,244,93,203                      ; vminps        %ymm3,%ymm1,%ymm1
@@ -1259,7 +1266,7 @@ PUBLIC _sk_unpremul_hsw
 _sk_unpremul_hsw LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,200,0                ; vcmpeqps      %ymm8,%ymm3,%ymm9
-  DB  196,98,125,24,21,204,52,0,0         ; vbroadcastss  0x34cc(%rip),%ymm10        # 48e0 <_sk_callback_hsw+0x1cc>
+  DB  196,98,125,24,21,203,52,0,0         ; vbroadcastss  0x34cb(%rip),%ymm10        # 48fc <_sk_callback_hsw+0x1cc>
   DB  197,44,94,211                       ; vdivps        %ymm3,%ymm10,%ymm10
   DB  196,67,45,74,192,144                ; vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
@@ -1270,16 +1277,16 @@ _sk_unpremul_hsw LABEL PROC
 
 PUBLIC _sk_from_srgb_hsw
 _sk_from_srgb_hsw LABEL PROC
-  DB  196,98,125,24,5,173,52,0,0          ; vbroadcastss  0x34ad(%rip),%ymm8        # 48e4 <_sk_callback_hsw+0x1d0>
+  DB  196,98,125,24,5,172,52,0,0          ; vbroadcastss  0x34ac(%rip),%ymm8        # 4900 <_sk_callback_hsw+0x1d0>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  197,124,89,208                      ; vmulps        %ymm0,%ymm0,%ymm10
-  DB  196,98,125,24,29,159,52,0,0         ; vbroadcastss  0x349f(%rip),%ymm11        # 48e8 <_sk_callback_hsw+0x1d4>
-  DB  196,98,125,24,37,154,52,0,0         ; vbroadcastss  0x349a(%rip),%ymm12        # 48ec <_sk_callback_hsw+0x1d8>
+  DB  196,98,125,24,29,158,52,0,0         ; vbroadcastss  0x349e(%rip),%ymm11        # 4904 <_sk_callback_hsw+0x1d4>
+  DB  196,98,125,24,37,153,52,0,0         ; vbroadcastss  0x3499(%rip),%ymm12        # 4908 <_sk_callback_hsw+0x1d8>
   DB  196,65,124,40,236                   ; vmovaps       %ymm12,%ymm13
   DB  196,66,125,168,235                  ; vfmadd213ps   %ymm11,%ymm0,%ymm13
-  DB  196,98,125,24,53,139,52,0,0         ; vbroadcastss  0x348b(%rip),%ymm14        # 48f0 <_sk_callback_hsw+0x1dc>
+  DB  196,98,125,24,53,138,52,0,0         ; vbroadcastss  0x348a(%rip),%ymm14        # 490c <_sk_callback_hsw+0x1dc>
   DB  196,66,45,168,238                   ; vfmadd213ps   %ymm14,%ymm10,%ymm13
-  DB  196,98,125,24,21,129,52,0,0         ; vbroadcastss  0x3481(%rip),%ymm10        # 48f4 <_sk_callback_hsw+0x1e0>
+  DB  196,98,125,24,21,128,52,0,0         ; vbroadcastss  0x3480(%rip),%ymm10        # 4910 <_sk_callback_hsw+0x1e0>
   DB  196,193,124,194,194,1               ; vcmpltps      %ymm10,%ymm0,%ymm0
   DB  196,195,21,74,193,0                 ; vblendvps     %ymm0,%ymm9,%ymm13,%ymm0
   DB  196,65,116,89,200                   ; vmulps        %ymm8,%ymm1,%ymm9
@@ -1303,16 +1310,16 @@ _sk_to_srgb_hsw LABEL PROC
   DB  197,124,82,192                      ; vrsqrtps      %ymm0,%ymm8
   DB  196,65,124,83,200                   ; vrcpps        %ymm8,%ymm9
   DB  196,65,124,82,208                   ; vrsqrtps      %ymm8,%ymm10
-  DB  196,98,125,24,5,27,52,0,0           ; vbroadcastss  0x341b(%rip),%ymm8        # 48f8 <_sk_callback_hsw+0x1e4>
+  DB  196,98,125,24,5,26,52,0,0           ; vbroadcastss  0x341a(%rip),%ymm8        # 4914 <_sk_callback_hsw+0x1e4>
   DB  196,65,124,89,216                   ; vmulps        %ymm8,%ymm0,%ymm11
-  DB  196,98,125,24,37,17,52,0,0          ; vbroadcastss  0x3411(%rip),%ymm12        # 48fc <_sk_callback_hsw+0x1e8>
-  DB  196,98,125,24,45,12,52,0,0          ; vbroadcastss  0x340c(%rip),%ymm13        # 4900 <_sk_callback_hsw+0x1ec>
+  DB  196,98,125,24,37,16,52,0,0          ; vbroadcastss  0x3410(%rip),%ymm12        # 4918 <_sk_callback_hsw+0x1e8>
+  DB  196,98,125,24,45,11,52,0,0          ; vbroadcastss  0x340b(%rip),%ymm13        # 491c <_sk_callback_hsw+0x1ec>
   DB  196,66,21,168,204                   ; vfmadd213ps   %ymm12,%ymm13,%ymm9
-  DB  196,98,125,24,53,2,52,0,0           ; vbroadcastss  0x3402(%rip),%ymm14        # 4904 <_sk_callback_hsw+0x1f0>
+  DB  196,98,125,24,53,1,52,0,0           ; vbroadcastss  0x3401(%rip),%ymm14        # 4920 <_sk_callback_hsw+0x1f0>
   DB  196,66,13,184,202                   ; vfmadd231ps   %ymm10,%ymm14,%ymm9
-  DB  196,98,125,24,21,248,51,0,0         ; vbroadcastss  0x33f8(%rip),%ymm10        # 4908 <_sk_callback_hsw+0x1f4>
+  DB  196,98,125,24,21,247,51,0,0         ; vbroadcastss  0x33f7(%rip),%ymm10        # 4924 <_sk_callback_hsw+0x1f4>
   DB  196,65,44,93,201                    ; vminps        %ymm9,%ymm10,%ymm9
-  DB  196,98,125,24,61,238,51,0,0         ; vbroadcastss  0x33ee(%rip),%ymm15        # 490c <_sk_callback_hsw+0x1f8>
+  DB  196,98,125,24,61,237,51,0,0         ; vbroadcastss  0x33ed(%rip),%ymm15        # 4928 <_sk_callback_hsw+0x1f8>
   DB  196,193,124,194,199,1               ; vcmpltps      %ymm15,%ymm0,%ymm0
   DB  196,195,53,74,195,0                 ; vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   DB  197,124,82,201                      ; vrsqrtps      %ymm1,%ymm9
@@ -1343,26 +1350,26 @@ _sk_rgb_to_hsl_hsw LABEL PROC
   DB  197,124,93,201                      ; vminps        %ymm1,%ymm0,%ymm9
   DB  197,52,93,202                       ; vminps        %ymm2,%ymm9,%ymm9
   DB  196,65,60,92,209                    ; vsubps        %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,29,104,51,0,0         ; vbroadcastss  0x3368(%rip),%ymm11        # 4910 <_sk_callback_hsw+0x1fc>
+  DB  196,98,125,24,29,103,51,0,0         ; vbroadcastss  0x3367(%rip),%ymm11        # 492c <_sk_callback_hsw+0x1fc>
   DB  196,65,36,94,218                    ; vdivps        %ymm10,%ymm11,%ymm11
   DB  197,116,92,226                      ; vsubps        %ymm2,%ymm1,%ymm12
   DB  197,116,194,234,1                   ; vcmpltps      %ymm2,%ymm1,%ymm13
-  DB  196,98,125,24,53,85,51,0,0          ; vbroadcastss  0x3355(%rip),%ymm14        # 4914 <_sk_callback_hsw+0x200>
+  DB  196,98,125,24,53,84,51,0,0          ; vbroadcastss  0x3354(%rip),%ymm14        # 4930 <_sk_callback_hsw+0x200>
   DB  196,65,4,87,255                     ; vxorps        %ymm15,%ymm15,%ymm15
   DB  196,67,5,74,238,208                 ; vblendvps     %ymm13,%ymm14,%ymm15,%ymm13
   DB  196,66,37,168,229                   ; vfmadd213ps   %ymm13,%ymm11,%ymm12
   DB  197,236,92,208                      ; vsubps        %ymm0,%ymm2,%ymm2
   DB  197,124,92,233                      ; vsubps        %ymm1,%ymm0,%ymm13
-  DB  196,98,125,24,53,60,51,0,0          ; vbroadcastss  0x333c(%rip),%ymm14        # 491c <_sk_callback_hsw+0x208>
+  DB  196,98,125,24,53,59,51,0,0          ; vbroadcastss  0x333b(%rip),%ymm14        # 4938 <_sk_callback_hsw+0x208>
   DB  196,66,37,168,238                   ; vfmadd213ps   %ymm14,%ymm11,%ymm13
-  DB  196,98,125,24,53,42,51,0,0          ; vbroadcastss  0x332a(%rip),%ymm14        # 4918 <_sk_callback_hsw+0x204>
+  DB  196,98,125,24,53,41,51,0,0          ; vbroadcastss  0x3329(%rip),%ymm14        # 4934 <_sk_callback_hsw+0x204>
   DB  196,194,37,168,214                  ; vfmadd213ps   %ymm14,%ymm11,%ymm2
   DB  197,188,194,201,0                   ; vcmpeqps      %ymm1,%ymm8,%ymm1
   DB  196,227,21,74,202,16                ; vblendvps     %ymm1,%ymm2,%ymm13,%ymm1
   DB  197,188,194,192,0                   ; vcmpeqps      %ymm0,%ymm8,%ymm0
   DB  196,195,117,74,196,0                ; vblendvps     %ymm0,%ymm12,%ymm1,%ymm0
   DB  196,193,60,88,201                   ; vaddps        %ymm9,%ymm8,%ymm1
-  DB  196,98,125,24,29,13,51,0,0          ; vbroadcastss  0x330d(%rip),%ymm11        # 4924 <_sk_callback_hsw+0x210>
+  DB  196,98,125,24,29,12,51,0,0          ; vbroadcastss  0x330c(%rip),%ymm11        # 4940 <_sk_callback_hsw+0x210>
   DB  196,193,116,89,211                  ; vmulps        %ymm11,%ymm1,%ymm2
   DB  197,36,194,218,1                    ; vcmpltps      %ymm2,%ymm11,%ymm11
   DB  196,65,12,92,224                    ; vsubps        %ymm8,%ymm14,%ymm12
@@ -1372,7 +1379,7 @@ _sk_rgb_to_hsl_hsw LABEL PROC
   DB  197,172,94,201                      ; vdivps        %ymm1,%ymm10,%ymm1
   DB  196,195,125,74,199,128              ; vblendvps     %ymm8,%ymm15,%ymm0,%ymm0
   DB  196,195,117,74,207,128              ; vblendvps     %ymm8,%ymm15,%ymm1,%ymm1
-  DB  196,98,125,24,5,208,50,0,0          ; vbroadcastss  0x32d0(%rip),%ymm8        # 4920 <_sk_callback_hsw+0x20c>
+  DB  196,98,125,24,5,207,50,0,0          ; vbroadcastss  0x32cf(%rip),%ymm8        # 493c <_sk_callback_hsw+0x20c>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -1387,30 +1394,30 @@ _sk_hsl_to_rgb_hsw LABEL PROC
   DB  197,252,17,28,36                    ; vmovups       %ymm3,(%rsp)
   DB  197,252,40,233                      ; vmovaps       %ymm1,%ymm5
   DB  197,252,40,224                      ; vmovaps       %ymm0,%ymm4
-  DB  196,98,125,24,5,151,50,0,0          ; vbroadcastss  0x3297(%rip),%ymm8        # 4928 <_sk_callback_hsw+0x214>
+  DB  196,98,125,24,5,150,50,0,0          ; vbroadcastss  0x3296(%rip),%ymm8        # 4944 <_sk_callback_hsw+0x214>
   DB  197,60,194,202,2                    ; vcmpleps      %ymm2,%ymm8,%ymm9
   DB  197,84,89,210                       ; vmulps        %ymm2,%ymm5,%ymm10
   DB  196,65,84,92,218                    ; vsubps        %ymm10,%ymm5,%ymm11
   DB  196,67,45,74,203,144                ; vblendvps     %ymm9,%ymm11,%ymm10,%ymm9
   DB  197,52,88,210                       ; vaddps        %ymm2,%ymm9,%ymm10
-  DB  196,98,125,24,13,122,50,0,0         ; vbroadcastss  0x327a(%rip),%ymm9        # 492c <_sk_callback_hsw+0x218>
+  DB  196,98,125,24,13,121,50,0,0         ; vbroadcastss  0x3279(%rip),%ymm9        # 4948 <_sk_callback_hsw+0x218>
   DB  196,66,109,170,202                  ; vfmsub213ps   %ymm10,%ymm2,%ymm9
-  DB  196,98,125,24,29,112,50,0,0         ; vbroadcastss  0x3270(%rip),%ymm11        # 4930 <_sk_callback_hsw+0x21c>
+  DB  196,98,125,24,29,111,50,0,0         ; vbroadcastss  0x326f(%rip),%ymm11        # 494c <_sk_callback_hsw+0x21c>
   DB  196,65,92,88,219                    ; vaddps        %ymm11,%ymm4,%ymm11
   DB  196,67,125,8,227,1                  ; vroundps      $0x1,%ymm11,%ymm12
   DB  196,65,36,92,252                    ; vsubps        %ymm12,%ymm11,%ymm15
   DB  196,65,44,92,217                    ; vsubps        %ymm9,%ymm10,%ymm11
-  DB  196,98,125,24,45,90,50,0,0          ; vbroadcastss  0x325a(%rip),%ymm13        # 4938 <_sk_callback_hsw+0x224>
+  DB  196,98,125,24,45,89,50,0,0          ; vbroadcastss  0x3259(%rip),%ymm13        # 4954 <_sk_callback_hsw+0x224>
   DB  196,193,4,89,197                    ; vmulps        %ymm13,%ymm15,%ymm0
-  DB  196,98,125,24,53,80,50,0,0          ; vbroadcastss  0x3250(%rip),%ymm14        # 493c <_sk_callback_hsw+0x228>
+  DB  196,98,125,24,53,79,50,0,0          ; vbroadcastss  0x324f(%rip),%ymm14        # 4958 <_sk_callback_hsw+0x228>
   DB  197,12,92,224                       ; vsubps        %ymm0,%ymm14,%ymm12
   DB  196,66,37,168,225                   ; vfmadd213ps   %ymm9,%ymm11,%ymm12
-  DB  196,226,125,24,29,54,50,0,0         ; vbroadcastss  0x3236(%rip),%ymm3        # 4934 <_sk_callback_hsw+0x220>
+  DB  196,226,125,24,29,53,50,0,0         ; vbroadcastss  0x3235(%rip),%ymm3        # 4950 <_sk_callback_hsw+0x220>
   DB  196,193,100,194,255,2               ; vcmpleps      %ymm15,%ymm3,%ymm7
   DB  196,195,29,74,249,112               ; vblendvps     %ymm7,%ymm9,%ymm12,%ymm7
   DB  196,65,60,194,231,2                 ; vcmpleps      %ymm15,%ymm8,%ymm12
   DB  196,227,45,74,255,192               ; vblendvps     %ymm12,%ymm7,%ymm10,%ymm7
-  DB  196,98,125,24,37,33,50,0,0          ; vbroadcastss  0x3221(%rip),%ymm12        # 4940 <_sk_callback_hsw+0x22c>
+  DB  196,98,125,24,37,32,50,0,0          ; vbroadcastss  0x3220(%rip),%ymm12        # 495c <_sk_callback_hsw+0x22c>
   DB  196,65,28,194,255,2                 ; vcmpleps      %ymm15,%ymm12,%ymm15
   DB  196,194,37,168,193                  ; vfmadd213ps   %ymm9,%ymm11,%ymm0
   DB  196,99,125,74,255,240               ; vblendvps     %ymm15,%ymm7,%ymm0,%ymm15
@@ -1426,7 +1433,7 @@ _sk_hsl_to_rgb_hsw LABEL PROC
   DB  197,156,194,192,2                   ; vcmpleps      %ymm0,%ymm12,%ymm0
   DB  196,194,37,168,249                  ; vfmadd213ps   %ymm9,%ymm11,%ymm7
   DB  196,227,69,74,201,0                 ; vblendvps     %ymm0,%ymm1,%ymm7,%ymm1
-  DB  196,226,125,24,5,205,49,0,0         ; vbroadcastss  0x31cd(%rip),%ymm0        # 4944 <_sk_callback_hsw+0x230>
+  DB  196,226,125,24,5,204,49,0,0         ; vbroadcastss  0x31cc(%rip),%ymm0        # 4960 <_sk_callback_hsw+0x230>
   DB  197,220,88,192                      ; vaddps        %ymm0,%ymm4,%ymm0
   DB  196,227,125,8,224,1                 ; vroundps      $0x1,%ymm0,%ymm4
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
@@ -1472,11 +1479,11 @@ _sk_scale_u8_hsw LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,51                              ; jne           185d <_sk_scale_u8_hsw+0x43>
+  DB  117,51                              ; jne           187a <_sk_scale_u8_hsw+0x43>
   DB  197,122,126,0                       ; vmovq         (%rax),%xmm8
   DB  196,66,125,49,192                   ; vpmovzxbd     %xmm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,7,49,0,0           ; vbroadcastss  0x3107(%rip),%ymm9        # 4948 <_sk_callback_hsw+0x234>
+  DB  196,98,125,24,13,6,49,0,0           ; vbroadcastss  0x3106(%rip),%ymm9        # 4964 <_sk_callback_hsw+0x234>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -1494,9 +1501,9 @@ _sk_scale_u8_hsw LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           1865 <_sk_scale_u8_hsw+0x4b>
+  DB  117,234                             ; jne           1882 <_sk_scale_u8_hsw+0x4b>
   DB  196,65,249,110,193                  ; vmovq         %r9,%xmm8
-  DB  235,172                             ; jmp           182e <_sk_scale_u8_hsw+0x14>
+  DB  235,172                             ; jmp           184b <_sk_scale_u8_hsw+0x14>
 
 PUBLIC _sk_lerp_1_float_hsw
 _sk_lerp_1_float_hsw LABEL PROC
@@ -1520,11 +1527,11 @@ _sk_lerp_u8_hsw LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,71                              ; jne           1908 <_sk_lerp_u8_hsw+0x57>
+  DB  117,71                              ; jne           1925 <_sk_lerp_u8_hsw+0x57>
   DB  197,122,126,0                       ; vmovq         (%rax),%xmm8
   DB  196,66,125,49,192                   ; vpmovzxbd     %xmm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,116,48,0,0         ; vbroadcastss  0x3074(%rip),%ymm9        # 494c <_sk_callback_hsw+0x238>
+  DB  196,98,125,24,13,115,48,0,0         ; vbroadcastss  0x3073(%rip),%ymm9        # 4968 <_sk_callback_hsw+0x238>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
   DB  196,226,61,168,196                  ; vfmadd213ps   %ymm4,%ymm8,%ymm0
@@ -1546,32 +1553,32 @@ _sk_lerp_u8_hsw LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           1910 <_sk_lerp_u8_hsw+0x5f>
+  DB  117,234                             ; jne           192d <_sk_lerp_u8_hsw+0x5f>
   DB  196,65,249,110,193                  ; vmovq         %r9,%xmm8
-  DB  235,152                             ; jmp           18c5 <_sk_lerp_u8_hsw+0x14>
+  DB  235,152                             ; jmp           18e2 <_sk_lerp_u8_hsw+0x14>
 
 PUBLIC _sk_lerp_565_hsw
 _sk_lerp_565_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,169,0,0,0                    ; jne           19e4 <_sk_lerp_565_hsw+0xb7>
+  DB  15,133,169,0,0,0                    ; jne           1a01 <_sk_lerp_565_hsw+0xb7>
   DB  196,65,122,111,4,122                ; vmovdqu       (%r10,%rdi,2),%xmm8
   DB  196,66,125,51,192                   ; vpmovzxwd     %xmm8,%ymm8
-  DB  196,98,125,88,13,1,48,0,0           ; vpbroadcastd  0x3001(%rip),%ymm9        # 4950 <_sk_callback_hsw+0x23c>
+  DB  196,98,125,88,13,0,48,0,0           ; vpbroadcastd  0x3000(%rip),%ymm9        # 496c <_sk_callback_hsw+0x23c>
   DB  196,65,61,219,201                   ; vpand         %ymm9,%ymm8,%ymm9
   DB  196,65,124,91,201                   ; vcvtdq2ps     %ymm9,%ymm9
-  DB  196,98,125,24,21,242,47,0,0         ; vbroadcastss  0x2ff2(%rip),%ymm10        # 4954 <_sk_callback_hsw+0x240>
+  DB  196,98,125,24,21,241,47,0,0         ; vbroadcastss  0x2ff1(%rip),%ymm10        # 4970 <_sk_callback_hsw+0x240>
   DB  196,65,52,89,202                    ; vmulps        %ymm10,%ymm9,%ymm9
-  DB  196,98,125,88,21,232,47,0,0         ; vpbroadcastd  0x2fe8(%rip),%ymm10        # 4958 <_sk_callback_hsw+0x244>
+  DB  196,98,125,88,21,231,47,0,0         ; vpbroadcastd  0x2fe7(%rip),%ymm10        # 4974 <_sk_callback_hsw+0x244>
   DB  196,65,61,219,210                   ; vpand         %ymm10,%ymm8,%ymm10
   DB  196,65,124,91,210                   ; vcvtdq2ps     %ymm10,%ymm10
-  DB  196,98,125,24,29,217,47,0,0         ; vbroadcastss  0x2fd9(%rip),%ymm11        # 495c <_sk_callback_hsw+0x248>
+  DB  196,98,125,24,29,216,47,0,0         ; vbroadcastss  0x2fd8(%rip),%ymm11        # 4978 <_sk_callback_hsw+0x248>
   DB  196,65,44,89,211                    ; vmulps        %ymm11,%ymm10,%ymm10
-  DB  196,98,125,88,29,207,47,0,0         ; vpbroadcastd  0x2fcf(%rip),%ymm11        # 4960 <_sk_callback_hsw+0x24c>
+  DB  196,98,125,88,29,206,47,0,0         ; vpbroadcastd  0x2fce(%rip),%ymm11        # 497c <_sk_callback_hsw+0x24c>
   DB  196,65,61,219,195                   ; vpand         %ymm11,%ymm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,29,192,47,0,0         ; vbroadcastss  0x2fc0(%rip),%ymm11        # 4964 <_sk_callback_hsw+0x250>
+  DB  196,98,125,24,29,191,47,0,0         ; vbroadcastss  0x2fbf(%rip),%ymm11        # 4980 <_sk_callback_hsw+0x250>
   DB  196,65,60,89,195                    ; vmulps        %ymm11,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
   DB  196,226,53,168,196                  ; vfmadd213ps   %ymm4,%ymm9,%ymm0
@@ -1592,9 +1599,9 @@ _sk_lerp_565_hsw LABEL PROC
   DB  196,65,57,239,192                   ; vpxor         %xmm8,%xmm8,%xmm8
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,68,255,255,255               ; ja            1941 <_sk_lerp_565_hsw+0x14>
+  DB  15,135,68,255,255,255               ; ja            195e <_sk_lerp_565_hsw+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,76,0,0,0                  ; lea           0x4c(%rip),%r9        # 1a54 <_sk_lerp_565_hsw+0x127>
+  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 1a70 <_sk_lerp_565_hsw+0x126>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -1606,28 +1613,27 @@ _sk_lerp_565_hsw LABEL PROC
   DB  196,65,57,196,68,122,4,2            ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,2,1            ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,4,122,0               ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
-  DB  233,239,254,255,255                 ; jmpq          1941 <_sk_lerp_565_hsw+0x14>
-  DB  102,144                             ; xchg          %ax,%ax
-  DB  242,255                             ; repnz         (bad)
+  DB  233,239,254,255,255                 ; jmpq          195e <_sk_lerp_565_hsw+0x14>
+  DB  144                                 ; nop
+  DB  243,255                             ; repz          (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  234                                 ; (bad)
+  DB  235,255                             ; jmp           1a75 <_sk_lerp_565_hsw+0x12b>
   DB  255                                 ; (bad)
+  DB  255,227                             ; jmpq          *%rbx
   DB  255                                 ; (bad)
-  DB  255,226                             ; jmpq          *%rdx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
+  DB  219,255                             ; (bad)
   DB  255                                 ; (bad)
-  DB  218,255                             ; (bad)
+  DB  255,211                             ; callq         *%rbx
   DB  255                                 ; (bad)
-  DB  255,210                             ; callq         *%rdx
   DB  255                                 ; (bad)
-  DB  255                                 ; (bad)
-  DB  255,202                             ; dec           %edx
+  DB  255,203                             ; dec           %ebx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  189                                 ; .byte         0xbd
+  DB  190                                 ; .byte         0xbe
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
@@ -1639,23 +1645,23 @@ _sk_load_tables_hsw LABEL PROC
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  76,3,8                              ; add           (%rax),%r9
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,105                             ; jne           1aee <_sk_load_tables_hsw+0x7e>
+  DB  117,105                             ; jne           1b0a <_sk_load_tables_hsw+0x7e>
   DB  196,193,126,111,25                  ; vmovdqu       (%r9),%ymm3
-  DB  197,229,219,13,142,49,0,0           ; vpand         0x318e(%rip),%ymm3,%ymm1        # 4c20 <_sk_callback_hsw+0x50c>
+  DB  197,229,219,13,146,49,0,0           ; vpand         0x3192(%rip),%ymm3,%ymm1        # 4c40 <_sk_callback_hsw+0x510>
   DB  196,65,61,118,192                   ; vpcmpeqd      %ymm8,%ymm8,%ymm8
   DB  72,139,72,8                         ; mov           0x8(%rax),%rcx
   DB  76,139,72,16                        ; mov           0x10(%rax),%r9
   DB  197,237,118,210                     ; vpcmpeqd      %ymm2,%ymm2,%ymm2
   DB  196,226,109,146,4,137               ; vgatherdps    %ymm2,(%rcx,%ymm1,4),%ymm0
-  DB  196,226,101,0,21,142,49,0,0         ; vpshufb       0x318e(%rip),%ymm3,%ymm2        # 4c40 <_sk_callback_hsw+0x52c>
+  DB  196,226,101,0,21,146,49,0,0         ; vpshufb       0x3192(%rip),%ymm3,%ymm2        # 4c60 <_sk_callback_hsw+0x530>
   DB  196,65,53,118,201                   ; vpcmpeqd      %ymm9,%ymm9,%ymm9
   DB  196,194,53,146,12,145               ; vgatherdps    %ymm9,(%r9,%ymm2,4),%ymm1
   DB  72,139,64,24                        ; mov           0x18(%rax),%rax
-  DB  196,98,101,0,13,150,49,0,0          ; vpshufb       0x3196(%rip),%ymm3,%ymm9        # 4c60 <_sk_callback_hsw+0x54c>
+  DB  196,98,101,0,13,154,49,0,0          ; vpshufb       0x319a(%rip),%ymm3,%ymm9        # 4c80 <_sk_callback_hsw+0x550>
   DB  196,162,61,146,20,136               ; vgatherdps    %ymm8,(%rax,%ymm9,4),%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,134,46,0,0          ; vbroadcastss  0x2e86(%rip),%ymm8        # 4968 <_sk_callback_hsw+0x254>
+  DB  196,98,125,24,5,134,46,0,0          ; vbroadcastss  0x2e86(%rip),%ymm8        # 4984 <_sk_callback_hsw+0x254>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,137,193                          ; mov           %r8,%rcx
@@ -1668,7 +1674,7 @@ _sk_load_tables_hsw LABEL PROC
   DB  196,193,249,110,194                 ; vmovq         %r10,%xmm0
   DB  196,226,125,33,192                  ; vpmovsxbd     %xmm0,%ymm0
   DB  196,194,125,140,25                  ; vpmaskmovd    (%r9),%ymm0,%ymm3
-  DB  233,115,255,255,255                 ; jmpq          1a8a <_sk_load_tables_hsw+0x1a>
+  DB  233,115,255,255,255                 ; jmpq          1aa6 <_sk_load_tables_hsw+0x1a>
 
 PUBLIC _sk_load_tables_u16_be_hsw
 _sk_load_tables_u16_be_hsw LABEL PROC
@@ -1676,7 +1682,7 @@ _sk_load_tables_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,201,0,0,0                    ; jne           1bf6 <_sk_load_tables_u16_be_hsw+0xdf>
+  DB  15,133,201,0,0,0                    ; jne           1c12 <_sk_load_tables_u16_be_hsw+0xdf>
   DB  196,1,121,16,4,72                   ; vmovupd       (%r8,%r9,2),%xmm8
   DB  196,129,121,16,84,72,16             ; vmovupd       0x10(%r8,%r9,2),%xmm2
   DB  196,129,121,16,92,72,32             ; vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -1692,7 +1698,7 @@ _sk_load_tables_u16_be_hsw LABEL PROC
   DB  197,185,108,200                     ; vpunpcklqdq   %xmm0,%xmm8,%xmm1
   DB  197,185,109,208                     ; vpunpckhqdq   %xmm0,%xmm8,%xmm2
   DB  197,49,108,195                      ; vpunpcklqdq   %xmm3,%xmm9,%xmm8
-  DB  197,121,111,21,34,50,0,0            ; vmovdqa       0x3222(%rip),%xmm10        # 4da0 <_sk_callback_hsw+0x68c>
+  DB  197,121,111,21,38,50,0,0            ; vmovdqa       0x3226(%rip),%xmm10        # 4dc0 <_sk_callback_hsw+0x690>
   DB  196,193,113,219,194                 ; vpand         %xmm10,%xmm1,%xmm0
   DB  196,226,125,51,200                  ; vpmovzxwd     %xmm0,%ymm1
   DB  196,65,37,118,219                   ; vpcmpeqd      %ymm11,%ymm11,%ymm11
@@ -1714,36 +1720,36 @@ _sk_load_tables_u16_be_hsw LABEL PROC
   DB  197,185,235,219                     ; vpor          %xmm3,%xmm8,%xmm3
   DB  196,226,125,51,219                  ; vpmovzxwd     %xmm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,127,45,0,0          ; vbroadcastss  0x2d7f(%rip),%ymm8        # 496c <_sk_callback_hsw+0x258>
+  DB  196,98,125,24,5,127,45,0,0          ; vbroadcastss  0x2d7f(%rip),%ymm8        # 4988 <_sk_callback_hsw+0x258>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
   DB  196,1,123,16,4,72                   ; vmovsd        (%r8,%r9,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            1c5c <_sk_load_tables_u16_be_hsw+0x145>
+  DB  116,85                              ; je            1c78 <_sk_load_tables_u16_be_hsw+0x145>
   DB  196,1,57,22,68,72,8                 ; vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            1c5c <_sk_load_tables_u16_be_hsw+0x145>
+  DB  114,72                              ; jb            1c78 <_sk_load_tables_u16_be_hsw+0x145>
   DB  196,129,123,16,84,72,16             ; vmovsd        0x10(%r8,%r9,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            1c69 <_sk_load_tables_u16_be_hsw+0x152>
+  DB  116,72                              ; je            1c85 <_sk_load_tables_u16_be_hsw+0x152>
   DB  196,129,105,22,84,72,24             ; vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            1c69 <_sk_load_tables_u16_be_hsw+0x152>
+  DB  114,59                              ; jb            1c85 <_sk_load_tables_u16_be_hsw+0x152>
   DB  196,129,123,16,92,72,32             ; vmovsd        0x20(%r8,%r9,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,9,255,255,255                ; je            1b48 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  15,132,9,255,255,255                ; je            1b64 <_sk_load_tables_u16_be_hsw+0x31>
   DB  196,129,97,22,92,72,40              ; vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,248,254,255,255              ; jb            1b48 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  15,130,248,254,255,255              ; jb            1b64 <_sk_load_tables_u16_be_hsw+0x31>
   DB  196,1,122,126,76,72,48              ; vmovq         0x30(%r8,%r9,2),%xmm9
-  DB  233,236,254,255,255                 ; jmpq          1b48 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  233,236,254,255,255                 ; jmpq          1b64 <_sk_load_tables_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,223,254,255,255                 ; jmpq          1b48 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  233,223,254,255,255                 ; jmpq          1b64 <_sk_load_tables_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,214,254,255,255                 ; jmpq          1b48 <_sk_load_tables_u16_be_hsw+0x31>
+  DB  233,214,254,255,255                 ; jmpq          1b64 <_sk_load_tables_u16_be_hsw+0x31>
 
 PUBLIC _sk_load_tables_rgb_u16_be_hsw
 _sk_load_tables_rgb_u16_be_hsw LABEL PROC
@@ -1751,7 +1757,7 @@ _sk_load_tables_rgb_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,127                       ; lea           (%rdi,%rdi,2),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,193,0,0,0                    ; jne           1d45 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
+  DB  15,133,193,0,0,0                    ; jne           1d61 <_sk_load_tables_rgb_u16_be_hsw+0xd3>
   DB  196,129,122,111,4,72                ; vmovdqu       (%r8,%r9,2),%xmm0
   DB  196,129,122,111,84,72,12            ; vmovdqu       0xc(%r8,%r9,2),%xmm2
   DB  196,129,122,111,76,72,24            ; vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -1772,7 +1778,7 @@ _sk_load_tables_rgb_u16_be_hsw LABEL PROC
   DB  197,185,108,218                     ; vpunpcklqdq   %xmm2,%xmm8,%xmm3
   DB  197,185,109,210                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm2
   DB  197,121,108,193                     ; vpunpcklqdq   %xmm1,%xmm0,%xmm8
-  DB  197,121,111,13,194,48,0,0           ; vmovdqa       0x30c2(%rip),%xmm9        # 4db0 <_sk_callback_hsw+0x69c>
+  DB  197,121,111,13,198,48,0,0           ; vmovdqa       0x30c6(%rip),%xmm9        # 4dd0 <_sk_callback_hsw+0x6a0>
   DB  196,193,97,219,193                  ; vpand         %xmm9,%xmm3,%xmm0
   DB  196,226,125,51,200                  ; vpmovzxwd     %xmm0,%ymm1
   DB  197,229,118,219                     ; vpcmpeqd      %ymm3,%ymm3,%ymm3
@@ -1789,41 +1795,41 @@ _sk_load_tables_rgb_u16_be_hsw LABEL PROC
   DB  196,98,125,51,194                   ; vpmovzxwd     %xmm2,%ymm8
   DB  196,162,101,146,20,128              ; vgatherdps    %ymm3,(%rax,%ymm8,4),%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,45,44,0,0         ; vbroadcastss  0x2c2d(%rip),%ymm3        # 4970 <_sk_callback_hsw+0x25c>
+  DB  196,226,125,24,29,45,44,0,0         ; vbroadcastss  0x2c2d(%rip),%ymm3        # 498c <_sk_callback_hsw+0x25c>
   DB  255,224                             ; jmpq          *%rax
   DB  196,129,121,110,4,72                ; vmovd         (%r8,%r9,2),%xmm0
   DB  196,129,121,196,68,72,4,2           ; vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           1d5e <_sk_load_tables_rgb_u16_be_hsw+0xec>
-  DB  233,90,255,255,255                  ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  117,5                               ; jne           1d7a <_sk_load_tables_rgb_u16_be_hsw+0xec>
+  DB  233,90,255,255,255                  ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   DB  196,129,121,110,76,72,6             ; vmovd         0x6(%r8,%r9,2),%xmm1
   DB  196,1,113,196,68,72,10,2            ; vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            1d8d <_sk_load_tables_rgb_u16_be_hsw+0x11b>
+  DB  114,26                              ; jb            1da9 <_sk_load_tables_rgb_u16_be_hsw+0x11b>
   DB  196,129,121,110,76,72,12            ; vmovd         0xc(%r8,%r9,2),%xmm1
   DB  196,129,113,196,84,72,16,2          ; vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           1d92 <_sk_load_tables_rgb_u16_be_hsw+0x120>
-  DB  233,43,255,255,255                  ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  DB  233,38,255,255,255                  ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           1dae <_sk_load_tables_rgb_u16_be_hsw+0x120>
+  DB  233,43,255,255,255                  ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,38,255,255,255                  ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   DB  196,129,121,110,76,72,18            ; vmovd         0x12(%r8,%r9,2),%xmm1
   DB  196,1,113,196,76,72,22,2            ; vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            1dc1 <_sk_load_tables_rgb_u16_be_hsw+0x14f>
+  DB  114,26                              ; jb            1ddd <_sk_load_tables_rgb_u16_be_hsw+0x14f>
   DB  196,129,121,110,76,72,24            ; vmovd         0x18(%r8,%r9,2),%xmm1
   DB  196,129,113,196,76,72,28,2          ; vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           1dc6 <_sk_load_tables_rgb_u16_be_hsw+0x154>
-  DB  233,247,254,255,255                 ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  DB  233,242,254,255,255                 ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           1de2 <_sk_load_tables_rgb_u16_be_hsw+0x154>
+  DB  233,247,254,255,255                 ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,242,254,255,255                 ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
   DB  196,129,121,110,92,72,30            ; vmovd         0x1e(%r8,%r9,2),%xmm3
   DB  196,1,97,196,92,72,34,2             ; vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            1def <_sk_load_tables_rgb_u16_be_hsw+0x17d>
+  DB  114,20                              ; jb            1e0b <_sk_load_tables_rgb_u16_be_hsw+0x17d>
   DB  196,129,121,110,92,72,36            ; vmovd         0x24(%r8,%r9,2),%xmm3
   DB  196,129,97,196,92,72,40,2           ; vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  DB  233,201,254,255,255                 ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
-  DB  233,196,254,255,255                 ; jmpq          1cb8 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,201,254,255,255                 ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
+  DB  233,196,254,255,255                 ; jmpq          1cd4 <_sk_load_tables_rgb_u16_be_hsw+0x46>
 
 PUBLIC _sk_byte_tables_hsw
 _sk_byte_tables_hsw LABEL PROC
@@ -1834,7 +1840,7 @@ _sk_byte_tables_hsw LABEL PROC
   DB  65,84                               ; push          %r12
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,107,43,0,0          ; vbroadcastss  0x2b6b(%rip),%ymm8        # 4974 <_sk_callback_hsw+0x260>
+  DB  196,98,125,24,5,107,43,0,0          ; vbroadcastss  0x2b6b(%rip),%ymm8        # 4990 <_sk_callback_hsw+0x260>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,195,249,22,192,1                ; vpextrq       $0x1,%xmm0,%r8
@@ -1871,7 +1877,7 @@ _sk_byte_tables_hsw LABEL PROC
   DB  196,227,121,32,197,7                ; vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,188,42,0,0         ; vbroadcastss  0x2abc(%rip),%ymm9        # 4978 <_sk_callback_hsw+0x264>
+  DB  196,98,125,24,13,188,42,0,0         ; vbroadcastss  0x2abc(%rip),%ymm9        # 4994 <_sk_callback_hsw+0x264>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -2030,7 +2036,7 @@ _sk_byte_tables_rgb_hsw LABEL PROC
   DB  196,227,121,32,197,7                ; vpinsrb       $0x7,%ebp,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,245,39,0,0         ; vbroadcastss  0x27f5(%rip),%ymm9        # 497c <_sk_callback_hsw+0x268>
+  DB  196,98,125,24,13,245,39,0,0         ; vbroadcastss  0x27f5(%rip),%ymm9        # 4998 <_sk_callback_hsw+0x268>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -2183,33 +2189,33 @@ _sk_parametric_r_hsw LABEL PROC
   DB  196,66,125,168,211                  ; vfmadd213ps   %ymm11,%ymm0,%ymm10
   DB  196,226,125,24,0                    ; vbroadcastss  (%rax),%ymm0
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,168,37,0,0         ; vbroadcastss  0x25a8(%rip),%ymm12        # 4980 <_sk_callback_hsw+0x26c>
-  DB  196,98,125,24,45,163,37,0,0         ; vbroadcastss  0x25a3(%rip),%ymm13        # 4984 <_sk_callback_hsw+0x270>
+  DB  196,98,125,24,37,168,37,0,0         ; vbroadcastss  0x25a8(%rip),%ymm12        # 499c <_sk_callback_hsw+0x26c>
+  DB  196,98,125,24,45,163,37,0,0         ; vbroadcastss  0x25a3(%rip),%ymm13        # 49a0 <_sk_callback_hsw+0x270>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,153,37,0,0         ; vbroadcastss  0x2599(%rip),%ymm13        # 4988 <_sk_callback_hsw+0x274>
+  DB  196,98,125,24,45,153,37,0,0         ; vbroadcastss  0x2599(%rip),%ymm13        # 49a4 <_sk_callback_hsw+0x274>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,143,37,0,0         ; vbroadcastss  0x258f(%rip),%ymm13        # 498c <_sk_callback_hsw+0x278>
+  DB  196,98,125,24,45,143,37,0,0         ; vbroadcastss  0x258f(%rip),%ymm13        # 49a8 <_sk_callback_hsw+0x278>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,133,37,0,0         ; vbroadcastss  0x2585(%rip),%ymm11        # 4990 <_sk_callback_hsw+0x27c>
+  DB  196,98,125,24,29,133,37,0,0         ; vbroadcastss  0x2585(%rip),%ymm11        # 49ac <_sk_callback_hsw+0x27c>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,123,37,0,0         ; vbroadcastss  0x257b(%rip),%ymm12        # 4994 <_sk_callback_hsw+0x280>
+  DB  196,98,125,24,37,123,37,0,0         ; vbroadcastss  0x257b(%rip),%ymm12        # 49b0 <_sk_callback_hsw+0x280>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,113,37,0,0         ; vbroadcastss  0x2571(%rip),%ymm12        # 4998 <_sk_callback_hsw+0x284>
+  DB  196,98,125,24,37,113,37,0,0         ; vbroadcastss  0x2571(%rip),%ymm12        # 49b4 <_sk_callback_hsw+0x284>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  196,99,125,8,208,1                  ; vroundps      $0x1,%ymm0,%ymm10
   DB  196,65,124,92,210                   ; vsubps        %ymm10,%ymm0,%ymm10
-  DB  196,98,125,24,29,82,37,0,0          ; vbroadcastss  0x2552(%rip),%ymm11        # 499c <_sk_callback_hsw+0x288>
+  DB  196,98,125,24,29,82,37,0,0          ; vbroadcastss  0x2552(%rip),%ymm11        # 49b8 <_sk_callback_hsw+0x288>
   DB  196,193,124,88,195                  ; vaddps        %ymm11,%ymm0,%ymm0
-  DB  196,98,125,24,29,72,37,0,0          ; vbroadcastss  0x2548(%rip),%ymm11        # 49a0 <_sk_callback_hsw+0x28c>
+  DB  196,98,125,24,29,72,37,0,0          ; vbroadcastss  0x2548(%rip),%ymm11        # 49bc <_sk_callback_hsw+0x28c>
   DB  196,98,45,172,216                   ; vfnmadd213ps  %ymm0,%ymm10,%ymm11
-  DB  196,226,125,24,5,62,37,0,0          ; vbroadcastss  0x253e(%rip),%ymm0        # 49a4 <_sk_callback_hsw+0x290>
+  DB  196,226,125,24,5,62,37,0,0          ; vbroadcastss  0x253e(%rip),%ymm0        # 49c0 <_sk_callback_hsw+0x290>
   DB  196,193,124,92,194                  ; vsubps        %ymm10,%ymm0,%ymm0
-  DB  196,98,125,24,21,52,37,0,0          ; vbroadcastss  0x2534(%rip),%ymm10        # 49a8 <_sk_callback_hsw+0x294>
+  DB  196,98,125,24,21,52,37,0,0          ; vbroadcastss  0x2534(%rip),%ymm10        # 49c4 <_sk_callback_hsw+0x294>
   DB  197,172,94,192                      ; vdivps        %ymm0,%ymm10,%ymm0
   DB  197,164,88,192                      ; vaddps        %ymm0,%ymm11,%ymm0
-  DB  196,98,125,24,21,39,37,0,0          ; vbroadcastss  0x2527(%rip),%ymm10        # 49ac <_sk_callback_hsw+0x298>
+  DB  196,98,125,24,21,39,37,0,0          ; vbroadcastss  0x2527(%rip),%ymm10        # 49c8 <_sk_callback_hsw+0x298>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2217,7 +2223,7 @@ _sk_parametric_r_hsw LABEL PROC
   DB  196,195,125,74,193,128              ; vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,124,95,192                  ; vmaxps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,254,36,0,0          ; vbroadcastss  0x24fe(%rip),%ymm8        # 49b0 <_sk_callback_hsw+0x29c>
+  DB  196,98,125,24,5,254,36,0,0          ; vbroadcastss  0x24fe(%rip),%ymm8        # 49cc <_sk_callback_hsw+0x29c>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2235,33 +2241,33 @@ _sk_parametric_g_hsw LABEL PROC
   DB  196,66,117,168,211                  ; vfmadd213ps   %ymm11,%ymm1,%ymm10
   DB  196,226,125,24,8                    ; vbroadcastss  (%rax),%ymm1
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,182,36,0,0         ; vbroadcastss  0x24b6(%rip),%ymm12        # 49b4 <_sk_callback_hsw+0x2a0>
-  DB  196,98,125,24,45,177,36,0,0         ; vbroadcastss  0x24b1(%rip),%ymm13        # 49b8 <_sk_callback_hsw+0x2a4>
+  DB  196,98,125,24,37,182,36,0,0         ; vbroadcastss  0x24b6(%rip),%ymm12        # 49d0 <_sk_callback_hsw+0x2a0>
+  DB  196,98,125,24,45,177,36,0,0         ; vbroadcastss  0x24b1(%rip),%ymm13        # 49d4 <_sk_callback_hsw+0x2a4>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,167,36,0,0         ; vbroadcastss  0x24a7(%rip),%ymm13        # 49bc <_sk_callback_hsw+0x2a8>
+  DB  196,98,125,24,45,167,36,0,0         ; vbroadcastss  0x24a7(%rip),%ymm13        # 49d8 <_sk_callback_hsw+0x2a8>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,157,36,0,0         ; vbroadcastss  0x249d(%rip),%ymm13        # 49c0 <_sk_callback_hsw+0x2ac>
+  DB  196,98,125,24,45,157,36,0,0         ; vbroadcastss  0x249d(%rip),%ymm13        # 49dc <_sk_callback_hsw+0x2ac>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,147,36,0,0         ; vbroadcastss  0x2493(%rip),%ymm11        # 49c4 <_sk_callback_hsw+0x2b0>
+  DB  196,98,125,24,29,147,36,0,0         ; vbroadcastss  0x2493(%rip),%ymm11        # 49e0 <_sk_callback_hsw+0x2b0>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,137,36,0,0         ; vbroadcastss  0x2489(%rip),%ymm12        # 49c8 <_sk_callback_hsw+0x2b4>
+  DB  196,98,125,24,37,137,36,0,0         ; vbroadcastss  0x2489(%rip),%ymm12        # 49e4 <_sk_callback_hsw+0x2b4>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,127,36,0,0         ; vbroadcastss  0x247f(%rip),%ymm12        # 49cc <_sk_callback_hsw+0x2b8>
+  DB  196,98,125,24,37,127,36,0,0         ; vbroadcastss  0x247f(%rip),%ymm12        # 49e8 <_sk_callback_hsw+0x2b8>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  196,99,125,8,209,1                  ; vroundps      $0x1,%ymm1,%ymm10
   DB  196,65,116,92,210                   ; vsubps        %ymm10,%ymm1,%ymm10
-  DB  196,98,125,24,29,96,36,0,0          ; vbroadcastss  0x2460(%rip),%ymm11        # 49d0 <_sk_callback_hsw+0x2bc>
+  DB  196,98,125,24,29,96,36,0,0          ; vbroadcastss  0x2460(%rip),%ymm11        # 49ec <_sk_callback_hsw+0x2bc>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,86,36,0,0          ; vbroadcastss  0x2456(%rip),%ymm11        # 49d4 <_sk_callback_hsw+0x2c0>
+  DB  196,98,125,24,29,86,36,0,0          ; vbroadcastss  0x2456(%rip),%ymm11        # 49f0 <_sk_callback_hsw+0x2c0>
   DB  196,98,45,172,217                   ; vfnmadd213ps  %ymm1,%ymm10,%ymm11
-  DB  196,226,125,24,13,76,36,0,0         ; vbroadcastss  0x244c(%rip),%ymm1        # 49d8 <_sk_callback_hsw+0x2c4>
+  DB  196,226,125,24,13,76,36,0,0         ; vbroadcastss  0x244c(%rip),%ymm1        # 49f4 <_sk_callback_hsw+0x2c4>
   DB  196,193,116,92,202                  ; vsubps        %ymm10,%ymm1,%ymm1
-  DB  196,98,125,24,21,66,36,0,0          ; vbroadcastss  0x2442(%rip),%ymm10        # 49dc <_sk_callback_hsw+0x2c8>
+  DB  196,98,125,24,21,66,36,0,0          ; vbroadcastss  0x2442(%rip),%ymm10        # 49f8 <_sk_callback_hsw+0x2c8>
   DB  197,172,94,201                      ; vdivps        %ymm1,%ymm10,%ymm1
   DB  197,164,88,201                      ; vaddps        %ymm1,%ymm11,%ymm1
-  DB  196,98,125,24,21,53,36,0,0          ; vbroadcastss  0x2435(%rip),%ymm10        # 49e0 <_sk_callback_hsw+0x2cc>
+  DB  196,98,125,24,21,53,36,0,0          ; vbroadcastss  0x2435(%rip),%ymm10        # 49fc <_sk_callback_hsw+0x2cc>
   DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2269,7 +2275,7 @@ _sk_parametric_g_hsw LABEL PROC
   DB  196,195,117,74,201,128              ; vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,116,95,200                  ; vmaxps        %ymm8,%ymm1,%ymm1
-  DB  196,98,125,24,5,12,36,0,0           ; vbroadcastss  0x240c(%rip),%ymm8        # 49e4 <_sk_callback_hsw+0x2d0>
+  DB  196,98,125,24,5,12,36,0,0           ; vbroadcastss  0x240c(%rip),%ymm8        # 4a00 <_sk_callback_hsw+0x2d0>
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2287,33 +2293,33 @@ _sk_parametric_b_hsw LABEL PROC
   DB  196,66,109,168,211                  ; vfmadd213ps   %ymm11,%ymm2,%ymm10
   DB  196,226,125,24,16                   ; vbroadcastss  (%rax),%ymm2
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,196,35,0,0         ; vbroadcastss  0x23c4(%rip),%ymm12        # 49e8 <_sk_callback_hsw+0x2d4>
-  DB  196,98,125,24,45,191,35,0,0         ; vbroadcastss  0x23bf(%rip),%ymm13        # 49ec <_sk_callback_hsw+0x2d8>
+  DB  196,98,125,24,37,196,35,0,0         ; vbroadcastss  0x23c4(%rip),%ymm12        # 4a04 <_sk_callback_hsw+0x2d4>
+  DB  196,98,125,24,45,191,35,0,0         ; vbroadcastss  0x23bf(%rip),%ymm13        # 4a08 <_sk_callback_hsw+0x2d8>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,181,35,0,0         ; vbroadcastss  0x23b5(%rip),%ymm13        # 49f0 <_sk_callback_hsw+0x2dc>
+  DB  196,98,125,24,45,181,35,0,0         ; vbroadcastss  0x23b5(%rip),%ymm13        # 4a0c <_sk_callback_hsw+0x2dc>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,171,35,0,0         ; vbroadcastss  0x23ab(%rip),%ymm13        # 49f4 <_sk_callback_hsw+0x2e0>
+  DB  196,98,125,24,45,171,35,0,0         ; vbroadcastss  0x23ab(%rip),%ymm13        # 4a10 <_sk_callback_hsw+0x2e0>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,161,35,0,0         ; vbroadcastss  0x23a1(%rip),%ymm11        # 49f8 <_sk_callback_hsw+0x2e4>
+  DB  196,98,125,24,29,161,35,0,0         ; vbroadcastss  0x23a1(%rip),%ymm11        # 4a14 <_sk_callback_hsw+0x2e4>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,151,35,0,0         ; vbroadcastss  0x2397(%rip),%ymm12        # 49fc <_sk_callback_hsw+0x2e8>
+  DB  196,98,125,24,37,151,35,0,0         ; vbroadcastss  0x2397(%rip),%ymm12        # 4a18 <_sk_callback_hsw+0x2e8>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,141,35,0,0         ; vbroadcastss  0x238d(%rip),%ymm12        # 4a00 <_sk_callback_hsw+0x2ec>
+  DB  196,98,125,24,37,141,35,0,0         ; vbroadcastss  0x238d(%rip),%ymm12        # 4a1c <_sk_callback_hsw+0x2ec>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  196,99,125,8,210,1                  ; vroundps      $0x1,%ymm2,%ymm10
   DB  196,65,108,92,210                   ; vsubps        %ymm10,%ymm2,%ymm10
-  DB  196,98,125,24,29,110,35,0,0         ; vbroadcastss  0x236e(%rip),%ymm11        # 4a04 <_sk_callback_hsw+0x2f0>
+  DB  196,98,125,24,29,110,35,0,0         ; vbroadcastss  0x236e(%rip),%ymm11        # 4a20 <_sk_callback_hsw+0x2f0>
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
-  DB  196,98,125,24,29,100,35,0,0         ; vbroadcastss  0x2364(%rip),%ymm11        # 4a08 <_sk_callback_hsw+0x2f4>
+  DB  196,98,125,24,29,100,35,0,0         ; vbroadcastss  0x2364(%rip),%ymm11        # 4a24 <_sk_callback_hsw+0x2f4>
   DB  196,98,45,172,218                   ; vfnmadd213ps  %ymm2,%ymm10,%ymm11
-  DB  196,226,125,24,21,90,35,0,0         ; vbroadcastss  0x235a(%rip),%ymm2        # 4a0c <_sk_callback_hsw+0x2f8>
+  DB  196,226,125,24,21,90,35,0,0         ; vbroadcastss  0x235a(%rip),%ymm2        # 4a28 <_sk_callback_hsw+0x2f8>
   DB  196,193,108,92,210                  ; vsubps        %ymm10,%ymm2,%ymm2
-  DB  196,98,125,24,21,80,35,0,0          ; vbroadcastss  0x2350(%rip),%ymm10        # 4a10 <_sk_callback_hsw+0x2fc>
+  DB  196,98,125,24,21,80,35,0,0          ; vbroadcastss  0x2350(%rip),%ymm10        # 4a2c <_sk_callback_hsw+0x2fc>
   DB  197,172,94,210                      ; vdivps        %ymm2,%ymm10,%ymm2
   DB  197,164,88,210                      ; vaddps        %ymm2,%ymm11,%ymm2
-  DB  196,98,125,24,21,67,35,0,0          ; vbroadcastss  0x2343(%rip),%ymm10        # 4a14 <_sk_callback_hsw+0x300>
+  DB  196,98,125,24,21,67,35,0,0          ; vbroadcastss  0x2343(%rip),%ymm10        # 4a30 <_sk_callback_hsw+0x300>
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  197,253,91,210                      ; vcvtps2dq     %ymm2,%ymm2
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2321,7 +2327,7 @@ _sk_parametric_b_hsw LABEL PROC
   DB  196,195,109,74,209,128              ; vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,108,95,208                  ; vmaxps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,26,35,0,0           ; vbroadcastss  0x231a(%rip),%ymm8        # 4a18 <_sk_callback_hsw+0x304>
+  DB  196,98,125,24,5,26,35,0,0           ; vbroadcastss  0x231a(%rip),%ymm8        # 4a34 <_sk_callback_hsw+0x304>
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2339,33 +2345,33 @@ _sk_parametric_a_hsw LABEL PROC
   DB  196,66,101,168,211                  ; vfmadd213ps   %ymm11,%ymm3,%ymm10
   DB  196,226,125,24,24                   ; vbroadcastss  (%rax),%ymm3
   DB  196,65,124,91,218                   ; vcvtdq2ps     %ymm10,%ymm11
-  DB  196,98,125,24,37,210,34,0,0         ; vbroadcastss  0x22d2(%rip),%ymm12        # 4a1c <_sk_callback_hsw+0x308>
-  DB  196,98,125,24,45,205,34,0,0         ; vbroadcastss  0x22cd(%rip),%ymm13        # 4a20 <_sk_callback_hsw+0x30c>
+  DB  196,98,125,24,37,210,34,0,0         ; vbroadcastss  0x22d2(%rip),%ymm12        # 4a38 <_sk_callback_hsw+0x308>
+  DB  196,98,125,24,45,205,34,0,0         ; vbroadcastss  0x22cd(%rip),%ymm13        # 4a3c <_sk_callback_hsw+0x30c>
   DB  196,65,44,84,213                    ; vandps        %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,195,34,0,0         ; vbroadcastss  0x22c3(%rip),%ymm13        # 4a24 <_sk_callback_hsw+0x310>
+  DB  196,98,125,24,45,195,34,0,0         ; vbroadcastss  0x22c3(%rip),%ymm13        # 4a40 <_sk_callback_hsw+0x310>
   DB  196,65,44,86,213                    ; vorps         %ymm13,%ymm10,%ymm10
-  DB  196,98,125,24,45,185,34,0,0         ; vbroadcastss  0x22b9(%rip),%ymm13        # 4a28 <_sk_callback_hsw+0x314>
+  DB  196,98,125,24,45,185,34,0,0         ; vbroadcastss  0x22b9(%rip),%ymm13        # 4a44 <_sk_callback_hsw+0x314>
   DB  196,66,37,184,236                   ; vfmadd231ps   %ymm12,%ymm11,%ymm13
-  DB  196,98,125,24,29,175,34,0,0         ; vbroadcastss  0x22af(%rip),%ymm11        # 4a2c <_sk_callback_hsw+0x318>
+  DB  196,98,125,24,29,175,34,0,0         ; vbroadcastss  0x22af(%rip),%ymm11        # 4a48 <_sk_callback_hsw+0x318>
   DB  196,66,45,172,221                   ; vfnmadd213ps  %ymm13,%ymm10,%ymm11
-  DB  196,98,125,24,37,165,34,0,0         ; vbroadcastss  0x22a5(%rip),%ymm12        # 4a30 <_sk_callback_hsw+0x31c>
+  DB  196,98,125,24,37,165,34,0,0         ; vbroadcastss  0x22a5(%rip),%ymm12        # 4a4c <_sk_callback_hsw+0x31c>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,155,34,0,0         ; vbroadcastss  0x229b(%rip),%ymm12        # 4a34 <_sk_callback_hsw+0x320>
+  DB  196,98,125,24,37,155,34,0,0         ; vbroadcastss  0x229b(%rip),%ymm12        # 4a50 <_sk_callback_hsw+0x320>
   DB  196,65,28,94,210                    ; vdivps        %ymm10,%ymm12,%ymm10
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
   DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
   DB  196,99,125,8,211,1                  ; vroundps      $0x1,%ymm3,%ymm10
   DB  196,65,100,92,210                   ; vsubps        %ymm10,%ymm3,%ymm10
-  DB  196,98,125,24,29,124,34,0,0         ; vbroadcastss  0x227c(%rip),%ymm11        # 4a38 <_sk_callback_hsw+0x324>
+  DB  196,98,125,24,29,124,34,0,0         ; vbroadcastss  0x227c(%rip),%ymm11        # 4a54 <_sk_callback_hsw+0x324>
   DB  196,193,100,88,219                  ; vaddps        %ymm11,%ymm3,%ymm3
-  DB  196,98,125,24,29,114,34,0,0         ; vbroadcastss  0x2272(%rip),%ymm11        # 4a3c <_sk_callback_hsw+0x328>
+  DB  196,98,125,24,29,114,34,0,0         ; vbroadcastss  0x2272(%rip),%ymm11        # 4a58 <_sk_callback_hsw+0x328>
   DB  196,98,45,172,219                   ; vfnmadd213ps  %ymm3,%ymm10,%ymm11
-  DB  196,226,125,24,29,104,34,0,0        ; vbroadcastss  0x2268(%rip),%ymm3        # 4a40 <_sk_callback_hsw+0x32c>
+  DB  196,226,125,24,29,104,34,0,0        ; vbroadcastss  0x2268(%rip),%ymm3        # 4a5c <_sk_callback_hsw+0x32c>
   DB  196,193,100,92,218                  ; vsubps        %ymm10,%ymm3,%ymm3
-  DB  196,98,125,24,21,94,34,0,0          ; vbroadcastss  0x225e(%rip),%ymm10        # 4a44 <_sk_callback_hsw+0x330>
+  DB  196,98,125,24,21,94,34,0,0          ; vbroadcastss  0x225e(%rip),%ymm10        # 4a60 <_sk_callback_hsw+0x330>
   DB  197,172,94,219                      ; vdivps        %ymm3,%ymm10,%ymm3
   DB  197,164,88,219                      ; vaddps        %ymm3,%ymm11,%ymm3
-  DB  196,98,125,24,21,81,34,0,0          ; vbroadcastss  0x2251(%rip),%ymm10        # 4a48 <_sk_callback_hsw+0x334>
+  DB  196,98,125,24,21,81,34,0,0          ; vbroadcastss  0x2251(%rip),%ymm10        # 4a64 <_sk_callback_hsw+0x334>
   DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
   DB  197,253,91,219                      ; vcvtps2dq     %ymm3,%ymm3
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -2373,33 +2379,33 @@ _sk_parametric_a_hsw LABEL PROC
   DB  196,195,101,74,217,128              ; vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,100,95,216                  ; vmaxps        %ymm8,%ymm3,%ymm3
-  DB  196,98,125,24,5,40,34,0,0           ; vbroadcastss  0x2228(%rip),%ymm8        # 4a4c <_sk_callback_hsw+0x338>
+  DB  196,98,125,24,5,40,34,0,0           ; vbroadcastss  0x2228(%rip),%ymm8        # 4a68 <_sk_callback_hsw+0x338>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_lab_to_xyz_hsw
 _sk_lab_to_xyz_hsw LABEL PROC
-  DB  196,98,125,24,5,26,34,0,0           ; vbroadcastss  0x221a(%rip),%ymm8        # 4a50 <_sk_callback_hsw+0x33c>
-  DB  196,98,125,24,13,21,34,0,0          ; vbroadcastss  0x2215(%rip),%ymm9        # 4a54 <_sk_callback_hsw+0x340>
-  DB  196,98,125,24,21,16,34,0,0          ; vbroadcastss  0x2210(%rip),%ymm10        # 4a58 <_sk_callback_hsw+0x344>
+  DB  196,98,125,24,5,26,34,0,0           ; vbroadcastss  0x221a(%rip),%ymm8        # 4a6c <_sk_callback_hsw+0x33c>
+  DB  196,98,125,24,13,21,34,0,0          ; vbroadcastss  0x2215(%rip),%ymm9        # 4a70 <_sk_callback_hsw+0x340>
+  DB  196,98,125,24,21,16,34,0,0          ; vbroadcastss  0x2210(%rip),%ymm10        # 4a74 <_sk_callback_hsw+0x344>
   DB  196,194,53,168,202                  ; vfmadd213ps   %ymm10,%ymm9,%ymm1
   DB  196,194,53,168,210                  ; vfmadd213ps   %ymm10,%ymm9,%ymm2
-  DB  196,98,125,24,13,1,34,0,0           ; vbroadcastss  0x2201(%rip),%ymm9        # 4a5c <_sk_callback_hsw+0x348>
+  DB  196,98,125,24,13,1,34,0,0           ; vbroadcastss  0x2201(%rip),%ymm9        # 4a78 <_sk_callback_hsw+0x348>
   DB  196,66,125,184,200                  ; vfmadd231ps   %ymm8,%ymm0,%ymm9
-  DB  196,226,125,24,5,247,33,0,0         ; vbroadcastss  0x21f7(%rip),%ymm0        # 4a60 <_sk_callback_hsw+0x34c>
+  DB  196,226,125,24,5,247,33,0,0         ; vbroadcastss  0x21f7(%rip),%ymm0        # 4a7c <_sk_callback_hsw+0x34c>
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
-  DB  196,98,125,24,5,238,33,0,0          ; vbroadcastss  0x21ee(%rip),%ymm8        # 4a64 <_sk_callback_hsw+0x350>
+  DB  196,98,125,24,5,238,33,0,0          ; vbroadcastss  0x21ee(%rip),%ymm8        # 4a80 <_sk_callback_hsw+0x350>
   DB  196,98,117,168,192                  ; vfmadd213ps   %ymm0,%ymm1,%ymm8
-  DB  196,98,125,24,13,228,33,0,0         ; vbroadcastss  0x21e4(%rip),%ymm9        # 4a68 <_sk_callback_hsw+0x354>
+  DB  196,98,125,24,13,228,33,0,0         ; vbroadcastss  0x21e4(%rip),%ymm9        # 4a84 <_sk_callback_hsw+0x354>
   DB  196,98,109,172,200                  ; vfnmadd213ps  %ymm0,%ymm2,%ymm9
   DB  196,193,60,89,200                   ; vmulps        %ymm8,%ymm8,%ymm1
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
-  DB  196,226,125,24,21,209,33,0,0        ; vbroadcastss  0x21d1(%rip),%ymm2        # 4a6c <_sk_callback_hsw+0x358>
+  DB  196,226,125,24,21,209,33,0,0        ; vbroadcastss  0x21d1(%rip),%ymm2        # 4a88 <_sk_callback_hsw+0x358>
   DB  197,108,194,209,1                   ; vcmpltps      %ymm1,%ymm2,%ymm10
-  DB  196,98,125,24,29,199,33,0,0         ; vbroadcastss  0x21c7(%rip),%ymm11        # 4a70 <_sk_callback_hsw+0x35c>
+  DB  196,98,125,24,29,199,33,0,0         ; vbroadcastss  0x21c7(%rip),%ymm11        # 4a8c <_sk_callback_hsw+0x35c>
   DB  196,65,60,88,195                    ; vaddps        %ymm11,%ymm8,%ymm8
-  DB  196,98,125,24,37,189,33,0,0         ; vbroadcastss  0x21bd(%rip),%ymm12        # 4a74 <_sk_callback_hsw+0x360>
+  DB  196,98,125,24,37,189,33,0,0         ; vbroadcastss  0x21bd(%rip),%ymm12        # 4a90 <_sk_callback_hsw+0x360>
   DB  196,65,60,89,196                    ; vmulps        %ymm12,%ymm8,%ymm8
   DB  196,99,61,74,193,160                ; vblendvps     %ymm10,%ymm1,%ymm8,%ymm8
   DB  197,252,89,200                      ; vmulps        %ymm0,%ymm0,%ymm1
@@ -2414,9 +2420,9 @@ _sk_lab_to_xyz_hsw LABEL PROC
   DB  196,65,52,88,203                    ; vaddps        %ymm11,%ymm9,%ymm9
   DB  196,65,52,89,204                    ; vmulps        %ymm12,%ymm9,%ymm9
   DB  196,227,53,74,208,32                ; vblendvps     %ymm2,%ymm0,%ymm9,%ymm2
-  DB  196,226,125,24,5,114,33,0,0         ; vbroadcastss  0x2172(%rip),%ymm0        # 4a78 <_sk_callback_hsw+0x364>
+  DB  196,226,125,24,5,114,33,0,0         ; vbroadcastss  0x2172(%rip),%ymm0        # 4a94 <_sk_callback_hsw+0x364>
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
-  DB  196,98,125,24,5,105,33,0,0          ; vbroadcastss  0x2169(%rip),%ymm8        # 4a7c <_sk_callback_hsw+0x368>
+  DB  196,98,125,24,5,105,33,0,0          ; vbroadcastss  0x2169(%rip),%ymm8        # 4a98 <_sk_callback_hsw+0x368>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2428,11 +2434,11 @@ _sk_load_a8_hsw LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,45                              ; jne           2959 <_sk_load_a8_hsw+0x3d>
+  DB  117,45                              ; jne           2975 <_sk_load_a8_hsw+0x3d>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,62,33,0,0         ; vbroadcastss  0x213e(%rip),%ymm1        # 4a80 <_sk_callback_hsw+0x36c>
+  DB  196,226,125,24,13,62,33,0,0         ; vbroadcastss  0x213e(%rip),%ymm1        # 4a9c <_sk_callback_hsw+0x36c>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -2449,9 +2455,9 @@ _sk_load_a8_hsw LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           2961 <_sk_load_a8_hsw+0x45>
+  DB  117,234                             ; jne           297d <_sk_load_a8_hsw+0x45>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,178                             ; jmp           2930 <_sk_load_a8_hsw+0x14>
+  DB  235,178                             ; jmp           294c <_sk_load_a8_hsw+0x14>
 
 PUBLIC _sk_gather_a8_hsw
 _sk_gather_a8_hsw LABEL PROC
@@ -2495,7 +2501,7 @@ _sk_gather_a8_hsw LABEL PROC
   DB  196,227,121,32,192,7                ; vpinsrb       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,73,32,0,0         ; vbroadcastss  0x2049(%rip),%ymm1        # 4a84 <_sk_callback_hsw+0x370>
+  DB  196,226,125,24,13,73,32,0,0         ; vbroadcastss  0x2049(%rip),%ymm1        # 4aa0 <_sk_callback_hsw+0x370>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -2511,14 +2517,14 @@ PUBLIC _sk_store_a8_hsw
 _sk_store_a8_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,36,32,0,0           ; vbroadcastss  0x2024(%rip),%ymm8        # 4a88 <_sk_callback_hsw+0x374>
+  DB  196,98,125,24,5,36,32,0,0           ; vbroadcastss  0x2024(%rip),%ymm8        # 4aa4 <_sk_callback_hsw+0x374>
   DB  196,65,100,89,192                   ; vmulps        %ymm8,%ymm3,%ymm8
   DB  196,65,125,91,192                   ; vcvtps2dq     %ymm8,%ymm8
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  196,65,57,103,192                   ; vpackuswb     %xmm8,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           2a8d <_sk_store_a8_hsw+0x37>
+  DB  117,10                              ; jne           2aa9 <_sk_store_a8_hsw+0x37>
   DB  196,65,123,17,4,58                  ; vmovsd        %xmm8,(%r10,%rdi,1)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2526,10 +2532,10 @@ _sk_store_a8_hsw LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            2a89 <_sk_store_a8_hsw+0x33>
+  DB  119,236                             ; ja            2aa5 <_sk_store_a8_hsw+0x33>
   DB  196,66,121,48,192                   ; vpmovzxbw     %xmm8,%xmm8
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 2af0 <_sk_store_a8_hsw+0x9a>
+  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 2b0c <_sk_store_a8_hsw+0x9a>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2540,7 +2546,7 @@ _sk_store_a8_hsw LABEL PROC
   DB  196,67,121,20,68,58,2,4             ; vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   DB  196,67,121,20,68,58,1,2             ; vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   DB  196,67,121,20,4,58,0                ; vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  DB  235,154                             ; jmp           2a89 <_sk_store_a8_hsw+0x33>
+  DB  235,154                             ; jmp           2aa5 <_sk_store_a8_hsw+0x33>
   DB  144                                 ; nop
   DB  246,255                             ; idiv          %bh
   DB  255                                 ; (bad)
@@ -2572,14 +2578,14 @@ _sk_load_g8_hsw LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,50                              ; jne           2b4e <_sk_load_g8_hsw+0x42>
+  DB  117,50                              ; jne           2b6a <_sk_load_g8_hsw+0x42>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,90,31,0,0         ; vbroadcastss  0x1f5a(%rip),%ymm1        # 4a8c <_sk_callback_hsw+0x378>
+  DB  196,226,125,24,13,90,31,0,0         ; vbroadcastss  0x1f5a(%rip),%ymm1        # 4aa8 <_sk_callback_hsw+0x378>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,79,31,0,0         ; vbroadcastss  0x1f4f(%rip),%ymm3        # 4a90 <_sk_callback_hsw+0x37c>
+  DB  196,226,125,24,29,79,31,0,0         ; vbroadcastss  0x1f4f(%rip),%ymm3        # 4aac <_sk_callback_hsw+0x37c>
   DB  76,137,193                          ; mov           %r8,%rcx
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
@@ -2593,9 +2599,9 @@ _sk_load_g8_hsw LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           2b56 <_sk_load_g8_hsw+0x4a>
+  DB  117,234                             ; jne           2b72 <_sk_load_g8_hsw+0x4a>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,173                             ; jmp           2b20 <_sk_load_g8_hsw+0x14>
+  DB  235,173                             ; jmp           2b3c <_sk_load_g8_hsw+0x14>
 
 PUBLIC _sk_gather_g8_hsw
 _sk_gather_g8_hsw LABEL PROC
@@ -2639,10 +2645,10 @@ _sk_gather_g8_hsw LABEL PROC
   DB  196,227,121,32,192,7                ; vpinsrb       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,49,192                  ; vpmovzxbd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,100,30,0,0        ; vbroadcastss  0x1e64(%rip),%ymm1        # 4a94 <_sk_callback_hsw+0x380>
+  DB  196,226,125,24,13,100,30,0,0        ; vbroadcastss  0x1e64(%rip),%ymm1        # 4ab0 <_sk_callback_hsw+0x380>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,89,30,0,0         ; vbroadcastss  0x1e59(%rip),%ymm3        # 4a98 <_sk_callback_hsw+0x384>
+  DB  196,226,125,24,29,89,30,0,0         ; vbroadcastss  0x1e59(%rip),%ymm3        # 4ab4 <_sk_callback_hsw+0x384>
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
   DB  91                                  ; pop           %rbx
@@ -2656,9 +2662,9 @@ _sk_gather_i8_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            2c5f <_sk_gather_i8_hsw+0xf>
+  DB  116,5                               ; je            2c7b <_sk_gather_i8_hsw+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           2c61 <_sk_gather_i8_hsw+0x11>
+  DB  235,2                               ; jmp           2c7d <_sk_gather_i8_hsw+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,87                               ; push          %r15
   DB  65,86                               ; push          %r14
@@ -2696,14 +2702,14 @@ _sk_gather_i8_hsw LABEL PROC
   DB  73,139,64,8                         ; mov           0x8(%r8),%rax
   DB  197,245,118,201                     ; vpcmpeqd      %ymm1,%ymm1,%ymm1
   DB  196,226,117,144,28,128              ; vpgatherdd    %ymm1,(%rax,%ymm0,4),%ymm3
-  DB  197,229,219,5,113,31,0,0            ; vpand         0x1f71(%rip),%ymm3,%ymm0        # 4c80 <_sk_callback_hsw+0x56c>
+  DB  197,229,219,5,117,31,0,0            ; vpand         0x1f75(%rip),%ymm3,%ymm0        # 4ca0 <_sk_callback_hsw+0x570>
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,128,29,0,0          ; vbroadcastss  0x1d80(%rip),%ymm8        # 4a9c <_sk_callback_hsw+0x388>
+  DB  196,98,125,24,5,128,29,0,0          ; vbroadcastss  0x1d80(%rip),%ymm8        # 4ab8 <_sk_callback_hsw+0x388>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,226,101,0,13,118,31,0,0         ; vpshufb       0x1f76(%rip),%ymm3,%ymm1        # 4ca0 <_sk_callback_hsw+0x58c>
+  DB  196,226,101,0,13,122,31,0,0         ; vpshufb       0x1f7a(%rip),%ymm3,%ymm1        # 4cc0 <_sk_callback_hsw+0x590>
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,226,101,0,21,132,31,0,0         ; vpshufb       0x1f84(%rip),%ymm3,%ymm2        # 4cc0 <_sk_callback_hsw+0x5ac>
+  DB  196,226,101,0,21,136,31,0,0         ; vpshufb       0x1f88(%rip),%ymm3,%ymm2        # 4ce0 <_sk_callback_hsw+0x5b0>
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
@@ -2722,35 +2728,35 @@ _sk_load_565_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,114                             ; jne           2ddc <_sk_load_565_hsw+0x7c>
+  DB  117,114                             ; jne           2df8 <_sk_load_565_hsw+0x7c>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  196,226,125,51,208                  ; vpmovzxwd     %xmm0,%ymm2
-  DB  196,226,125,88,5,34,29,0,0          ; vpbroadcastd  0x1d22(%rip),%ymm0        # 4aa0 <_sk_callback_hsw+0x38c>
+  DB  196,226,125,88,5,34,29,0,0          ; vpbroadcastd  0x1d22(%rip),%ymm0        # 4abc <_sk_callback_hsw+0x38c>
   DB  197,237,219,192                     ; vpand         %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,21,29,0,0         ; vbroadcastss  0x1d15(%rip),%ymm1        # 4aa4 <_sk_callback_hsw+0x390>
+  DB  196,226,125,24,13,21,29,0,0         ; vbroadcastss  0x1d15(%rip),%ymm1        # 4ac0 <_sk_callback_hsw+0x390>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,12,29,0,0         ; vpbroadcastd  0x1d0c(%rip),%ymm1        # 4aa8 <_sk_callback_hsw+0x394>
+  DB  196,226,125,88,13,12,29,0,0         ; vpbroadcastd  0x1d0c(%rip),%ymm1        # 4ac4 <_sk_callback_hsw+0x394>
   DB  197,237,219,201                     ; vpand         %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,255,28,0,0        ; vbroadcastss  0x1cff(%rip),%ymm3        # 4aac <_sk_callback_hsw+0x398>
+  DB  196,226,125,24,29,255,28,0,0        ; vbroadcastss  0x1cff(%rip),%ymm3        # 4ac8 <_sk_callback_hsw+0x398>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,88,29,246,28,0,0        ; vpbroadcastd  0x1cf6(%rip),%ymm3        # 4ab0 <_sk_callback_hsw+0x39c>
+  DB  196,226,125,88,29,246,28,0,0        ; vpbroadcastd  0x1cf6(%rip),%ymm3        # 4acc <_sk_callback_hsw+0x39c>
   DB  197,237,219,211                     ; vpand         %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,233,28,0,0        ; vbroadcastss  0x1ce9(%rip),%ymm3        # 4ab4 <_sk_callback_hsw+0x3a0>
+  DB  196,226,125,24,29,233,28,0,0        ; vbroadcastss  0x1ce9(%rip),%ymm3        # 4ad0 <_sk_callback_hsw+0x3a0>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,222,28,0,0        ; vbroadcastss  0x1cde(%rip),%ymm3        # 4ab8 <_sk_callback_hsw+0x3a4>
+  DB  196,226,125,24,29,222,28,0,0        ; vbroadcastss  0x1cde(%rip),%ymm3        # 4ad4 <_sk_callback_hsw+0x3a4>
   DB  255,224                             ; jmpq          *%rax
   DB  65,137,200                          ; mov           %ecx,%r8d
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,128                             ; ja            2d70 <_sk_load_565_hsw+0x10>
+  DB  119,128                             ; ja            2d8c <_sk_load_565_hsw+0x10>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 2e44 <_sk_load_565_hsw+0xe4>
+  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 2e60 <_sk_load_565_hsw+0xe4>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2762,7 +2768,7 @@ _sk_load_565_hsw LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,44,255,255,255                  ; jmpq          2d70 <_sk_load_565_hsw+0x10>
+  DB  233,44,255,255,255                  ; jmpq          2d8c <_sk_load_565_hsw+0x10>
   DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -2830,23 +2836,23 @@ _sk_gather_565_hsw LABEL PROC
   DB  65,15,183,4,88                      ; movzwl        (%r8,%rbx,2),%eax
   DB  197,249,196,192,7                   ; vpinsrw       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,51,208                  ; vpmovzxwd     %xmm0,%ymm2
-  DB  196,226,125,88,5,161,27,0,0         ; vpbroadcastd  0x1ba1(%rip),%ymm0        # 4abc <_sk_callback_hsw+0x3a8>
+  DB  196,226,125,88,5,161,27,0,0         ; vpbroadcastd  0x1ba1(%rip),%ymm0        # 4ad8 <_sk_callback_hsw+0x3a8>
   DB  197,237,219,192                     ; vpand         %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,148,27,0,0        ; vbroadcastss  0x1b94(%rip),%ymm1        # 4ac0 <_sk_callback_hsw+0x3ac>
+  DB  196,226,125,24,13,148,27,0,0        ; vbroadcastss  0x1b94(%rip),%ymm1        # 4adc <_sk_callback_hsw+0x3ac>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,139,27,0,0        ; vpbroadcastd  0x1b8b(%rip),%ymm1        # 4ac4 <_sk_callback_hsw+0x3b0>
+  DB  196,226,125,88,13,139,27,0,0        ; vpbroadcastd  0x1b8b(%rip),%ymm1        # 4ae0 <_sk_callback_hsw+0x3b0>
   DB  197,237,219,201                     ; vpand         %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,126,27,0,0        ; vbroadcastss  0x1b7e(%rip),%ymm3        # 4ac8 <_sk_callback_hsw+0x3b4>
+  DB  196,226,125,24,29,126,27,0,0        ; vbroadcastss  0x1b7e(%rip),%ymm3        # 4ae4 <_sk_callback_hsw+0x3b4>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,88,29,117,27,0,0        ; vpbroadcastd  0x1b75(%rip),%ymm3        # 4acc <_sk_callback_hsw+0x3b8>
+  DB  196,226,125,88,29,117,27,0,0        ; vpbroadcastd  0x1b75(%rip),%ymm3        # 4ae8 <_sk_callback_hsw+0x3b8>
   DB  197,237,219,211                     ; vpand         %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,104,27,0,0        ; vbroadcastss  0x1b68(%rip),%ymm3        # 4ad0 <_sk_callback_hsw+0x3bc>
+  DB  196,226,125,24,29,104,27,0,0        ; vbroadcastss  0x1b68(%rip),%ymm3        # 4aec <_sk_callback_hsw+0x3bc>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,93,27,0,0         ; vbroadcastss  0x1b5d(%rip),%ymm3        # 4ad4 <_sk_callback_hsw+0x3c0>
+  DB  196,226,125,24,29,93,27,0,0         ; vbroadcastss  0x1b5d(%rip),%ymm3        # 4af0 <_sk_callback_hsw+0x3c0>
   DB  91                                  ; pop           %rbx
   DB  65,92                               ; pop           %r12
   DB  65,94                               ; pop           %r14
@@ -2857,11 +2863,11 @@ PUBLIC _sk_store_565_hsw
 _sk_store_565_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,74,27,0,0           ; vbroadcastss  0x1b4a(%rip),%ymm8        # 4ad8 <_sk_callback_hsw+0x3c4>
+  DB  196,98,125,24,5,74,27,0,0           ; vbroadcastss  0x1b4a(%rip),%ymm8        # 4af4 <_sk_callback_hsw+0x3c4>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,53,114,241,11               ; vpslld        $0xb,%ymm9,%ymm9
-  DB  196,98,125,24,21,53,27,0,0          ; vbroadcastss  0x1b35(%rip),%ymm10        # 4adc <_sk_callback_hsw+0x3c8>
+  DB  196,98,125,24,21,53,27,0,0          ; vbroadcastss  0x1b35(%rip),%ymm10        # 4af8 <_sk_callback_hsw+0x3c8>
   DB  196,65,116,89,210                   ; vmulps        %ymm10,%ymm1,%ymm10
   DB  196,65,125,91,210                   ; vcvtps2dq     %ymm10,%ymm10
   DB  196,193,45,114,242,5                ; vpslld        $0x5,%ymm10,%ymm10
@@ -2872,7 +2878,7 @@ _sk_store_565_hsw LABEL PROC
   DB  196,67,125,57,193,1                 ; vextracti128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           2fe5 <_sk_store_565_hsw+0x65>
+  DB  117,10                              ; jne           3001 <_sk_store_565_hsw+0x65>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2880,9 +2886,9 @@ _sk_store_565_hsw LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            2fe1 <_sk_store_565_hsw+0x61>
+  DB  119,236                             ; ja            2ffd <_sk_store_565_hsw+0x61>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3044 <_sk_store_565_hsw+0xc4>
+  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3060 <_sk_store_565_hsw+0xc4>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2893,7 +2899,7 @@ _sk_store_565_hsw LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           2fe1 <_sk_store_565_hsw+0x61>
+  DB  235,159                             ; jmp           2ffd <_sk_store_565_hsw+0x61>
   DB  102,144                             ; xchg          %ax,%ax
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -2924,28 +2930,28 @@ _sk_load_4444_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,138,0,0,0                    ; jne           30f8 <_sk_load_4444_hsw+0x98>
+  DB  15,133,138,0,0,0                    ; jne           3114 <_sk_load_4444_hsw+0x98>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  196,226,125,51,216                  ; vpmovzxwd     %xmm0,%ymm3
-  DB  196,226,125,88,5,94,26,0,0          ; vpbroadcastd  0x1a5e(%rip),%ymm0        # 4ae0 <_sk_callback_hsw+0x3cc>
+  DB  196,226,125,88,5,94,26,0,0          ; vpbroadcastd  0x1a5e(%rip),%ymm0        # 4afc <_sk_callback_hsw+0x3cc>
   DB  197,229,219,192                     ; vpand         %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,81,26,0,0         ; vbroadcastss  0x1a51(%rip),%ymm1        # 4ae4 <_sk_callback_hsw+0x3d0>
+  DB  196,226,125,24,13,81,26,0,0         ; vbroadcastss  0x1a51(%rip),%ymm1        # 4b00 <_sk_callback_hsw+0x3d0>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,72,26,0,0         ; vpbroadcastd  0x1a48(%rip),%ymm1        # 4ae8 <_sk_callback_hsw+0x3d4>
+  DB  196,226,125,88,13,72,26,0,0         ; vpbroadcastd  0x1a48(%rip),%ymm1        # 4b04 <_sk_callback_hsw+0x3d4>
   DB  197,229,219,201                     ; vpand         %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,59,26,0,0         ; vbroadcastss  0x1a3b(%rip),%ymm2        # 4aec <_sk_callback_hsw+0x3d8>
+  DB  196,226,125,24,21,59,26,0,0         ; vbroadcastss  0x1a3b(%rip),%ymm2        # 4b08 <_sk_callback_hsw+0x3d8>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,88,21,50,26,0,0         ; vpbroadcastd  0x1a32(%rip),%ymm2        # 4af0 <_sk_callback_hsw+0x3dc>
+  DB  196,226,125,88,21,50,26,0,0         ; vpbroadcastd  0x1a32(%rip),%ymm2        # 4b0c <_sk_callback_hsw+0x3dc>
   DB  197,229,219,210                     ; vpand         %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,37,26,0,0           ; vbroadcastss  0x1a25(%rip),%ymm8        # 4af4 <_sk_callback_hsw+0x3e0>
+  DB  196,98,125,24,5,37,26,0,0           ; vbroadcastss  0x1a25(%rip),%ymm8        # 4b10 <_sk_callback_hsw+0x3e0>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,88,5,27,26,0,0           ; vpbroadcastd  0x1a1b(%rip),%ymm8        # 4af8 <_sk_callback_hsw+0x3e4>
+  DB  196,98,125,88,5,27,26,0,0           ; vpbroadcastd  0x1a1b(%rip),%ymm8        # 4b14 <_sk_callback_hsw+0x3e4>
   DB  196,193,101,219,216                 ; vpand         %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,13,26,0,0           ; vbroadcastss  0x1a0d(%rip),%ymm8        # 4afc <_sk_callback_hsw+0x3e8>
+  DB  196,98,125,24,5,13,26,0,0           ; vbroadcastss  0x1a0d(%rip),%ymm8        # 4b18 <_sk_callback_hsw+0x3e8>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2954,9 +2960,9 @@ _sk_load_4444_hsw LABEL PROC
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,100,255,255,255              ; ja            3074 <_sk_load_4444_hsw+0x14>
+  DB  15,135,100,255,255,255              ; ja            3090 <_sk_load_4444_hsw+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 3164 <_sk_load_4444_hsw+0x104>
+  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 3180 <_sk_load_4444_hsw+0x104>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -2968,7 +2974,7 @@ _sk_load_4444_hsw LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,16,255,255,255                  ; jmpq          3074 <_sk_load_4444_hsw+0x14>
+  DB  233,16,255,255,255                  ; jmpq          3090 <_sk_load_4444_hsw+0x14>
   DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -3036,25 +3042,25 @@ _sk_gather_4444_hsw LABEL PROC
   DB  65,15,183,4,88                      ; movzwl        (%r8,%rbx,2),%eax
   DB  197,249,196,192,7                   ; vpinsrw       $0x7,%eax,%xmm0,%xmm0
   DB  196,226,125,51,216                  ; vpmovzxwd     %xmm0,%ymm3
-  DB  196,226,125,88,5,197,24,0,0         ; vpbroadcastd  0x18c5(%rip),%ymm0        # 4b00 <_sk_callback_hsw+0x3ec>
+  DB  196,226,125,88,5,197,24,0,0         ; vpbroadcastd  0x18c5(%rip),%ymm0        # 4b1c <_sk_callback_hsw+0x3ec>
   DB  197,229,219,192                     ; vpand         %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,184,24,0,0        ; vbroadcastss  0x18b8(%rip),%ymm1        # 4b04 <_sk_callback_hsw+0x3f0>
+  DB  196,226,125,24,13,184,24,0,0        ; vbroadcastss  0x18b8(%rip),%ymm1        # 4b20 <_sk_callback_hsw+0x3f0>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,88,13,175,24,0,0        ; vpbroadcastd  0x18af(%rip),%ymm1        # 4b08 <_sk_callback_hsw+0x3f4>
+  DB  196,226,125,88,13,175,24,0,0        ; vpbroadcastd  0x18af(%rip),%ymm1        # 4b24 <_sk_callback_hsw+0x3f4>
   DB  197,229,219,201                     ; vpand         %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,162,24,0,0        ; vbroadcastss  0x18a2(%rip),%ymm2        # 4b0c <_sk_callback_hsw+0x3f8>
+  DB  196,226,125,24,21,162,24,0,0        ; vbroadcastss  0x18a2(%rip),%ymm2        # 4b28 <_sk_callback_hsw+0x3f8>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,88,21,153,24,0,0        ; vpbroadcastd  0x1899(%rip),%ymm2        # 4b10 <_sk_callback_hsw+0x3fc>
+  DB  196,226,125,88,21,153,24,0,0        ; vpbroadcastd  0x1899(%rip),%ymm2        # 4b2c <_sk_callback_hsw+0x3fc>
   DB  197,229,219,210                     ; vpand         %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,140,24,0,0          ; vbroadcastss  0x188c(%rip),%ymm8        # 4b14 <_sk_callback_hsw+0x400>
+  DB  196,98,125,24,5,140,24,0,0          ; vbroadcastss  0x188c(%rip),%ymm8        # 4b30 <_sk_callback_hsw+0x400>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,88,5,130,24,0,0          ; vpbroadcastd  0x1882(%rip),%ymm8        # 4b18 <_sk_callback_hsw+0x404>
+  DB  196,98,125,88,5,130,24,0,0          ; vpbroadcastd  0x1882(%rip),%ymm8        # 4b34 <_sk_callback_hsw+0x404>
   DB  196,193,101,219,216                 ; vpand         %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,116,24,0,0          ; vbroadcastss  0x1874(%rip),%ymm8        # 4b1c <_sk_callback_hsw+0x408>
+  DB  196,98,125,24,5,116,24,0,0          ; vbroadcastss  0x1874(%rip),%ymm8        # 4b38 <_sk_callback_hsw+0x408>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -3067,7 +3073,7 @@ PUBLIC _sk_store_4444_hsw
 _sk_store_4444_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,90,24,0,0           ; vbroadcastss  0x185a(%rip),%ymm8        # 4b20 <_sk_callback_hsw+0x40c>
+  DB  196,98,125,24,5,90,24,0,0           ; vbroadcastss  0x185a(%rip),%ymm8        # 4b3c <_sk_callback_hsw+0x40c>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,53,114,241,12               ; vpslld        $0xc,%ymm9,%ymm9
@@ -3085,7 +3091,7 @@ _sk_store_4444_hsw LABEL PROC
   DB  196,67,125,57,193,1                 ; vextracti128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           3329 <_sk_store_4444_hsw+0x71>
+  DB  117,10                              ; jne           3345 <_sk_store_4444_hsw+0x71>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -3093,9 +3099,9 @@ _sk_store_4444_hsw LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            3325 <_sk_store_4444_hsw+0x6d>
+  DB  119,236                             ; ja            3341 <_sk_store_4444_hsw+0x6d>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3388 <_sk_store_4444_hsw+0xd0>
+  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 33a4 <_sk_store_4444_hsw+0xd0>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -3106,7 +3112,7 @@ _sk_store_4444_hsw LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           3325 <_sk_store_4444_hsw+0x6d>
+  DB  235,159                             ; jmp           3341 <_sk_store_4444_hsw+0x6d>
   DB  102,144                             ; xchg          %ax,%ax
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -3139,16 +3145,16 @@ _sk_load_8888_hsw LABEL PROC
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  76,3,8                              ; add           (%rax),%r9
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,88                              ; jne           3411 <_sk_load_8888_hsw+0x6d>
+  DB  117,88                              ; jne           342d <_sk_load_8888_hsw+0x6d>
   DB  196,193,126,111,25                  ; vmovdqu       (%r9),%ymm3
-  DB  197,229,219,5,26,25,0,0             ; vpand         0x191a(%rip),%ymm3,%ymm0        # 4ce0 <_sk_callback_hsw+0x5cc>
+  DB  197,229,219,5,30,25,0,0             ; vpand         0x191e(%rip),%ymm3,%ymm0        # 4d00 <_sk_callback_hsw+0x5d0>
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,81,23,0,0           ; vbroadcastss  0x1751(%rip),%ymm8        # 4b24 <_sk_callback_hsw+0x410>
+  DB  196,98,125,24,5,81,23,0,0           ; vbroadcastss  0x1751(%rip),%ymm8        # 4b40 <_sk_callback_hsw+0x410>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,226,101,0,13,31,25,0,0          ; vpshufb       0x191f(%rip),%ymm3,%ymm1        # 4d00 <_sk_callback_hsw+0x5ec>
+  DB  196,226,101,0,13,35,25,0,0          ; vpshufb       0x1923(%rip),%ymm3,%ymm1        # 4d20 <_sk_callback_hsw+0x5f0>
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,226,101,0,21,45,25,0,0          ; vpshufb       0x192d(%rip),%ymm3,%ymm2        # 4d20 <_sk_callback_hsw+0x60c>
+  DB  196,226,101,0,21,49,25,0,0          ; vpshufb       0x1931(%rip),%ymm3,%ymm2        # 4d40 <_sk_callback_hsw+0x610>
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
@@ -3165,7 +3171,7 @@ _sk_load_8888_hsw LABEL PROC
   DB  196,225,249,110,192                 ; vmovq         %rax,%xmm0
   DB  196,226,125,33,192                  ; vpmovsxbd     %xmm0,%ymm0
   DB  196,194,125,140,25                  ; vpmaskmovd    (%r9),%ymm0,%ymm3
-  DB  235,135                             ; jmp           33be <_sk_load_8888_hsw+0x1a>
+  DB  235,135                             ; jmp           33da <_sk_load_8888_hsw+0x1a>
 
 PUBLIC _sk_gather_8888_hsw
 _sk_gather_8888_hsw LABEL PROC
@@ -3178,14 +3184,14 @@ _sk_gather_8888_hsw LABEL PROC
   DB  197,245,254,192                     ; vpaddd        %ymm0,%ymm1,%ymm0
   DB  197,245,118,201                     ; vpcmpeqd      %ymm1,%ymm1,%ymm1
   DB  196,194,117,144,28,128              ; vpgatherdd    %ymm1,(%r8,%ymm0,4),%ymm3
-  DB  197,229,219,5,219,24,0,0            ; vpand         0x18db(%rip),%ymm3,%ymm0        # 4d40 <_sk_callback_hsw+0x62c>
+  DB  197,229,219,5,223,24,0,0            ; vpand         0x18df(%rip),%ymm3,%ymm0        # 4d60 <_sk_callback_hsw+0x630>
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,182,22,0,0          ; vbroadcastss  0x16b6(%rip),%ymm8        # 4b28 <_sk_callback_hsw+0x414>
+  DB  196,98,125,24,5,182,22,0,0          ; vbroadcastss  0x16b6(%rip),%ymm8        # 4b44 <_sk_callback_hsw+0x414>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,226,101,0,13,224,24,0,0         ; vpshufb       0x18e0(%rip),%ymm3,%ymm1        # 4d60 <_sk_callback_hsw+0x64c>
+  DB  196,226,101,0,13,228,24,0,0         ; vpshufb       0x18e4(%rip),%ymm3,%ymm1        # 4d80 <_sk_callback_hsw+0x650>
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,226,101,0,21,238,24,0,0         ; vpshufb       0x18ee(%rip),%ymm3,%ymm2        # 4d80 <_sk_callback_hsw+0x66c>
+  DB  196,226,101,0,21,242,24,0,0         ; vpshufb       0x18f2(%rip),%ymm3,%ymm2        # 4da0 <_sk_callback_hsw+0x670>
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,229,114,211,24                  ; vpsrld        $0x18,%ymm3,%ymm3
@@ -3200,7 +3206,7 @@ _sk_store_8888_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  76,3,8                              ; add           (%rax),%r9
-  DB  196,98,125,24,5,102,22,0,0          ; vbroadcastss  0x1666(%rip),%ymm8        # 4b2c <_sk_callback_hsw+0x418>
+  DB  196,98,125,24,5,102,22,0,0          ; vbroadcastss  0x1666(%rip),%ymm8        # 4b48 <_sk_callback_hsw+0x418>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,65,116,89,208                   ; vmulps        %ymm8,%ymm1,%ymm10
@@ -3216,7 +3222,7 @@ _sk_store_8888_hsw LABEL PROC
   DB  196,65,45,235,192                   ; vpor          %ymm8,%ymm10,%ymm8
   DB  196,65,53,235,192                   ; vpor          %ymm8,%ymm9,%ymm8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,12                              ; jne           3520 <_sk_store_8888_hsw+0x73>
+  DB  117,12                              ; jne           353c <_sk_store_8888_hsw+0x73>
   DB  196,65,126,127,1                    ; vmovdqu       %ymm8,(%r9)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,137,193                          ; mov           %r8,%rcx
@@ -3229,14 +3235,14 @@ _sk_store_8888_hsw LABEL PROC
   DB  196,97,249,110,200                  ; vmovq         %rax,%xmm9
   DB  196,66,125,33,201                   ; vpmovsxbd     %xmm9,%ymm9
   DB  196,66,53,142,1                     ; vpmaskmovd    %ymm8,%ymm9,(%r9)
-  DB  235,211                             ; jmp           3519 <_sk_store_8888_hsw+0x6c>
+  DB  235,211                             ; jmp           3535 <_sk_store_8888_hsw+0x6c>
 
 PUBLIC _sk_load_f16_hsw
 _sk_load_f16_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,97                              ; jne           35b1 <_sk_load_f16_hsw+0x6b>
+  DB  117,97                              ; jne           35cd <_sk_load_f16_hsw+0x6b>
   DB  197,121,16,4,248                    ; vmovupd       (%rax,%rdi,8),%xmm8
   DB  197,249,16,84,248,16                ; vmovupd       0x10(%rax,%rdi,8),%xmm2
   DB  197,249,16,92,248,32                ; vmovupd       0x20(%rax,%rdi,8),%xmm3
@@ -3262,29 +3268,29 @@ _sk_load_f16_hsw LABEL PROC
   DB  197,123,16,4,248                    ; vmovsd        (%rax,%rdi,8),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,79                              ; je            3610 <_sk_load_f16_hsw+0xca>
+  DB  116,79                              ; je            362c <_sk_load_f16_hsw+0xca>
   DB  197,57,22,68,248,8                  ; vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,67                              ; jb            3610 <_sk_load_f16_hsw+0xca>
+  DB  114,67                              ; jb            362c <_sk_load_f16_hsw+0xca>
   DB  197,251,16,84,248,16                ; vmovsd        0x10(%rax,%rdi,8),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,68                              ; je            361d <_sk_load_f16_hsw+0xd7>
+  DB  116,68                              ; je            3639 <_sk_load_f16_hsw+0xd7>
   DB  197,233,22,84,248,24                ; vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,56                              ; jb            361d <_sk_load_f16_hsw+0xd7>
+  DB  114,56                              ; jb            3639 <_sk_load_f16_hsw+0xd7>
   DB  197,251,16,92,248,32                ; vmovsd        0x20(%rax,%rdi,8),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,114,255,255,255              ; je            3567 <_sk_load_f16_hsw+0x21>
+  DB  15,132,114,255,255,255              ; je            3583 <_sk_load_f16_hsw+0x21>
   DB  197,225,22,92,248,40                ; vmovhpd       0x28(%rax,%rdi,8),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,98,255,255,255               ; jb            3567 <_sk_load_f16_hsw+0x21>
+  DB  15,130,98,255,255,255               ; jb            3583 <_sk_load_f16_hsw+0x21>
   DB  197,122,126,76,248,48               ; vmovq         0x30(%rax,%rdi,8),%xmm9
-  DB  233,87,255,255,255                  ; jmpq          3567 <_sk_load_f16_hsw+0x21>
+  DB  233,87,255,255,255                  ; jmpq          3583 <_sk_load_f16_hsw+0x21>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,74,255,255,255                  ; jmpq          3567 <_sk_load_f16_hsw+0x21>
+  DB  233,74,255,255,255                  ; jmpq          3583 <_sk_load_f16_hsw+0x21>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,65,255,255,255                  ; jmpq          3567 <_sk_load_f16_hsw+0x21>
+  DB  233,65,255,255,255                  ; jmpq          3583 <_sk_load_f16_hsw+0x21>
 
 PUBLIC _sk_gather_f16_hsw
 _sk_gather_f16_hsw LABEL PROC
@@ -3338,7 +3344,7 @@ _sk_store_f16_hsw LABEL PROC
   DB  196,65,57,98,205                    ; vpunpckldq    %xmm13,%xmm8,%xmm9
   DB  196,65,57,106,197                   ; vpunpckhdq    %xmm13,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,27                              ; jne           3715 <_sk_store_f16_hsw+0x65>
+  DB  117,27                              ; jne           3731 <_sk_store_f16_hsw+0x65>
   DB  197,120,17,28,248                   ; vmovups       %xmm11,(%rax,%rdi,8)
   DB  197,120,17,84,248,16                ; vmovups       %xmm10,0x10(%rax,%rdi,8)
   DB  197,120,17,76,248,32                ; vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -3347,22 +3353,22 @@ _sk_store_f16_hsw LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  197,121,214,28,248                  ; vmovq         %xmm11,(%rax,%rdi,8)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,241                             ; je            3711 <_sk_store_f16_hsw+0x61>
+  DB  116,241                             ; je            372d <_sk_store_f16_hsw+0x61>
   DB  197,121,23,92,248,8                 ; vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,229                             ; jb            3711 <_sk_store_f16_hsw+0x61>
+  DB  114,229                             ; jb            372d <_sk_store_f16_hsw+0x61>
   DB  197,121,214,84,248,16               ; vmovq         %xmm10,0x10(%rax,%rdi,8)
-  DB  116,221                             ; je            3711 <_sk_store_f16_hsw+0x61>
+  DB  116,221                             ; je            372d <_sk_store_f16_hsw+0x61>
   DB  197,121,23,84,248,24                ; vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,209                             ; jb            3711 <_sk_store_f16_hsw+0x61>
+  DB  114,209                             ; jb            372d <_sk_store_f16_hsw+0x61>
   DB  197,121,214,76,248,32               ; vmovq         %xmm9,0x20(%rax,%rdi,8)
-  DB  116,201                             ; je            3711 <_sk_store_f16_hsw+0x61>
+  DB  116,201                             ; je            372d <_sk_store_f16_hsw+0x61>
   DB  197,121,23,76,248,40                ; vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,189                             ; jb            3711 <_sk_store_f16_hsw+0x61>
+  DB  114,189                             ; jb            372d <_sk_store_f16_hsw+0x61>
   DB  197,121,214,68,248,48               ; vmovq         %xmm8,0x30(%rax,%rdi,8)
-  DB  235,181                             ; jmp           3711 <_sk_store_f16_hsw+0x61>
+  DB  235,181                             ; jmp           372d <_sk_store_f16_hsw+0x61>
 
 PUBLIC _sk_load_u16_be_hsw
 _sk_load_u16_be_hsw LABEL PROC
@@ -3370,7 +3376,7 @@ _sk_load_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,204,0,0,0                    ; jne           383e <_sk_load_u16_be_hsw+0xe2>
+  DB  15,133,204,0,0,0                    ; jne           385a <_sk_load_u16_be_hsw+0xe2>
   DB  196,65,121,16,4,64                  ; vmovupd       (%r8,%rax,2),%xmm8
   DB  196,193,121,16,84,64,16             ; vmovupd       0x10(%r8,%rax,2),%xmm2
   DB  196,193,121,16,92,64,32             ; vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -3389,7 +3395,7 @@ _sk_load_u16_be_hsw LABEL PROC
   DB  197,241,235,192                     ; vpor          %xmm0,%xmm1,%xmm0
   DB  196,226,125,51,192                  ; vpmovzxwd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,21,93,19,0,0          ; vbroadcastss  0x135d(%rip),%ymm10        # 4b30 <_sk_callback_hsw+0x41c>
+  DB  196,98,125,24,21,93,19,0,0          ; vbroadcastss  0x135d(%rip),%ymm10        # 4b4c <_sk_callback_hsw+0x41c>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,185,109,202                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -3417,29 +3423,29 @@ _sk_load_u16_be_hsw LABEL PROC
   DB  196,65,123,16,4,64                  ; vmovsd        (%r8,%rax,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            38a4 <_sk_load_u16_be_hsw+0x148>
+  DB  116,85                              ; je            38c0 <_sk_load_u16_be_hsw+0x148>
   DB  196,65,57,22,68,64,8                ; vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            38a4 <_sk_load_u16_be_hsw+0x148>
+  DB  114,72                              ; jb            38c0 <_sk_load_u16_be_hsw+0x148>
   DB  196,193,123,16,84,64,16             ; vmovsd        0x10(%r8,%rax,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            38b1 <_sk_load_u16_be_hsw+0x155>
+  DB  116,72                              ; je            38cd <_sk_load_u16_be_hsw+0x155>
   DB  196,193,105,22,84,64,24             ; vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            38b1 <_sk_load_u16_be_hsw+0x155>
+  DB  114,59                              ; jb            38cd <_sk_load_u16_be_hsw+0x155>
   DB  196,193,123,16,92,64,32             ; vmovsd        0x20(%r8,%rax,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,6,255,255,255                ; je            378d <_sk_load_u16_be_hsw+0x31>
+  DB  15,132,6,255,255,255                ; je            37a9 <_sk_load_u16_be_hsw+0x31>
   DB  196,193,97,22,92,64,40              ; vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,245,254,255,255              ; jb            378d <_sk_load_u16_be_hsw+0x31>
+  DB  15,130,245,254,255,255              ; jb            37a9 <_sk_load_u16_be_hsw+0x31>
   DB  196,65,122,126,76,64,48             ; vmovq         0x30(%r8,%rax,2),%xmm9
-  DB  233,233,254,255,255                 ; jmpq          378d <_sk_load_u16_be_hsw+0x31>
+  DB  233,233,254,255,255                 ; jmpq          37a9 <_sk_load_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,220,254,255,255                 ; jmpq          378d <_sk_load_u16_be_hsw+0x31>
+  DB  233,220,254,255,255                 ; jmpq          37a9 <_sk_load_u16_be_hsw+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,211,254,255,255                 ; jmpq          378d <_sk_load_u16_be_hsw+0x31>
+  DB  233,211,254,255,255                 ; jmpq          37a9 <_sk_load_u16_be_hsw+0x31>
 
 PUBLIC _sk_load_rgb_u16_be_hsw
 _sk_load_rgb_u16_be_hsw LABEL PROC
@@ -3447,7 +3453,7 @@ _sk_load_rgb_u16_be_hsw LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,127                        ; lea           (%rdi,%rdi,2),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,204,0,0,0                    ; jne           3998 <_sk_load_rgb_u16_be_hsw+0xde>
+  DB  15,133,204,0,0,0                    ; jne           39b4 <_sk_load_rgb_u16_be_hsw+0xde>
   DB  196,193,122,111,4,64                ; vmovdqu       (%r8,%rax,2),%xmm0
   DB  196,193,122,111,84,64,12            ; vmovdqu       0xc(%r8,%rax,2),%xmm2
   DB  196,193,122,111,76,64,24            ; vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -3471,7 +3477,7 @@ _sk_load_rgb_u16_be_hsw LABEL PROC
   DB  197,241,235,192                     ; vpor          %xmm0,%xmm1,%xmm0
   DB  196,226,125,51,192                  ; vpmovzxwd     %xmm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,21,238,17,0,0         ; vbroadcastss  0x11ee(%rip),%ymm10        # 4b34 <_sk_callback_hsw+0x420>
+  DB  196,98,125,24,21,238,17,0,0         ; vbroadcastss  0x11ee(%rip),%ymm10        # 4b50 <_sk_callback_hsw+0x420>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,185,109,202                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -3488,48 +3494,48 @@ _sk_load_rgb_u16_be_hsw LABEL PROC
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,162,17,0,0        ; vbroadcastss  0x11a2(%rip),%ymm3        # 4b38 <_sk_callback_hsw+0x424>
+  DB  196,226,125,24,29,162,17,0,0        ; vbroadcastss  0x11a2(%rip),%ymm3        # 4b54 <_sk_callback_hsw+0x424>
   DB  255,224                             ; jmpq          *%rax
   DB  196,193,121,110,4,64                ; vmovd         (%r8,%rax,2),%xmm0
   DB  196,193,121,196,68,64,4,2           ; vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           39b1 <_sk_load_rgb_u16_be_hsw+0xf7>
-  DB  233,79,255,255,255                  ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  117,5                               ; jne           39cd <_sk_load_rgb_u16_be_hsw+0xf7>
+  DB  233,79,255,255,255                  ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
   DB  196,193,121,110,76,64,6             ; vmovd         0x6(%r8,%rax,2),%xmm1
   DB  196,65,113,196,68,64,10,2           ; vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            39e0 <_sk_load_rgb_u16_be_hsw+0x126>
+  DB  114,26                              ; jb            39fc <_sk_load_rgb_u16_be_hsw+0x126>
   DB  196,193,121,110,76,64,12            ; vmovd         0xc(%r8,%rax,2),%xmm1
   DB  196,193,113,196,84,64,16,2          ; vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           39e5 <_sk_load_rgb_u16_be_hsw+0x12b>
-  DB  233,32,255,255,255                  ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
-  DB  233,27,255,255,255                  ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           3a01 <_sk_load_rgb_u16_be_hsw+0x12b>
+  DB  233,32,255,255,255                  ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,27,255,255,255                  ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
   DB  196,193,121,110,76,64,18            ; vmovd         0x12(%r8,%rax,2),%xmm1
   DB  196,65,113,196,76,64,22,2           ; vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            3a14 <_sk_load_rgb_u16_be_hsw+0x15a>
+  DB  114,26                              ; jb            3a30 <_sk_load_rgb_u16_be_hsw+0x15a>
   DB  196,193,121,110,76,64,24            ; vmovd         0x18(%r8,%rax,2),%xmm1
   DB  196,193,113,196,76,64,28,2          ; vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           3a19 <_sk_load_rgb_u16_be_hsw+0x15f>
-  DB  233,236,254,255,255                 ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
-  DB  233,231,254,255,255                 ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  117,10                              ; jne           3a35 <_sk_load_rgb_u16_be_hsw+0x15f>
+  DB  233,236,254,255,255                 ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,231,254,255,255                 ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
   DB  196,193,121,110,92,64,30            ; vmovd         0x1e(%r8,%rax,2),%xmm3
   DB  196,65,97,196,92,64,34,2            ; vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            3a42 <_sk_load_rgb_u16_be_hsw+0x188>
+  DB  114,20                              ; jb            3a5e <_sk_load_rgb_u16_be_hsw+0x188>
   DB  196,193,121,110,92,64,36            ; vmovd         0x24(%r8,%rax,2),%xmm3
   DB  196,193,97,196,92,64,40,2           ; vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  DB  233,190,254,255,255                 ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
-  DB  233,185,254,255,255                 ; jmpq          3900 <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,190,254,255,255                 ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
+  DB  233,185,254,255,255                 ; jmpq          391c <_sk_load_rgb_u16_be_hsw+0x46>
 
 PUBLIC _sk_store_u16_be_hsw
 _sk_store_u16_be_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
-  DB  196,98,125,24,5,223,16,0,0          ; vbroadcastss  0x10df(%rip),%ymm8        # 4b3c <_sk_callback_hsw+0x428>
+  DB  196,98,125,24,5,223,16,0,0          ; vbroadcastss  0x10df(%rip),%ymm8        # 4b58 <_sk_callback_hsw+0x428>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,67,125,25,202,1                 ; vextractf128  $0x1,%ymm9,%xmm10
@@ -3567,7 +3573,7 @@ _sk_store_u16_be_hsw LABEL PROC
   DB  196,65,17,98,200                    ; vpunpckldq    %xmm8,%xmm13,%xmm9
   DB  196,65,17,106,192                   ; vpunpckhdq    %xmm8,%xmm13,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,31                              ; jne           3b41 <_sk_store_u16_be_hsw+0xfa>
+  DB  117,31                              ; jne           3b5d <_sk_store_u16_be_hsw+0xfa>
   DB  196,65,120,17,28,64                 ; vmovups       %xmm11,(%r8,%rax,2)
   DB  196,65,120,17,84,64,16              ; vmovups       %xmm10,0x10(%r8,%rax,2)
   DB  196,65,120,17,76,64,32              ; vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -3576,31 +3582,31 @@ _sk_store_u16_be_hsw LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,214,28,64                ; vmovq         %xmm11,(%r8,%rax,2)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  116,240                             ; je            3b59 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,23,92,64,8               ; vmovhpd       %xmm11,0x8(%r8,%rax,2)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  114,227                             ; jb            3b59 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,214,84,64,16             ; vmovq         %xmm10,0x10(%r8,%rax,2)
-  DB  116,218                             ; je            3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  116,218                             ; je            3b59 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,23,84,64,24              ; vmovhpd       %xmm10,0x18(%r8,%rax,2)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  114,205                             ; jb            3b59 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,214,76,64,32             ; vmovq         %xmm9,0x20(%r8,%rax,2)
-  DB  116,196                             ; je            3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  116,196                             ; je            3b59 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,23,76,64,40              ; vmovhpd       %xmm9,0x28(%r8,%rax,2)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,183                             ; jb            3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  114,183                             ; jb            3b59 <_sk_store_u16_be_hsw+0xf6>
   DB  196,65,121,214,68,64,48             ; vmovq         %xmm8,0x30(%r8,%rax,2)
-  DB  235,174                             ; jmp           3b3d <_sk_store_u16_be_hsw+0xf6>
+  DB  235,174                             ; jmp           3b59 <_sk_store_u16_be_hsw+0xf6>
 
 PUBLIC _sk_load_f32_hsw
 _sk_load_f32_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  119,110                             ; ja            3c05 <_sk_load_f32_hsw+0x76>
+  DB  119,110                             ; ja            3c21 <_sk_load_f32_hsw+0x76>
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
-  DB  76,141,21,135,0,0,0                 ; lea           0x87(%rip),%r10        # 3c30 <_sk_load_f32_hsw+0xa1>
+  DB  76,141,21,135,0,0,0                 ; lea           0x87(%rip),%r10        # 3c4c <_sk_load_f32_hsw+0xa1>
   DB  73,99,4,138                         ; movslq        (%r10,%rcx,4),%rax
   DB  76,1,208                            ; add           %r10,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -3659,7 +3665,7 @@ _sk_store_f32_hsw LABEL PROC
   DB  196,65,37,20,196                    ; vunpcklpd     %ymm12,%ymm11,%ymm8
   DB  196,65,37,21,220                    ; vunpckhpd     %ymm12,%ymm11,%ymm11
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,55                              ; jne           3cbd <_sk_store_f32_hsw+0x6d>
+  DB  117,55                              ; jne           3cd9 <_sk_store_f32_hsw+0x6d>
   DB  196,67,45,24,225,1                  ; vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   DB  196,67,61,24,235,1                  ; vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   DB  196,67,45,6,201,49                  ; vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -3672,22 +3678,22 @@ _sk_store_f32_hsw LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,17,20,128                ; vmovupd       %xmm10,(%r8,%rax,4)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            3cb9 <_sk_store_f32_hsw+0x69>
+  DB  116,240                             ; je            3cd5 <_sk_store_f32_hsw+0x69>
   DB  196,65,121,17,76,128,16             ; vmovupd       %xmm9,0x10(%r8,%rax,4)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            3cb9 <_sk_store_f32_hsw+0x69>
+  DB  114,227                             ; jb            3cd5 <_sk_store_f32_hsw+0x69>
   DB  196,65,121,17,68,128,32             ; vmovupd       %xmm8,0x20(%r8,%rax,4)
-  DB  116,218                             ; je            3cb9 <_sk_store_f32_hsw+0x69>
+  DB  116,218                             ; je            3cd5 <_sk_store_f32_hsw+0x69>
   DB  196,65,121,17,92,128,48             ; vmovupd       %xmm11,0x30(%r8,%rax,4)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            3cb9 <_sk_store_f32_hsw+0x69>
+  DB  114,205                             ; jb            3cd5 <_sk_store_f32_hsw+0x69>
   DB  196,67,125,25,84,128,64,1           ; vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  DB  116,195                             ; je            3cb9 <_sk_store_f32_hsw+0x69>
+  DB  116,195                             ; je            3cd5 <_sk_store_f32_hsw+0x69>
   DB  196,67,125,25,76,128,80,1           ; vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,181                             ; jb            3cb9 <_sk_store_f32_hsw+0x69>
+  DB  114,181                             ; jb            3cd5 <_sk_store_f32_hsw+0x69>
   DB  196,67,125,25,68,128,96,1           ; vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  DB  235,171                             ; jmp           3cb9 <_sk_store_f32_hsw+0x69>
+  DB  235,171                             ; jmp           3cd5 <_sk_store_f32_hsw+0x69>
 
 PUBLIC _sk_clamp_x_hsw
 _sk_clamp_x_hsw LABEL PROC
@@ -3771,11 +3777,11 @@ _sk_mirror_y_hsw LABEL PROC
 
 PUBLIC _sk_luminance_to_alpha_hsw
 _sk_luminance_to_alpha_hsw LABEL PROC
-  DB  196,226,125,24,29,47,13,0,0         ; vbroadcastss  0xd2f(%rip),%ymm3        # 4b40 <_sk_callback_hsw+0x42c>
-  DB  196,98,125,24,5,42,13,0,0           ; vbroadcastss  0xd2a(%rip),%ymm8        # 4b44 <_sk_callback_hsw+0x430>
+  DB  196,226,125,24,29,47,13,0,0         ; vbroadcastss  0xd2f(%rip),%ymm3        # 4b5c <_sk_callback_hsw+0x42c>
+  DB  196,98,125,24,5,42,13,0,0           ; vbroadcastss  0xd2a(%rip),%ymm8        # 4b60 <_sk_callback_hsw+0x430>
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  196,226,125,184,203                 ; vfmadd231ps   %ymm3,%ymm0,%ymm1
-  DB  196,226,125,24,29,27,13,0,0         ; vbroadcastss  0xd1b(%rip),%ymm3        # 4b48 <_sk_callback_hsw+0x434>
+  DB  196,226,125,24,29,27,13,0,0         ; vbroadcastss  0xd1b(%rip),%ymm3        # 4b64 <_sk_callback_hsw+0x434>
   DB  196,226,109,168,217                 ; vfmadd213ps   %ymm1,%ymm2,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -3908,9 +3914,9 @@ _sk_evenly_spaced_gradient_hsw LABEL PROC
   DB  76,139,64,8                         ; mov           0x8(%rax),%r8
   DB  77,137,202                          ; mov           %r9,%r10
   DB  73,255,202                          ; dec           %r10
-  DB  120,7                               ; js            4068 <_sk_evenly_spaced_gradient_hsw+0x18>
+  DB  120,7                               ; js            4084 <_sk_evenly_spaced_gradient_hsw+0x18>
   DB  196,193,242,42,202                  ; vcvtsi2ss     %r10,%xmm1,%xmm1
-  DB  235,22                              ; jmp           407e <_sk_evenly_spaced_gradient_hsw+0x2e>
+  DB  235,22                              ; jmp           409a <_sk_evenly_spaced_gradient_hsw+0x2e>
   DB  77,137,211                          ; mov           %r10,%r11
   DB  73,209,235                          ; shr           %r11
   DB  65,131,226,1                        ; and           $0x1,%r10d
@@ -3921,7 +3927,7 @@ _sk_evenly_spaced_gradient_hsw LABEL PROC
   DB  197,244,89,200                      ; vmulps        %ymm0,%ymm1,%ymm1
   DB  197,126,91,217                      ; vcvttps2dq    %ymm1,%ymm11
   DB  73,131,249,8                        ; cmp           $0x8,%r9
-  DB  119,70                              ; ja            40d7 <_sk_evenly_spaced_gradient_hsw+0x87>
+  DB  119,70                              ; ja            40f3 <_sk_evenly_spaced_gradient_hsw+0x87>
   DB  196,66,37,22,0                      ; vpermps       (%r8),%ymm11,%ymm8
   DB  76,139,64,40                        ; mov           0x28(%rax),%r8
   DB  196,66,37,22,8                      ; vpermps       (%r8),%ymm11,%ymm9
@@ -3937,7 +3943,7 @@ _sk_evenly_spaced_gradient_hsw LABEL PROC
   DB  196,194,37,22,24                    ; vpermps       (%r8),%ymm11,%ymm3
   DB  72,139,64,64                        ; mov           0x40(%rax),%rax
   DB  196,98,37,22,40                     ; vpermps       (%rax),%ymm11,%ymm13
-  DB  235,110                             ; jmp           4145 <_sk_evenly_spaced_gradient_hsw+0xf5>
+  DB  235,110                             ; jmp           4161 <_sk_evenly_spaced_gradient_hsw+0xf5>
   DB  196,65,13,118,246                   ; vpcmpeqd      %ymm14,%ymm14,%ymm14
   DB  197,245,118,201                     ; vpcmpeqd      %ymm1,%ymm1,%ymm1
   DB  196,2,117,146,4,152                 ; vgatherdps    %ymm1,(%r8,%ymm11,4),%ymm8
@@ -3974,11 +3980,11 @@ _sk_gradient_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  73,131,248,1                        ; cmp           $0x1,%r8
-  DB  15,134,180,0,0,0                    ; jbe           4224 <_sk_gradient_hsw+0xc3>
+  DB  15,134,180,0,0,0                    ; jbe           4240 <_sk_gradient_hsw+0xc3>
   DB  76,139,72,72                        ; mov           0x48(%rax),%r9
   DB  197,244,87,201                      ; vxorps        %ymm1,%ymm1,%ymm1
   DB  65,186,1,0,0,0                      ; mov           $0x1,%r10d
-  DB  196,226,125,24,21,197,9,0,0         ; vbroadcastss  0x9c5(%rip),%ymm2        # 4b4c <_sk_callback_hsw+0x438>
+  DB  196,226,125,24,21,197,9,0,0         ; vbroadcastss  0x9c5(%rip),%ymm2        # 4b68 <_sk_callback_hsw+0x438>
   DB  196,65,53,239,201                   ; vpxor         %ymm9,%ymm9,%ymm9
   DB  196,130,125,24,28,145               ; vbroadcastss  (%r9,%r10,4),%ymm3
   DB  197,228,194,216,2                   ; vcmpleps      %ymm0,%ymm3,%ymm3
@@ -3986,10 +3992,10 @@ _sk_gradient_hsw LABEL PROC
   DB  196,65,101,254,201                  ; vpaddd        %ymm9,%ymm3,%ymm9
   DB  73,255,194                          ; inc           %r10
   DB  77,57,208                           ; cmp           %r10,%r8
-  DB  117,226                             ; jne           418c <_sk_gradient_hsw+0x2b>
+  DB  117,226                             ; jne           41a8 <_sk_gradient_hsw+0x2b>
   DB  76,139,72,8                         ; mov           0x8(%rax),%r9
   DB  73,131,248,8                        ; cmp           $0x8,%r8
-  DB  118,121                             ; jbe           422d <_sk_gradient_hsw+0xcc>
+  DB  118,121                             ; jbe           4249 <_sk_gradient_hsw+0xcc>
   DB  196,65,13,118,246                   ; vpcmpeqd      %ymm14,%ymm14,%ymm14
   DB  197,245,118,201                     ; vpcmpeqd      %ymm1,%ymm1,%ymm1
   DB  196,2,117,146,4,137                 ; vgatherdps    %ymm1,(%r9,%ymm9,4),%ymm8
@@ -4013,7 +4019,7 @@ _sk_gradient_hsw LABEL PROC
   DB  196,130,21,146,28,136               ; vgatherdps    %ymm13,(%r8,%ymm9,4),%ymm3
   DB  72,139,64,64                        ; mov           0x40(%rax),%rax
   DB  196,34,13,146,44,136                ; vgatherdps    %ymm14,(%rax,%ymm9,4),%ymm13
-  DB  235,77                              ; jmp           4271 <_sk_gradient_hsw+0x110>
+  DB  235,77                              ; jmp           428d <_sk_gradient_hsw+0x110>
   DB  76,139,72,8                         ; mov           0x8(%rax),%r9
   DB  196,65,52,87,201                    ; vxorps        %ymm9,%ymm9,%ymm9
   DB  196,66,53,22,1                      ; vpermps       (%r9),%ymm9,%ymm8
@@ -4069,24 +4075,24 @@ _sk_xy_to_unit_angle_hsw LABEL PROC
   DB  196,65,52,95,226                    ; vmaxps        %ymm10,%ymm9,%ymm12
   DB  196,65,36,94,220                    ; vdivps        %ymm12,%ymm11,%ymm11
   DB  196,65,36,89,227                    ; vmulps        %ymm11,%ymm11,%ymm12
-  DB  196,98,125,24,45,68,8,0,0           ; vbroadcastss  0x844(%rip),%ymm13        # 4b50 <_sk_callback_hsw+0x43c>
-  DB  196,98,125,24,53,63,8,0,0           ; vbroadcastss  0x83f(%rip),%ymm14        # 4b54 <_sk_callback_hsw+0x440>
+  DB  196,98,125,24,45,68,8,0,0           ; vbroadcastss  0x844(%rip),%ymm13        # 4b6c <_sk_callback_hsw+0x43c>
+  DB  196,98,125,24,53,63,8,0,0           ; vbroadcastss  0x83f(%rip),%ymm14        # 4b70 <_sk_callback_hsw+0x440>
   DB  196,66,29,184,245                   ; vfmadd231ps   %ymm13,%ymm12,%ymm14
-  DB  196,98,125,24,45,53,8,0,0           ; vbroadcastss  0x835(%rip),%ymm13        # 4b58 <_sk_callback_hsw+0x444>
+  DB  196,98,125,24,45,53,8,0,0           ; vbroadcastss  0x835(%rip),%ymm13        # 4b74 <_sk_callback_hsw+0x444>
   DB  196,66,29,184,238                   ; vfmadd231ps   %ymm14,%ymm12,%ymm13
-  DB  196,98,125,24,53,43,8,0,0           ; vbroadcastss  0x82b(%rip),%ymm14        # 4b5c <_sk_callback_hsw+0x448>
+  DB  196,98,125,24,53,43,8,0,0           ; vbroadcastss  0x82b(%rip),%ymm14        # 4b78 <_sk_callback_hsw+0x448>
   DB  196,66,29,184,245                   ; vfmadd231ps   %ymm13,%ymm12,%ymm14
   DB  196,65,36,89,222                    ; vmulps        %ymm14,%ymm11,%ymm11
   DB  196,65,52,194,202,1                 ; vcmpltps      %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,21,22,8,0,0           ; vbroadcastss  0x816(%rip),%ymm10        # 4b60 <_sk_callback_hsw+0x44c>
+  DB  196,98,125,24,21,22,8,0,0           ; vbroadcastss  0x816(%rip),%ymm10        # 4b7c <_sk_callback_hsw+0x44c>
   DB  196,65,44,92,211                    ; vsubps        %ymm11,%ymm10,%ymm10
   DB  196,67,37,74,202,144                ; vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   DB  196,193,124,194,192,1               ; vcmpltps      %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,21,0,8,0,0            ; vbroadcastss  0x800(%rip),%ymm10        # 4b64 <_sk_callback_hsw+0x450>
+  DB  196,98,125,24,21,0,8,0,0            ; vbroadcastss  0x800(%rip),%ymm10        # 4b80 <_sk_callback_hsw+0x450>
   DB  196,65,44,92,209                    ; vsubps        %ymm9,%ymm10,%ymm10
   DB  196,195,53,74,194,0                 ; vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   DB  196,65,116,194,200,1                ; vcmpltps      %ymm8,%ymm1,%ymm9
-  DB  196,98,125,24,21,234,7,0,0          ; vbroadcastss  0x7ea(%rip),%ymm10        # 4b68 <_sk_callback_hsw+0x454>
+  DB  196,98,125,24,21,234,7,0,0          ; vbroadcastss  0x7ea(%rip),%ymm10        # 4b84 <_sk_callback_hsw+0x454>
   DB  197,44,92,208                       ; vsubps        %ymm0,%ymm10,%ymm10
   DB  196,195,125,74,194,144              ; vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   DB  196,65,124,194,200,3                ; vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -4105,7 +4111,7 @@ _sk_xy_to_radius_hsw LABEL PROC
 PUBLIC _sk_save_xy_hsw
 _sk_save_xy_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,183,7,0,0           ; vbroadcastss  0x7b7(%rip),%ymm8        # 4b6c <_sk_callback_hsw+0x458>
+  DB  196,98,125,24,5,183,7,0,0           ; vbroadcastss  0x7b7(%rip),%ymm8        # 4b88 <_sk_callback_hsw+0x458>
   DB  196,65,124,88,200                   ; vaddps        %ymm8,%ymm0,%ymm9
   DB  196,67,125,8,209,1                  ; vroundps      $0x1,%ymm9,%ymm10
   DB  196,65,52,92,202                    ; vsubps        %ymm10,%ymm9,%ymm9
@@ -4135,9 +4141,9 @@ _sk_accumulate_hsw LABEL PROC
 PUBLIC _sk_bilinear_nx_hsw
 _sk_bilinear_nx_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,75,7,0,0           ; vbroadcastss  0x74b(%rip),%ymm0        # 4b70 <_sk_callback_hsw+0x45c>
+  DB  196,226,125,24,5,75,7,0,0           ; vbroadcastss  0x74b(%rip),%ymm0        # 4b8c <_sk_callback_hsw+0x45c>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,66,7,0,0            ; vbroadcastss  0x742(%rip),%ymm8        # 4b74 <_sk_callback_hsw+0x460>
+  DB  196,98,125,24,5,66,7,0,0            ; vbroadcastss  0x742(%rip),%ymm8        # 4b90 <_sk_callback_hsw+0x460>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4146,7 +4152,7 @@ _sk_bilinear_nx_hsw LABEL PROC
 PUBLIC _sk_bilinear_px_hsw
 _sk_bilinear_px_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,42,7,0,0           ; vbroadcastss  0x72a(%rip),%ymm0        # 4b78 <_sk_callback_hsw+0x464>
+  DB  196,226,125,24,5,42,7,0,0           ; vbroadcastss  0x72a(%rip),%ymm0        # 4b94 <_sk_callback_hsw+0x464>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -4156,9 +4162,9 @@ _sk_bilinear_px_hsw LABEL PROC
 PUBLIC _sk_bilinear_ny_hsw
 _sk_bilinear_ny_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,14,7,0,0          ; vbroadcastss  0x70e(%rip),%ymm1        # 4b7c <_sk_callback_hsw+0x468>
+  DB  196,226,125,24,13,14,7,0,0          ; vbroadcastss  0x70e(%rip),%ymm1        # 4b98 <_sk_callback_hsw+0x468>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,4,7,0,0             ; vbroadcastss  0x704(%rip),%ymm8        # 4b80 <_sk_callback_hsw+0x46c>
+  DB  196,98,125,24,5,4,7,0,0             ; vbroadcastss  0x704(%rip),%ymm8        # 4b9c <_sk_callback_hsw+0x46c>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4167,7 +4173,7 @@ _sk_bilinear_ny_hsw LABEL PROC
 PUBLIC _sk_bilinear_py_hsw
 _sk_bilinear_py_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,236,6,0,0         ; vbroadcastss  0x6ec(%rip),%ymm1        # 4b84 <_sk_callback_hsw+0x470>
+  DB  196,226,125,24,13,236,6,0,0         ; vbroadcastss  0x6ec(%rip),%ymm1        # 4ba0 <_sk_callback_hsw+0x470>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -4177,13 +4183,13 @@ _sk_bilinear_py_hsw LABEL PROC
 PUBLIC _sk_bicubic_n3x_hsw
 _sk_bicubic_n3x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,207,6,0,0          ; vbroadcastss  0x6cf(%rip),%ymm0        # 4b88 <_sk_callback_hsw+0x474>
+  DB  196,226,125,24,5,207,6,0,0          ; vbroadcastss  0x6cf(%rip),%ymm0        # 4ba4 <_sk_callback_hsw+0x474>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,198,6,0,0           ; vbroadcastss  0x6c6(%rip),%ymm8        # 4b8c <_sk_callback_hsw+0x478>
+  DB  196,98,125,24,5,198,6,0,0           ; vbroadcastss  0x6c6(%rip),%ymm8        # 4ba8 <_sk_callback_hsw+0x478>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,183,6,0,0          ; vbroadcastss  0x6b7(%rip),%ymm10        # 4b90 <_sk_callback_hsw+0x47c>
-  DB  196,98,125,24,29,178,6,0,0          ; vbroadcastss  0x6b2(%rip),%ymm11        # 4b94 <_sk_callback_hsw+0x480>
+  DB  196,98,125,24,21,183,6,0,0          ; vbroadcastss  0x6b7(%rip),%ymm10        # 4bac <_sk_callback_hsw+0x47c>
+  DB  196,98,125,24,29,178,6,0,0          ; vbroadcastss  0x6b2(%rip),%ymm11        # 4bb0 <_sk_callback_hsw+0x480>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,36,89,193                    ; vmulps        %ymm9,%ymm11,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -4193,16 +4199,16 @@ _sk_bicubic_n3x_hsw LABEL PROC
 PUBLIC _sk_bicubic_n1x_hsw
 _sk_bicubic_n1x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,149,6,0,0          ; vbroadcastss  0x695(%rip),%ymm0        # 4b98 <_sk_callback_hsw+0x484>
+  DB  196,226,125,24,5,149,6,0,0          ; vbroadcastss  0x695(%rip),%ymm0        # 4bb4 <_sk_callback_hsw+0x484>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,140,6,0,0           ; vbroadcastss  0x68c(%rip),%ymm8        # 4b9c <_sk_callback_hsw+0x488>
+  DB  196,98,125,24,5,140,6,0,0           ; vbroadcastss  0x68c(%rip),%ymm8        # 4bb8 <_sk_callback_hsw+0x488>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,130,6,0,0          ; vbroadcastss  0x682(%rip),%ymm9        # 4ba0 <_sk_callback_hsw+0x48c>
-  DB  196,98,125,24,21,125,6,0,0          ; vbroadcastss  0x67d(%rip),%ymm10        # 4ba4 <_sk_callback_hsw+0x490>
+  DB  196,98,125,24,13,130,6,0,0          ; vbroadcastss  0x682(%rip),%ymm9        # 4bbc <_sk_callback_hsw+0x48c>
+  DB  196,98,125,24,21,125,6,0,0          ; vbroadcastss  0x67d(%rip),%ymm10        # 4bc0 <_sk_callback_hsw+0x490>
   DB  196,66,61,168,209                   ; vfmadd213ps   %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,13,115,6,0,0          ; vbroadcastss  0x673(%rip),%ymm9        # 4ba8 <_sk_callback_hsw+0x494>
+  DB  196,98,125,24,13,115,6,0,0          ; vbroadcastss  0x673(%rip),%ymm9        # 4bc4 <_sk_callback_hsw+0x494>
   DB  196,66,61,184,202                   ; vfmadd231ps   %ymm10,%ymm8,%ymm9
-  DB  196,98,125,24,21,105,6,0,0          ; vbroadcastss  0x669(%rip),%ymm10        # 4bac <_sk_callback_hsw+0x498>
+  DB  196,98,125,24,21,105,6,0,0          ; vbroadcastss  0x669(%rip),%ymm10        # 4bc8 <_sk_callback_hsw+0x498>
   DB  196,66,61,184,209                   ; vfmadd231ps   %ymm9,%ymm8,%ymm10
   DB  197,124,17,144,128,0,0,0            ; vmovups       %ymm10,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4211,14 +4217,14 @@ _sk_bicubic_n1x_hsw LABEL PROC
 PUBLIC _sk_bicubic_p1x_hsw
 _sk_bicubic_p1x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,81,6,0,0            ; vbroadcastss  0x651(%rip),%ymm8        # 4bb0 <_sk_callback_hsw+0x49c>
+  DB  196,98,125,24,5,81,6,0,0            ; vbroadcastss  0x651(%rip),%ymm8        # 4bcc <_sk_callback_hsw+0x49c>
   DB  197,188,88,0                        ; vaddps        (%rax),%ymm8,%ymm0
   DB  197,124,16,72,64                    ; vmovups       0x40(%rax),%ymm9
-  DB  196,98,125,24,21,67,6,0,0           ; vbroadcastss  0x643(%rip),%ymm10        # 4bb4 <_sk_callback_hsw+0x4a0>
-  DB  196,98,125,24,29,62,6,0,0           ; vbroadcastss  0x63e(%rip),%ymm11        # 4bb8 <_sk_callback_hsw+0x4a4>
+  DB  196,98,125,24,21,67,6,0,0           ; vbroadcastss  0x643(%rip),%ymm10        # 4bd0 <_sk_callback_hsw+0x4a0>
+  DB  196,98,125,24,29,62,6,0,0           ; vbroadcastss  0x63e(%rip),%ymm11        # 4bd4 <_sk_callback_hsw+0x4a4>
   DB  196,66,53,168,218                   ; vfmadd213ps   %ymm10,%ymm9,%ymm11
   DB  196,66,53,168,216                   ; vfmadd213ps   %ymm8,%ymm9,%ymm11
-  DB  196,98,125,24,5,47,6,0,0            ; vbroadcastss  0x62f(%rip),%ymm8        # 4bbc <_sk_callback_hsw+0x4a8>
+  DB  196,98,125,24,5,47,6,0,0            ; vbroadcastss  0x62f(%rip),%ymm8        # 4bd8 <_sk_callback_hsw+0x4a8>
   DB  196,66,53,184,195                   ; vfmadd231ps   %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4227,12 +4233,12 @@ _sk_bicubic_p1x_hsw LABEL PROC
 PUBLIC _sk_bicubic_p3x_hsw
 _sk_bicubic_p3x_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,23,6,0,0           ; vbroadcastss  0x617(%rip),%ymm0        # 4bc0 <_sk_callback_hsw+0x4ac>
+  DB  196,226,125,24,5,23,6,0,0           ; vbroadcastss  0x617(%rip),%ymm0        # 4bdc <_sk_callback_hsw+0x4ac>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,4,6,0,0            ; vbroadcastss  0x604(%rip),%ymm10        # 4bc4 <_sk_callback_hsw+0x4b0>
-  DB  196,98,125,24,29,255,5,0,0          ; vbroadcastss  0x5ff(%rip),%ymm11        # 4bc8 <_sk_callback_hsw+0x4b4>
+  DB  196,98,125,24,21,4,6,0,0            ; vbroadcastss  0x604(%rip),%ymm10        # 4be0 <_sk_callback_hsw+0x4b0>
+  DB  196,98,125,24,29,255,5,0,0          ; vbroadcastss  0x5ff(%rip),%ymm11        # 4be4 <_sk_callback_hsw+0x4b4>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,52,89,195                    ; vmulps        %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -4242,13 +4248,13 @@ _sk_bicubic_p3x_hsw LABEL PROC
 PUBLIC _sk_bicubic_n3y_hsw
 _sk_bicubic_n3y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,226,5,0,0         ; vbroadcastss  0x5e2(%rip),%ymm1        # 4bcc <_sk_callback_hsw+0x4b8>
+  DB  196,226,125,24,13,226,5,0,0         ; vbroadcastss  0x5e2(%rip),%ymm1        # 4be8 <_sk_callback_hsw+0x4b8>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,216,5,0,0           ; vbroadcastss  0x5d8(%rip),%ymm8        # 4bd0 <_sk_callback_hsw+0x4bc>
+  DB  196,98,125,24,5,216,5,0,0           ; vbroadcastss  0x5d8(%rip),%ymm8        # 4bec <_sk_callback_hsw+0x4bc>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,201,5,0,0          ; vbroadcastss  0x5c9(%rip),%ymm10        # 4bd4 <_sk_callback_hsw+0x4c0>
-  DB  196,98,125,24,29,196,5,0,0          ; vbroadcastss  0x5c4(%rip),%ymm11        # 4bd8 <_sk_callback_hsw+0x4c4>
+  DB  196,98,125,24,21,201,5,0,0          ; vbroadcastss  0x5c9(%rip),%ymm10        # 4bf0 <_sk_callback_hsw+0x4c0>
+  DB  196,98,125,24,29,196,5,0,0          ; vbroadcastss  0x5c4(%rip),%ymm11        # 4bf4 <_sk_callback_hsw+0x4c4>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,36,89,193                    ; vmulps        %ymm9,%ymm11,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -4258,16 +4264,16 @@ _sk_bicubic_n3y_hsw LABEL PROC
 PUBLIC _sk_bicubic_n1y_hsw
 _sk_bicubic_n1y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,167,5,0,0         ; vbroadcastss  0x5a7(%rip),%ymm1        # 4bdc <_sk_callback_hsw+0x4c8>
+  DB  196,226,125,24,13,167,5,0,0         ; vbroadcastss  0x5a7(%rip),%ymm1        # 4bf8 <_sk_callback_hsw+0x4c8>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,157,5,0,0           ; vbroadcastss  0x59d(%rip),%ymm8        # 4be0 <_sk_callback_hsw+0x4cc>
+  DB  196,98,125,24,5,157,5,0,0           ; vbroadcastss  0x59d(%rip),%ymm8        # 4bfc <_sk_callback_hsw+0x4cc>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,147,5,0,0          ; vbroadcastss  0x593(%rip),%ymm9        # 4be4 <_sk_callback_hsw+0x4d0>
-  DB  196,98,125,24,21,142,5,0,0          ; vbroadcastss  0x58e(%rip),%ymm10        # 4be8 <_sk_callback_hsw+0x4d4>
+  DB  196,98,125,24,13,147,5,0,0          ; vbroadcastss  0x593(%rip),%ymm9        # 4c00 <_sk_callback_hsw+0x4d0>
+  DB  196,98,125,24,21,142,5,0,0          ; vbroadcastss  0x58e(%rip),%ymm10        # 4c04 <_sk_callback_hsw+0x4d4>
   DB  196,66,61,168,209                   ; vfmadd213ps   %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,13,132,5,0,0          ; vbroadcastss  0x584(%rip),%ymm9        # 4bec <_sk_callback_hsw+0x4d8>
+  DB  196,98,125,24,13,132,5,0,0          ; vbroadcastss  0x584(%rip),%ymm9        # 4c08 <_sk_callback_hsw+0x4d8>
   DB  196,66,61,184,202                   ; vfmadd231ps   %ymm10,%ymm8,%ymm9
-  DB  196,98,125,24,21,122,5,0,0          ; vbroadcastss  0x57a(%rip),%ymm10        # 4bf0 <_sk_callback_hsw+0x4dc>
+  DB  196,98,125,24,21,122,5,0,0          ; vbroadcastss  0x57a(%rip),%ymm10        # 4c0c <_sk_callback_hsw+0x4dc>
   DB  196,66,61,184,209                   ; vfmadd231ps   %ymm9,%ymm8,%ymm10
   DB  197,124,17,144,160,0,0,0            ; vmovups       %ymm10,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4276,14 +4282,14 @@ _sk_bicubic_n1y_hsw LABEL PROC
 PUBLIC _sk_bicubic_p1y_hsw
 _sk_bicubic_p1y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,98,5,0,0            ; vbroadcastss  0x562(%rip),%ymm8        # 4bf4 <_sk_callback_hsw+0x4e0>
+  DB  196,98,125,24,5,98,5,0,0            ; vbroadcastss  0x562(%rip),%ymm8        # 4c10 <_sk_callback_hsw+0x4e0>
   DB  197,188,88,72,32                    ; vaddps        0x20(%rax),%ymm8,%ymm1
   DB  197,124,16,72,96                    ; vmovups       0x60(%rax),%ymm9
-  DB  196,98,125,24,21,83,5,0,0           ; vbroadcastss  0x553(%rip),%ymm10        # 4bf8 <_sk_callback_hsw+0x4e4>
-  DB  196,98,125,24,29,78,5,0,0           ; vbroadcastss  0x54e(%rip),%ymm11        # 4bfc <_sk_callback_hsw+0x4e8>
+  DB  196,98,125,24,21,83,5,0,0           ; vbroadcastss  0x553(%rip),%ymm10        # 4c14 <_sk_callback_hsw+0x4e4>
+  DB  196,98,125,24,29,78,5,0,0           ; vbroadcastss  0x54e(%rip),%ymm11        # 4c18 <_sk_callback_hsw+0x4e8>
   DB  196,66,53,168,218                   ; vfmadd213ps   %ymm10,%ymm9,%ymm11
   DB  196,66,53,168,216                   ; vfmadd213ps   %ymm8,%ymm9,%ymm11
-  DB  196,98,125,24,5,63,5,0,0            ; vbroadcastss  0x53f(%rip),%ymm8        # 4c00 <_sk_callback_hsw+0x4ec>
+  DB  196,98,125,24,5,63,5,0,0            ; vbroadcastss  0x53f(%rip),%ymm8        # 4c1c <_sk_callback_hsw+0x4ec>
   DB  196,66,53,184,195                   ; vfmadd231ps   %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -4292,12 +4298,12 @@ _sk_bicubic_p1y_hsw LABEL PROC
 PUBLIC _sk_bicubic_p3y_hsw
 _sk_bicubic_p3y_hsw LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,39,5,0,0          ; vbroadcastss  0x527(%rip),%ymm1        # 4c04 <_sk_callback_hsw+0x4f0>
+  DB  196,226,125,24,13,39,5,0,0          ; vbroadcastss  0x527(%rip),%ymm1        # 4c20 <_sk_callback_hsw+0x4f0>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,19,5,0,0           ; vbroadcastss  0x513(%rip),%ymm10        # 4c08 <_sk_callback_hsw+0x4f4>
-  DB  196,98,125,24,29,14,5,0,0           ; vbroadcastss  0x50e(%rip),%ymm11        # 4c0c <_sk_callback_hsw+0x4f8>
+  DB  196,98,125,24,21,19,5,0,0           ; vbroadcastss  0x513(%rip),%ymm10        # 4c24 <_sk_callback_hsw+0x4f4>
+  DB  196,98,125,24,29,14,5,0,0           ; vbroadcastss  0x50e(%rip),%ymm11        # 4c28 <_sk_callback_hsw+0x4f8>
   DB  196,66,61,168,218                   ; vfmadd213ps   %ymm10,%ymm8,%ymm11
   DB  196,65,52,89,195                    ; vmulps        %ymm11,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -4411,25 +4417,25 @@ ALIGN 4
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 48e1 <.literal4+0xb1>
+  DB  71,225,61                           ; rex.RXB       loope 48fd <.literal4+0xb1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 48f1 <.literal4+0xc1>
+  DB  71,225,61                           ; rex.RXB       loope 490d <.literal4+0xc1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 4901 <.literal4+0xd1>
+  DB  71,225,61                           ; rex.RXB       loope 491d <.literal4+0xd1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 4911 <.literal4+0xe1>
+  DB  71,225,61                           ; rex.RXB       loope 492d <.literal4+0xe1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -4479,7 +4485,7 @@ ALIGN 4
   DB  190,129,128,128,59                  ; mov           $0x3b808081,%esi
   DB  129,128,128,59,0,248,0,0,8,33       ; addl          $0x21080000,-0x7ffc480(%rax)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4961 <.literal4+0x131>
+  DB  224,7                               ; loopne        497d <.literal4+0x131>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -4495,10 +4501,10 @@ ALIGN 4
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
   DB  0,52,255                            ; add           %dh,(%rdi,%rdi,8)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4988 <.literal4+0x158>
+  DB  127,0                               ; jg            49a4 <.literal4+0x158>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4a01 <.literal4+0x1d1>
+  DB  119,115                             ; ja            4a1d <.literal4+0x1d1>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4512,10 +4518,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            49bc <.literal4+0x18c>
+  DB  127,0                               ; jg            49d8 <.literal4+0x18c>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4a35 <.literal4+0x205>
+  DB  119,115                             ; ja            4a51 <.literal4+0x205>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4529,10 +4535,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            49f0 <.literal4+0x1c0>
+  DB  127,0                               ; jg            4a0c <.literal4+0x1c0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4a69 <.literal4+0x239>
+  DB  119,115                             ; ja            4a85 <.literal4+0x239>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4546,10 +4552,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4a24 <.literal4+0x1f4>
+  DB  127,0                               ; jg            4a40 <.literal4+0x1f4>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4a9d <.literal4+0x26d>
+  DB  119,115                             ; ja            4ab9 <.literal4+0x26d>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -4562,7 +4568,7 @@ ALIGN 4
   DB  0,75,0                              ; add           %cl,0x0(%rbx)
   DB  0,128,63,0,0,200                    ; add           %al,-0x37ffffc1(%rax)
   DB  66,0,0                              ; rex.X         add %al,(%rax)
-  DB  127,67                              ; jg            4a9b <.literal4+0x26b>
+  DB  127,67                              ; jg            4ab7 <.literal4+0x26b>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -4574,10 +4580,10 @@ ALIGN 4
   DB  190,80,128,3,62                     ; mov           $0x3e038050,%esi
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           4abb <.literal4+0x28b>
+  DB  118,63                              ; jbe           4ad7 <.literal4+0x28b>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            4acf <.literal4+0x29f>
+  DB  127,67                              ; jg            4aeb <.literal4+0x29f>
   DB  129,128,128,59,0,0,128,63,129,128   ; addl          $0x80813f80,0x3b80(%rax)
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,128,63,129,128,128                ; add           %al,-0x7f7f7ec1(%rax)
@@ -4586,7 +4592,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4ab1 <.literal4+0x281>
+  DB  224,7                               ; loopne        4acd <.literal4+0x281>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -4598,7 +4604,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4acd <.literal4+0x29d>
+  DB  224,7                               ; loopne        4ae9 <.literal4+0x29d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -4609,7 +4615,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            4b22 <.literal4+0x2f2>
+  DB  124,66                              ; jl            4b3e <.literal4+0x2f2>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,55,0,15                 ; mov           %ecx,0xf003788(%rax)
@@ -4627,9 +4633,9 @@ ALIGN 4
   DB  137,136,136,59,15,0                 ; mov           %ecx,0xf3b88(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,61,0,0                  ; mov           %ecx,0x3d88(%rax)
-  DB  112,65                              ; jo            4b65 <.literal4+0x335>
+  DB  112,65                              ; jo            4b81 <.literal4+0x335>
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            4b73 <.literal4+0x343>
+  DB  127,67                              ; jg            4b8f <.literal4+0x343>
   DB  128,0,128                           ; addb          $0x80,(%rax)
   DB  55                                  ; (bad)
   DB  128,0,128                           ; addb          $0x80,(%rax)
@@ -4637,7 +4643,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  255                                 ; (bad)
-  DB  127,71                              ; jg            4b87 <.literal4+0x357>
+  DB  127,71                              ; jg            4ba3 <.literal4+0x357>
   DB  208                                 ; (bad)
   DB  179,89                              ; mov           $0x59,%bl
   DB  62,89                               ; ds            pop %rcx
@@ -4737,16 +4743,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004c48 <_sk_callback_hsw+0xa000534>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004c68 <_sk_callback_hsw+0xa000538>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004c50 <_sk_callback_hsw+0x1200053c>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004c70 <_sk_callback_hsw+0x12000540>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004c58 <_sk_callback_hsw+0x1a000544>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004c78 <_sk_callback_hsw+0x1a000548>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004c60 <_sk_callback_hsw+0x300054c>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004c80 <_sk_callback_hsw+0x3000550>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4789,16 +4795,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004ca8 <_sk_callback_hsw+0xa000594>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004cc8 <_sk_callback_hsw+0xa000598>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004cb0 <_sk_callback_hsw+0x1200059c>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004cd0 <_sk_callback_hsw+0x120005a0>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004cb8 <_sk_callback_hsw+0x1a0005a4>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004cd8 <_sk_callback_hsw+0x1a0005a8>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004cc0 <_sk_callback_hsw+0x30005ac>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004ce0 <_sk_callback_hsw+0x30005b0>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4841,16 +4847,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004d08 <_sk_callback_hsw+0xa0005f4>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004d28 <_sk_callback_hsw+0xa0005f8>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004d10 <_sk_callback_hsw+0x120005fc>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004d30 <_sk_callback_hsw+0x12000600>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004d18 <_sk_callback_hsw+0x1a000604>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004d38 <_sk_callback_hsw+0x1a000608>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004d20 <_sk_callback_hsw+0x300060c>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004d40 <_sk_callback_hsw+0x3000610>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -4893,16 +4899,16 @@ ALIGN 32
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004d68 <_sk_callback_hsw+0xa000654>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004d88 <_sk_callback_hsw+0xa000658>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004d70 <_sk_callback_hsw+0x1200065c>
+  DB  255,13,255,255,255,17               ; decl          0x11ffffff(%rip)        # 12004d90 <_sk_callback_hsw+0x12000660>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004d78 <_sk_callback_hsw+0x1a000664>
+  DB  255,21,255,255,255,25               ; callq         *0x19ffffff(%rip)        # 1a004d98 <_sk_callback_hsw+0x1a000668>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004d80 <_sk_callback_hsw+0x300066c>
+  DB  255,29,255,255,255,2                ; lcall         *0x2ffffff(%rip)        # 3004da0 <_sk_callback_hsw+0x3000670>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -5044,14 +5050,14 @@ _sk_seed_shader_avx LABEL PROC
   DB  197,249,112,192,0                   ; vpshufd       $0x0,%xmm0,%xmm0
   DB  196,227,125,24,192,1                ; vinsertf128   $0x1,%xmm0,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,220,98,0,0        ; vbroadcastss  0x62dc(%rip),%ymm1        # 643c <_sk_callback_avx+0x11a>
+  DB  196,226,125,24,13,248,98,0,0        ; vbroadcastss  0x62f8(%rip),%ymm1        # 6458 <_sk_callback_avx+0x11a>
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
   DB  197,252,88,2                        ; vaddps        (%rdx),%ymm0,%ymm0
   DB  196,226,125,24,16                   ; vbroadcastss  (%rax),%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  197,236,88,201                      ; vaddps        %ymm1,%ymm2,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,21,192,98,0,0        ; vbroadcastss  0x62c0(%rip),%ymm2        # 6440 <_sk_callback_avx+0x11e>
+  DB  196,226,125,24,21,220,98,0,0        ; vbroadcastss  0x62dc(%rip),%ymm2        # 645c <_sk_callback_avx+0x11e>
   DB  197,228,87,219                      ; vxorps        %ymm3,%ymm3,%ymm3
   DB  197,220,87,228                      ; vxorps        %ymm4,%ymm4,%ymm4
   DB  197,212,87,237                      ; vxorps        %ymm5,%ymm5,%ymm5
@@ -5071,7 +5077,7 @@ _sk_dither_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  196,66,125,24,8                     ; vbroadcastss  (%r8),%ymm9
   DB  196,65,60,87,209                    ; vxorps        %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,29,120,98,0,0         ; vbroadcastss  0x6278(%rip),%ymm11        # 6444 <_sk_callback_avx+0x122>
+  DB  196,98,125,24,29,148,98,0,0         ; vbroadcastss  0x6294(%rip),%ymm11        # 6460 <_sk_callback_avx+0x122>
   DB  196,65,44,84,203                    ; vandps        %ymm11,%ymm10,%ymm9
   DB  196,193,25,114,241,5                ; vpslld        $0x5,%xmm9,%xmm12
   DB  196,67,125,25,201,1                 ; vextractf128  $0x1,%ymm9,%xmm9
@@ -5082,8 +5088,8 @@ _sk_dither_avx LABEL PROC
   DB  196,67,125,25,219,1                 ; vextractf128  $0x1,%ymm11,%xmm11
   DB  196,193,33,114,243,4                ; vpslld        $0x4,%xmm11,%xmm11
   DB  196,67,29,24,219,1                  ; vinsertf128   $0x1,%xmm11,%ymm12,%ymm11
-  DB  196,98,125,24,37,57,98,0,0          ; vbroadcastss  0x6239(%rip),%ymm12        # 6448 <_sk_callback_avx+0x126>
-  DB  196,98,125,24,45,52,98,0,0          ; vbroadcastss  0x6234(%rip),%ymm13        # 644c <_sk_callback_avx+0x12a>
+  DB  196,98,125,24,37,85,98,0,0          ; vbroadcastss  0x6255(%rip),%ymm12        # 6464 <_sk_callback_avx+0x126>
+  DB  196,98,125,24,45,80,98,0,0          ; vbroadcastss  0x6250(%rip),%ymm13        # 6468 <_sk_callback_avx+0x12a>
   DB  196,65,44,84,245                    ; vandps        %ymm13,%ymm10,%ymm14
   DB  196,193,1,114,246,2                 ; vpslld        $0x2,%xmm14,%xmm15
   DB  196,67,125,25,246,1                 ; vextractf128  $0x1,%ymm14,%xmm14
@@ -5110,15 +5116,22 @@ _sk_dither_avx LABEL PROC
   DB  196,65,60,86,193                    ; vorps         %ymm9,%ymm8,%ymm8
   DB  196,65,60,86,194                    ; vorps         %ymm10,%ymm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,159,97,0,0         ; vbroadcastss  0x619f(%rip),%ymm9        # 6450 <_sk_callback_avx+0x12e>
+  DB  196,98,125,24,13,187,97,0,0         ; vbroadcastss  0x61bb(%rip),%ymm9        # 646c <_sk_callback_avx+0x12e>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,149,97,0,0         ; vbroadcastss  0x6195(%rip),%ymm9        # 6454 <_sk_callback_avx+0x132>
+  DB  196,98,125,24,13,177,97,0,0         ; vbroadcastss  0x61b1(%rip),%ymm9        # 6470 <_sk_callback_avx+0x132>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  196,98,125,24,72,8                  ; vbroadcastss  0x8(%rax),%ymm9
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,188,88,192                      ; vaddps        %ymm0,%ymm8,%ymm0
   DB  197,188,88,201                      ; vaddps        %ymm1,%ymm8,%ymm1
   DB  197,188,88,210                      ; vaddps        %ymm2,%ymm8,%ymm2
+  DB  197,252,93,195                      ; vminps        %ymm3,%ymm0,%ymm0
+  DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
+  DB  197,188,95,192                      ; vmaxps        %ymm0,%ymm8,%ymm0
+  DB  197,244,93,203                      ; vminps        %ymm3,%ymm1,%ymm1
+  DB  197,188,95,201                      ; vmaxps        %ymm1,%ymm8,%ymm1
+  DB  197,236,93,211                      ; vminps        %ymm3,%ymm2,%ymm2
+  DB  197,188,95,210                      ; vmaxps        %ymm2,%ymm8,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -5164,7 +5177,7 @@ _sk_clear_avx LABEL PROC
 PUBLIC _sk_srcatop_avx
 _sk_srcatop_avx LABEL PROC
   DB  197,252,89,199                      ; vmulps        %ymm7,%ymm0,%ymm0
-  DB  196,98,125,24,5,9,97,0,0            ; vbroadcastss  0x6109(%rip),%ymm8        # 6458 <_sk_callback_avx+0x136>
+  DB  196,98,125,24,5,8,97,0,0            ; vbroadcastss  0x6108(%rip),%ymm8        # 6474 <_sk_callback_avx+0x136>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,204                       ; vmulps        %ymm4,%ymm8,%ymm9
   DB  197,180,88,192                      ; vaddps        %ymm0,%ymm9,%ymm0
@@ -5183,7 +5196,7 @@ _sk_srcatop_avx LABEL PROC
 PUBLIC _sk_dstatop_avx
 _sk_dstatop_avx LABEL PROC
   DB  197,100,89,196                      ; vmulps        %ymm4,%ymm3,%ymm8
-  DB  196,98,125,24,13,203,96,0,0         ; vbroadcastss  0x60cb(%rip),%ymm9        # 645c <_sk_callback_avx+0x13a>
+  DB  196,98,125,24,13,202,96,0,0         ; vbroadcastss  0x60ca(%rip),%ymm9        # 6478 <_sk_callback_avx+0x13a>
   DB  197,52,92,207                       ; vsubps        %ymm7,%ymm9,%ymm9
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
   DB  197,188,88,192                      ; vaddps        %ymm0,%ymm8,%ymm0
@@ -5219,7 +5232,7 @@ _sk_dstin_avx LABEL PROC
 
 PUBLIC _sk_srcout_avx
 _sk_srcout_avx LABEL PROC
-  DB  196,98,125,24,5,106,96,0,0          ; vbroadcastss  0x606a(%rip),%ymm8        # 6460 <_sk_callback_avx+0x13e>
+  DB  196,98,125,24,5,105,96,0,0          ; vbroadcastss  0x6069(%rip),%ymm8        # 647c <_sk_callback_avx+0x13e>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -5230,7 +5243,7 @@ _sk_srcout_avx LABEL PROC
 
 PUBLIC _sk_dstout_avx
 _sk_dstout_avx LABEL PROC
-  DB  196,226,125,24,5,77,96,0,0          ; vbroadcastss  0x604d(%rip),%ymm0        # 6464 <_sk_callback_avx+0x142>
+  DB  196,226,125,24,5,76,96,0,0          ; vbroadcastss  0x604c(%rip),%ymm0        # 6480 <_sk_callback_avx+0x142>
   DB  197,252,92,219                      ; vsubps        %ymm3,%ymm0,%ymm3
   DB  197,228,89,196                      ; vmulps        %ymm4,%ymm3,%ymm0
   DB  197,228,89,205                      ; vmulps        %ymm5,%ymm3,%ymm1
@@ -5241,7 +5254,7 @@ _sk_dstout_avx LABEL PROC
 
 PUBLIC _sk_srcover_avx
 _sk_srcover_avx LABEL PROC
-  DB  196,98,125,24,5,48,96,0,0           ; vbroadcastss  0x6030(%rip),%ymm8        # 6468 <_sk_callback_avx+0x146>
+  DB  196,98,125,24,5,47,96,0,0           ; vbroadcastss  0x602f(%rip),%ymm8        # 6484 <_sk_callback_avx+0x146>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,204                       ; vmulps        %ymm4,%ymm8,%ymm9
   DB  197,180,88,192                      ; vaddps        %ymm0,%ymm9,%ymm0
@@ -5256,7 +5269,7 @@ _sk_srcover_avx LABEL PROC
 
 PUBLIC _sk_dstover_avx
 _sk_dstover_avx LABEL PROC
-  DB  196,98,125,24,5,3,96,0,0            ; vbroadcastss  0x6003(%rip),%ymm8        # 646c <_sk_callback_avx+0x14a>
+  DB  196,98,125,24,5,2,96,0,0            ; vbroadcastss  0x6002(%rip),%ymm8        # 6488 <_sk_callback_avx+0x14a>
   DB  197,60,92,199                       ; vsubps        %ymm7,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,252,88,196                      ; vaddps        %ymm4,%ymm0,%ymm0
@@ -5280,7 +5293,7 @@ _sk_modulate_avx LABEL PROC
 
 PUBLIC _sk_multiply_avx
 _sk_multiply_avx LABEL PROC
-  DB  196,98,125,24,5,194,95,0,0          ; vbroadcastss  0x5fc2(%rip),%ymm8        # 6470 <_sk_callback_avx+0x14e>
+  DB  196,98,125,24,5,193,95,0,0          ; vbroadcastss  0x5fc1(%rip),%ymm8        # 648c <_sk_callback_avx+0x14e>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,208                       ; vmulps        %ymm0,%ymm9,%ymm10
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5334,7 +5347,7 @@ _sk_screen_avx LABEL PROC
 
 PUBLIC _sk_xor__avx
 _sk_xor__avx LABEL PROC
-  DB  196,98,125,24,5,17,95,0,0           ; vbroadcastss  0x5f11(%rip),%ymm8        # 6474 <_sk_callback_avx+0x152>
+  DB  196,98,125,24,5,16,95,0,0           ; vbroadcastss  0x5f10(%rip),%ymm8        # 6490 <_sk_callback_avx+0x152>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,180,89,192                      ; vmulps        %ymm0,%ymm9,%ymm0
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5369,7 +5382,7 @@ _sk_darken_avx LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,95,209                  ; vmaxps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,145,94,0,0          ; vbroadcastss  0x5e91(%rip),%ymm8        # 6478 <_sk_callback_avx+0x156>
+  DB  196,98,125,24,5,144,94,0,0          ; vbroadcastss  0x5e90(%rip),%ymm8        # 6494 <_sk_callback_avx+0x156>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5393,7 +5406,7 @@ _sk_lighten_avx LABEL PROC
   DB  197,100,89,206                      ; vmulps        %ymm6,%ymm3,%ymm9
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,61,94,0,0           ; vbroadcastss  0x5e3d(%rip),%ymm8        # 647c <_sk_callback_avx+0x15a>
+  DB  196,98,125,24,5,60,94,0,0           ; vbroadcastss  0x5e3c(%rip),%ymm8        # 6498 <_sk_callback_avx+0x15a>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5420,7 +5433,7 @@ _sk_difference_avx LABEL PROC
   DB  196,193,108,93,209                  ; vminps        %ymm9,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,221,93,0,0          ; vbroadcastss  0x5ddd(%rip),%ymm8        # 6480 <_sk_callback_avx+0x15e>
+  DB  196,98,125,24,5,220,93,0,0          ; vbroadcastss  0x5ddc(%rip),%ymm8        # 649c <_sk_callback_avx+0x15e>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5441,7 +5454,7 @@ _sk_exclusion_avx LABEL PROC
   DB  197,236,89,214                      ; vmulps        %ymm6,%ymm2,%ymm2
   DB  197,236,88,210                      ; vaddps        %ymm2,%ymm2,%ymm2
   DB  197,188,92,210                      ; vsubps        %ymm2,%ymm8,%ymm2
-  DB  196,98,125,24,5,152,93,0,0          ; vbroadcastss  0x5d98(%rip),%ymm8        # 6484 <_sk_callback_avx+0x162>
+  DB  196,98,125,24,5,151,93,0,0          ; vbroadcastss  0x5d97(%rip),%ymm8        # 64a0 <_sk_callback_avx+0x162>
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
   DB  197,60,89,199                       ; vmulps        %ymm7,%ymm8,%ymm8
   DB  197,188,88,219                      ; vaddps        %ymm3,%ymm8,%ymm3
@@ -5450,7 +5463,7 @@ _sk_exclusion_avx LABEL PROC
 
 PUBLIC _sk_colorburn_avx
 _sk_colorburn_avx LABEL PROC
-  DB  196,98,125,24,5,131,93,0,0          ; vbroadcastss  0x5d83(%rip),%ymm8        # 6488 <_sk_callback_avx+0x166>
+  DB  196,98,125,24,5,130,93,0,0          ; vbroadcastss  0x5d82(%rip),%ymm8        # 64a4 <_sk_callback_avx+0x166>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,52,89,216                       ; vmulps        %ymm0,%ymm9,%ymm11
   DB  196,65,44,87,210                    ; vxorps        %ymm10,%ymm10,%ymm10
@@ -5510,7 +5523,7 @@ _sk_colorburn_avx LABEL PROC
 PUBLIC _sk_colordodge_avx
 _sk_colordodge_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
-  DB  196,98,125,24,13,127,92,0,0         ; vbroadcastss  0x5c7f(%rip),%ymm9        # 648c <_sk_callback_avx+0x16a>
+  DB  196,98,125,24,13,126,92,0,0         ; vbroadcastss  0x5c7e(%rip),%ymm9        # 64a8 <_sk_callback_avx+0x16a>
   DB  197,52,92,215                       ; vsubps        %ymm7,%ymm9,%ymm10
   DB  197,44,89,216                       ; vmulps        %ymm0,%ymm10,%ymm11
   DB  197,52,92,203                       ; vsubps        %ymm3,%ymm9,%ymm9
@@ -5565,7 +5578,7 @@ _sk_colordodge_avx LABEL PROC
 
 PUBLIC _sk_hardlight_avx
 _sk_hardlight_avx LABEL PROC
-  DB  196,98,125,24,5,145,91,0,0          ; vbroadcastss  0x5b91(%rip),%ymm8        # 6490 <_sk_callback_avx+0x16e>
+  DB  196,98,125,24,5,144,91,0,0          ; vbroadcastss  0x5b90(%rip),%ymm8        # 64ac <_sk_callback_avx+0x16e>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,200                       ; vmulps        %ymm0,%ymm10,%ymm9
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5618,7 +5631,7 @@ _sk_hardlight_avx LABEL PROC
 
 PUBLIC _sk_overlay_avx
 _sk_overlay_avx LABEL PROC
-  DB  196,98,125,24,5,186,90,0,0          ; vbroadcastss  0x5aba(%rip),%ymm8        # 6494 <_sk_callback_avx+0x172>
+  DB  196,98,125,24,5,185,90,0,0          ; vbroadcastss  0x5ab9(%rip),%ymm8        # 64b0 <_sk_callback_avx+0x172>
   DB  197,60,92,215                       ; vsubps        %ymm7,%ymm8,%ymm10
   DB  197,44,89,200                       ; vmulps        %ymm0,%ymm10,%ymm9
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5683,10 +5696,10 @@ _sk_softlight_avx LABEL PROC
   DB  196,65,60,88,192                    ; vaddps        %ymm8,%ymm8,%ymm8
   DB  196,65,60,89,216                    ; vmulps        %ymm8,%ymm8,%ymm11
   DB  196,65,60,88,195                    ; vaddps        %ymm11,%ymm8,%ymm8
-  DB  196,98,125,24,29,173,89,0,0         ; vbroadcastss  0x59ad(%rip),%ymm11        # 649c <_sk_callback_avx+0x17a>
+  DB  196,98,125,24,29,172,89,0,0         ; vbroadcastss  0x59ac(%rip),%ymm11        # 64b8 <_sk_callback_avx+0x17a>
   DB  196,65,28,88,235                    ; vaddps        %ymm11,%ymm12,%ymm13
   DB  196,65,20,89,192                    ; vmulps        %ymm8,%ymm13,%ymm8
-  DB  196,98,125,24,45,158,89,0,0         ; vbroadcastss  0x599e(%rip),%ymm13        # 64a0 <_sk_callback_avx+0x17e>
+  DB  196,98,125,24,45,157,89,0,0         ; vbroadcastss  0x599d(%rip),%ymm13        # 64bc <_sk_callback_avx+0x17e>
   DB  196,65,28,89,245                    ; vmulps        %ymm13,%ymm12,%ymm14
   DB  196,65,12,88,192                    ; vaddps        %ymm8,%ymm14,%ymm8
   DB  196,65,124,82,244                   ; vrsqrtps      %ymm12,%ymm14
@@ -5697,7 +5710,7 @@ _sk_softlight_avx LABEL PROC
   DB  197,4,194,255,2                     ; vcmpleps      %ymm7,%ymm15,%ymm15
   DB  196,67,13,74,240,240                ; vblendvps     %ymm15,%ymm8,%ymm14,%ymm14
   DB  197,116,88,249                      ; vaddps        %ymm1,%ymm1,%ymm15
-  DB  196,98,125,24,5,92,89,0,0           ; vbroadcastss  0x595c(%rip),%ymm8        # 6498 <_sk_callback_avx+0x176>
+  DB  196,98,125,24,5,91,89,0,0           ; vbroadcastss  0x595b(%rip),%ymm8        # 64b4 <_sk_callback_avx+0x176>
   DB  196,65,60,92,228                    ; vsubps        %ymm12,%ymm8,%ymm12
   DB  197,132,92,195                      ; vsubps        %ymm3,%ymm15,%ymm0
   DB  196,65,124,89,228                   ; vmulps        %ymm12,%ymm0,%ymm12
@@ -5824,12 +5837,12 @@ _sk_hue_avx LABEL PROC
   DB  196,65,28,89,219                    ; vmulps        %ymm11,%ymm12,%ymm11
   DB  196,65,36,94,222                    ; vdivps        %ymm14,%ymm11,%ymm11
   DB  196,67,37,74,224,240                ; vblendvps     %ymm15,%ymm8,%ymm11,%ymm12
-  DB  196,98,125,24,53,38,87,0,0          ; vbroadcastss  0x5726(%rip),%ymm14        # 64a4 <_sk_callback_avx+0x182>
+  DB  196,98,125,24,53,37,87,0,0          ; vbroadcastss  0x5725(%rip),%ymm14        # 64c0 <_sk_callback_avx+0x182>
   DB  196,65,92,89,222                    ; vmulps        %ymm14,%ymm4,%ymm11
-  DB  196,98,125,24,61,28,87,0,0          ; vbroadcastss  0x571c(%rip),%ymm15        # 64a8 <_sk_callback_avx+0x186>
+  DB  196,98,125,24,61,27,87,0,0          ; vbroadcastss  0x571b(%rip),%ymm15        # 64c4 <_sk_callback_avx+0x186>
   DB  196,65,84,89,239                    ; vmulps        %ymm15,%ymm5,%ymm13
   DB  196,65,36,88,221                    ; vaddps        %ymm13,%ymm11,%ymm11
-  DB  196,226,125,24,5,13,87,0,0          ; vbroadcastss  0x570d(%rip),%ymm0        # 64ac <_sk_callback_avx+0x18a>
+  DB  196,226,125,24,5,12,87,0,0          ; vbroadcastss  0x570c(%rip),%ymm0        # 64c8 <_sk_callback_avx+0x18a>
   DB  197,76,89,232                       ; vmulps        %ymm0,%ymm6,%ymm13
   DB  196,65,36,88,221                    ; vaddps        %ymm13,%ymm11,%ymm11
   DB  196,65,52,89,238                    ; vmulps        %ymm14,%ymm9,%ymm13
@@ -5890,7 +5903,7 @@ _sk_hue_avx LABEL PROC
   DB  196,65,36,95,208                    ; vmaxps        %ymm8,%ymm11,%ymm10
   DB  196,195,109,74,209,240              ; vblendvps     %ymm15,%ymm9,%ymm2,%ymm2
   DB  196,193,108,95,208                  ; vmaxps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,230,85,0,0          ; vbroadcastss  0x55e6(%rip),%ymm8        # 64b0 <_sk_callback_avx+0x18e>
+  DB  196,98,125,24,5,229,85,0,0          ; vbroadcastss  0x55e5(%rip),%ymm8        # 64cc <_sk_callback_avx+0x18e>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,180,89,201                      ; vmulps        %ymm1,%ymm9,%ymm1
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -5947,12 +5960,12 @@ _sk_saturation_avx LABEL PROC
   DB  196,65,28,89,219                    ; vmulps        %ymm11,%ymm12,%ymm11
   DB  196,65,36,94,222                    ; vdivps        %ymm14,%ymm11,%ymm11
   DB  196,67,37,74,224,240                ; vblendvps     %ymm15,%ymm8,%ymm11,%ymm12
-  DB  196,98,125,24,53,238,84,0,0         ; vbroadcastss  0x54ee(%rip),%ymm14        # 64b4 <_sk_callback_avx+0x192>
+  DB  196,98,125,24,53,237,84,0,0         ; vbroadcastss  0x54ed(%rip),%ymm14        # 64d0 <_sk_callback_avx+0x192>
   DB  196,65,92,89,222                    ; vmulps        %ymm14,%ymm4,%ymm11
-  DB  196,98,125,24,61,228,84,0,0         ; vbroadcastss  0x54e4(%rip),%ymm15        # 64b8 <_sk_callback_avx+0x196>
+  DB  196,98,125,24,61,227,84,0,0         ; vbroadcastss  0x54e3(%rip),%ymm15        # 64d4 <_sk_callback_avx+0x196>
   DB  196,65,84,89,239                    ; vmulps        %ymm15,%ymm5,%ymm13
   DB  196,65,36,88,221                    ; vaddps        %ymm13,%ymm11,%ymm11
-  DB  196,226,125,24,5,213,84,0,0         ; vbroadcastss  0x54d5(%rip),%ymm0        # 64bc <_sk_callback_avx+0x19a>
+  DB  196,226,125,24,5,212,84,0,0         ; vbroadcastss  0x54d4(%rip),%ymm0        # 64d8 <_sk_callback_avx+0x19a>
   DB  197,76,89,232                       ; vmulps        %ymm0,%ymm6,%ymm13
   DB  196,65,36,88,221                    ; vaddps        %ymm13,%ymm11,%ymm11
   DB  196,65,52,89,238                    ; vmulps        %ymm14,%ymm9,%ymm13
@@ -6013,7 +6026,7 @@ _sk_saturation_avx LABEL PROC
   DB  196,65,36,95,208                    ; vmaxps        %ymm8,%ymm11,%ymm10
   DB  196,195,109,74,209,240              ; vblendvps     %ymm15,%ymm9,%ymm2,%ymm2
   DB  196,193,108,95,208                  ; vmaxps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,174,83,0,0          ; vbroadcastss  0x53ae(%rip),%ymm8        # 64c0 <_sk_callback_avx+0x19e>
+  DB  196,98,125,24,5,173,83,0,0          ; vbroadcastss  0x53ad(%rip),%ymm8        # 64dc <_sk_callback_avx+0x19e>
   DB  197,60,92,207                       ; vsubps        %ymm7,%ymm8,%ymm9
   DB  197,180,89,201                      ; vmulps        %ymm1,%ymm9,%ymm1
   DB  197,60,92,195                       ; vsubps        %ymm3,%ymm8,%ymm8
@@ -6042,12 +6055,12 @@ _sk_color_avx LABEL PROC
   DB  197,252,17,68,36,32                 ; vmovups       %ymm0,0x20(%rsp)
   DB  197,124,89,199                      ; vmulps        %ymm7,%ymm0,%ymm8
   DB  197,116,89,207                      ; vmulps        %ymm7,%ymm1,%ymm9
-  DB  196,98,125,24,45,62,83,0,0          ; vbroadcastss  0x533e(%rip),%ymm13        # 64c4 <_sk_callback_avx+0x1a2>
+  DB  196,98,125,24,45,61,83,0,0          ; vbroadcastss  0x533d(%rip),%ymm13        # 64e0 <_sk_callback_avx+0x1a2>
   DB  196,65,92,89,213                    ; vmulps        %ymm13,%ymm4,%ymm10
-  DB  196,98,125,24,53,52,83,0,0          ; vbroadcastss  0x5334(%rip),%ymm14        # 64c8 <_sk_callback_avx+0x1a6>
+  DB  196,98,125,24,53,51,83,0,0          ; vbroadcastss  0x5333(%rip),%ymm14        # 64e4 <_sk_callback_avx+0x1a6>
   DB  196,65,84,89,222                    ; vmulps        %ymm14,%ymm5,%ymm11
   DB  196,65,44,88,211                    ; vaddps        %ymm11,%ymm10,%ymm10
-  DB  196,98,125,24,61,37,83,0,0          ; vbroadcastss  0x5325(%rip),%ymm15        # 64cc <_sk_callback_avx+0x1aa>
+  DB  196,98,125,24,61,36,83,0,0          ; vbroadcastss  0x5324(%rip),%ymm15        # 64e8 <_sk_callback_avx+0x1aa>
   DB  196,65,76,89,223                    ; vmulps        %ymm15,%ymm6,%ymm11
   DB  196,193,44,88,195                   ; vaddps        %ymm11,%ymm10,%ymm0
   DB  196,65,60,89,221                    ; vmulps        %ymm13,%ymm8,%ymm11
@@ -6110,7 +6123,7 @@ _sk_color_avx LABEL PROC
   DB  196,65,44,95,207                    ; vmaxps        %ymm15,%ymm10,%ymm9
   DB  196,195,37,74,192,0                 ; vblendvps     %ymm0,%ymm8,%ymm11,%ymm0
   DB  196,65,124,95,199                   ; vmaxps        %ymm15,%ymm0,%ymm8
-  DB  196,226,125,24,5,236,81,0,0         ; vbroadcastss  0x51ec(%rip),%ymm0        # 64d0 <_sk_callback_avx+0x1ae>
+  DB  196,226,125,24,5,235,81,0,0         ; vbroadcastss  0x51eb(%rip),%ymm0        # 64ec <_sk_callback_avx+0x1ae>
   DB  197,124,92,215                      ; vsubps        %ymm7,%ymm0,%ymm10
   DB  197,172,89,84,36,32                 ; vmulps        0x20(%rsp),%ymm10,%ymm2
   DB  197,124,92,219                      ; vsubps        %ymm3,%ymm0,%ymm11
@@ -6140,12 +6153,12 @@ _sk_luminosity_avx LABEL PROC
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
   DB  197,100,89,196                      ; vmulps        %ymm4,%ymm3,%ymm8
   DB  197,100,89,205                      ; vmulps        %ymm5,%ymm3,%ymm9
-  DB  196,98,125,24,45,120,81,0,0         ; vbroadcastss  0x5178(%rip),%ymm13        # 64d4 <_sk_callback_avx+0x1b2>
+  DB  196,98,125,24,45,119,81,0,0         ; vbroadcastss  0x5177(%rip),%ymm13        # 64f0 <_sk_callback_avx+0x1b2>
   DB  196,65,108,89,213                   ; vmulps        %ymm13,%ymm2,%ymm10
-  DB  196,98,125,24,53,110,81,0,0         ; vbroadcastss  0x516e(%rip),%ymm14        # 64d8 <_sk_callback_avx+0x1b6>
+  DB  196,98,125,24,53,109,81,0,0         ; vbroadcastss  0x516d(%rip),%ymm14        # 64f4 <_sk_callback_avx+0x1b6>
   DB  196,65,116,89,222                   ; vmulps        %ymm14,%ymm1,%ymm11
   DB  196,65,44,88,211                    ; vaddps        %ymm11,%ymm10,%ymm10
-  DB  196,98,125,24,61,95,81,0,0          ; vbroadcastss  0x515f(%rip),%ymm15        # 64dc <_sk_callback_avx+0x1ba>
+  DB  196,98,125,24,61,94,81,0,0          ; vbroadcastss  0x515e(%rip),%ymm15        # 64f8 <_sk_callback_avx+0x1ba>
   DB  196,65,28,89,223                    ; vmulps        %ymm15,%ymm12,%ymm11
   DB  196,193,44,88,195                   ; vaddps        %ymm11,%ymm10,%ymm0
   DB  196,65,60,89,221                    ; vmulps        %ymm13,%ymm8,%ymm11
@@ -6208,7 +6221,7 @@ _sk_luminosity_avx LABEL PROC
   DB  196,65,44,95,207                    ; vmaxps        %ymm15,%ymm10,%ymm9
   DB  196,195,37,74,192,0                 ; vblendvps     %ymm0,%ymm8,%ymm11,%ymm0
   DB  196,65,124,95,199                   ; vmaxps        %ymm15,%ymm0,%ymm8
-  DB  196,226,125,24,5,38,80,0,0          ; vbroadcastss  0x5026(%rip),%ymm0        # 64e0 <_sk_callback_avx+0x1be>
+  DB  196,226,125,24,5,37,80,0,0          ; vbroadcastss  0x5025(%rip),%ymm0        # 64fc <_sk_callback_avx+0x1be>
   DB  197,124,92,215                      ; vsubps        %ymm7,%ymm0,%ymm10
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  197,124,92,219                      ; vsubps        %ymm3,%ymm0,%ymm11
@@ -6241,7 +6254,7 @@ _sk_clamp_0_avx LABEL PROC
 
 PUBLIC _sk_clamp_1_avx
 _sk_clamp_1_avx LABEL PROC
-  DB  196,98,125,24,5,182,79,0,0          ; vbroadcastss  0x4fb6(%rip),%ymm8        # 64e4 <_sk_callback_avx+0x1c2>
+  DB  196,98,125,24,5,181,79,0,0          ; vbroadcastss  0x4fb5(%rip),%ymm8        # 6500 <_sk_callback_avx+0x1c2>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
@@ -6251,7 +6264,7 @@ _sk_clamp_1_avx LABEL PROC
 
 PUBLIC _sk_clamp_a_avx
 _sk_clamp_a_avx LABEL PROC
-  DB  196,98,125,24,5,153,79,0,0          ; vbroadcastss  0x4f99(%rip),%ymm8        # 64e8 <_sk_callback_avx+0x1c6>
+  DB  196,98,125,24,5,152,79,0,0          ; vbroadcastss  0x4f98(%rip),%ymm8        # 6504 <_sk_callback_avx+0x1c6>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  197,252,93,195                      ; vminps        %ymm3,%ymm0,%ymm0
   DB  197,244,93,203                      ; vminps        %ymm3,%ymm1,%ymm1
@@ -6323,7 +6336,7 @@ PUBLIC _sk_unpremul_avx
 _sk_unpremul_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,65,100,194,200,0                ; vcmpeqps      %ymm8,%ymm3,%ymm9
-  DB  196,98,125,24,21,225,78,0,0         ; vbroadcastss  0x4ee1(%rip),%ymm10        # 64ec <_sk_callback_avx+0x1ca>
+  DB  196,98,125,24,21,224,78,0,0         ; vbroadcastss  0x4ee0(%rip),%ymm10        # 6508 <_sk_callback_avx+0x1ca>
   DB  197,44,94,211                       ; vdivps        %ymm3,%ymm10,%ymm10
   DB  196,67,45,74,192,144                ; vblendvps     %ymm9,%ymm8,%ymm10,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
@@ -6334,17 +6347,17 @@ _sk_unpremul_avx LABEL PROC
 
 PUBLIC _sk_from_srgb_avx
 _sk_from_srgb_avx LABEL PROC
-  DB  196,98,125,24,5,194,78,0,0          ; vbroadcastss  0x4ec2(%rip),%ymm8        # 64f0 <_sk_callback_avx+0x1ce>
+  DB  196,98,125,24,5,193,78,0,0          ; vbroadcastss  0x4ec1(%rip),%ymm8        # 650c <_sk_callback_avx+0x1ce>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  197,124,89,208                      ; vmulps        %ymm0,%ymm0,%ymm10
-  DB  196,98,125,24,29,180,78,0,0         ; vbroadcastss  0x4eb4(%rip),%ymm11        # 64f4 <_sk_callback_avx+0x1d2>
+  DB  196,98,125,24,29,179,78,0,0         ; vbroadcastss  0x4eb3(%rip),%ymm11        # 6510 <_sk_callback_avx+0x1d2>
   DB  196,65,124,89,227                   ; vmulps        %ymm11,%ymm0,%ymm12
-  DB  196,98,125,24,45,170,78,0,0         ; vbroadcastss  0x4eaa(%rip),%ymm13        # 64f8 <_sk_callback_avx+0x1d6>
+  DB  196,98,125,24,45,169,78,0,0         ; vbroadcastss  0x4ea9(%rip),%ymm13        # 6514 <_sk_callback_avx+0x1d6>
   DB  196,65,28,88,229                    ; vaddps        %ymm13,%ymm12,%ymm12
   DB  196,65,44,89,212                    ; vmulps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,37,155,78,0,0         ; vbroadcastss  0x4e9b(%rip),%ymm12        # 64fc <_sk_callback_avx+0x1da>
+  DB  196,98,125,24,37,154,78,0,0         ; vbroadcastss  0x4e9a(%rip),%ymm12        # 6518 <_sk_callback_avx+0x1da>
   DB  196,65,44,88,212                    ; vaddps        %ymm12,%ymm10,%ymm10
-  DB  196,98,125,24,53,145,78,0,0         ; vbroadcastss  0x4e91(%rip),%ymm14        # 6500 <_sk_callback_avx+0x1de>
+  DB  196,98,125,24,53,144,78,0,0         ; vbroadcastss  0x4e90(%rip),%ymm14        # 651c <_sk_callback_avx+0x1de>
   DB  196,193,124,194,198,1               ; vcmpltps      %ymm14,%ymm0,%ymm0
   DB  196,195,45,74,193,0                 ; vblendvps     %ymm0,%ymm9,%ymm10,%ymm0
   DB  196,65,116,89,200                   ; vmulps        %ymm8,%ymm1,%ymm9
@@ -6371,18 +6384,18 @@ _sk_to_srgb_avx LABEL PROC
   DB  197,124,82,192                      ; vrsqrtps      %ymm0,%ymm8
   DB  196,65,124,83,200                   ; vrcpps        %ymm8,%ymm9
   DB  196,65,124,82,208                   ; vrsqrtps      %ymm8,%ymm10
-  DB  196,98,125,24,5,28,78,0,0           ; vbroadcastss  0x4e1c(%rip),%ymm8        # 6504 <_sk_callback_avx+0x1e2>
+  DB  196,98,125,24,5,27,78,0,0           ; vbroadcastss  0x4e1b(%rip),%ymm8        # 6520 <_sk_callback_avx+0x1e2>
   DB  196,65,124,89,216                   ; vmulps        %ymm8,%ymm0,%ymm11
-  DB  196,98,125,24,37,18,78,0,0          ; vbroadcastss  0x4e12(%rip),%ymm12        # 6508 <_sk_callback_avx+0x1e6>
+  DB  196,98,125,24,37,17,78,0,0          ; vbroadcastss  0x4e11(%rip),%ymm12        # 6524 <_sk_callback_avx+0x1e6>
   DB  196,65,52,89,204                    ; vmulps        %ymm12,%ymm9,%ymm9
-  DB  196,98,125,24,45,8,78,0,0           ; vbroadcastss  0x4e08(%rip),%ymm13        # 650c <_sk_callback_avx+0x1ea>
+  DB  196,98,125,24,45,7,78,0,0           ; vbroadcastss  0x4e07(%rip),%ymm13        # 6528 <_sk_callback_avx+0x1ea>
   DB  196,65,52,88,205                    ; vaddps        %ymm13,%ymm9,%ymm9
-  DB  196,98,125,24,53,254,77,0,0         ; vbroadcastss  0x4dfe(%rip),%ymm14        # 6510 <_sk_callback_avx+0x1ee>
+  DB  196,98,125,24,53,253,77,0,0         ; vbroadcastss  0x4dfd(%rip),%ymm14        # 652c <_sk_callback_avx+0x1ee>
   DB  196,65,44,89,214                    ; vmulps        %ymm14,%ymm10,%ymm10
   DB  196,65,44,88,201                    ; vaddps        %ymm9,%ymm10,%ymm9
-  DB  196,98,125,24,21,239,77,0,0         ; vbroadcastss  0x4def(%rip),%ymm10        # 6514 <_sk_callback_avx+0x1f2>
+  DB  196,98,125,24,21,238,77,0,0         ; vbroadcastss  0x4dee(%rip),%ymm10        # 6530 <_sk_callback_avx+0x1f2>
   DB  196,65,44,93,201                    ; vminps        %ymm9,%ymm10,%ymm9
-  DB  196,98,125,24,61,229,77,0,0         ; vbroadcastss  0x4de5(%rip),%ymm15        # 6518 <_sk_callback_avx+0x1f6>
+  DB  196,98,125,24,61,228,77,0,0         ; vbroadcastss  0x4de4(%rip),%ymm15        # 6534 <_sk_callback_avx+0x1f6>
   DB  196,193,124,194,199,1               ; vcmpltps      %ymm15,%ymm0,%ymm0
   DB  196,195,53,74,195,0                 ; vblendvps     %ymm0,%ymm11,%ymm9,%ymm0
   DB  197,124,82,201                      ; vrsqrtps      %ymm1,%ymm9
@@ -6417,7 +6430,7 @@ _sk_rgb_to_hsl_avx LABEL PROC
   DB  197,124,93,201                      ; vminps        %ymm1,%ymm0,%ymm9
   DB  197,52,93,202                       ; vminps        %ymm2,%ymm9,%ymm9
   DB  196,65,60,92,209                    ; vsubps        %ymm9,%ymm8,%ymm10
-  DB  196,98,125,24,29,75,77,0,0          ; vbroadcastss  0x4d4b(%rip),%ymm11        # 651c <_sk_callback_avx+0x1fa>
+  DB  196,98,125,24,29,74,77,0,0          ; vbroadcastss  0x4d4a(%rip),%ymm11        # 6538 <_sk_callback_avx+0x1fa>
   DB  196,65,36,94,218                    ; vdivps        %ymm10,%ymm11,%ymm11
   DB  197,116,92,226                      ; vsubps        %ymm2,%ymm1,%ymm12
   DB  196,65,28,89,227                    ; vmulps        %ymm11,%ymm12,%ymm12
@@ -6427,19 +6440,19 @@ _sk_rgb_to_hsl_avx LABEL PROC
   DB  196,193,108,89,211                  ; vmulps        %ymm11,%ymm2,%ymm2
   DB  197,252,92,201                      ; vsubps        %ymm1,%ymm0,%ymm1
   DB  196,193,116,89,203                  ; vmulps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,36,77,0,0          ; vbroadcastss  0x4d24(%rip),%ymm11        # 6528 <_sk_callback_avx+0x206>
+  DB  196,98,125,24,29,35,77,0,0          ; vbroadcastss  0x4d23(%rip),%ymm11        # 6544 <_sk_callback_avx+0x206>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,18,77,0,0          ; vbroadcastss  0x4d12(%rip),%ymm11        # 6524 <_sk_callback_avx+0x202>
+  DB  196,98,125,24,29,17,77,0,0          ; vbroadcastss  0x4d11(%rip),%ymm11        # 6540 <_sk_callback_avx+0x202>
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
   DB  196,227,117,74,202,224              ; vblendvps     %ymm14,%ymm2,%ymm1,%ymm1
-  DB  196,226,125,24,21,250,76,0,0        ; vbroadcastss  0x4cfa(%rip),%ymm2        # 6520 <_sk_callback_avx+0x1fe>
+  DB  196,226,125,24,21,249,76,0,0        ; vbroadcastss  0x4cf9(%rip),%ymm2        # 653c <_sk_callback_avx+0x1fe>
   DB  196,65,12,87,246                    ; vxorps        %ymm14,%ymm14,%ymm14
   DB  196,227,13,74,210,208               ; vblendvps     %ymm13,%ymm2,%ymm14,%ymm2
   DB  197,188,194,192,0                   ; vcmpeqps      %ymm0,%ymm8,%ymm0
   DB  196,193,108,88,212                  ; vaddps        %ymm12,%ymm2,%ymm2
   DB  196,227,117,74,194,0                ; vblendvps     %ymm0,%ymm2,%ymm1,%ymm0
   DB  196,193,60,88,201                   ; vaddps        %ymm9,%ymm8,%ymm1
-  DB  196,98,125,24,37,225,76,0,0         ; vbroadcastss  0x4ce1(%rip),%ymm12        # 6530 <_sk_callback_avx+0x20e>
+  DB  196,98,125,24,37,224,76,0,0         ; vbroadcastss  0x4ce0(%rip),%ymm12        # 654c <_sk_callback_avx+0x20e>
   DB  196,193,116,89,212                  ; vmulps        %ymm12,%ymm1,%ymm2
   DB  197,28,194,226,1                    ; vcmpltps      %ymm2,%ymm12,%ymm12
   DB  196,65,36,92,216                    ; vsubps        %ymm8,%ymm11,%ymm11
@@ -6449,7 +6462,7 @@ _sk_rgb_to_hsl_avx LABEL PROC
   DB  197,172,94,201                      ; vdivps        %ymm1,%ymm10,%ymm1
   DB  196,195,125,74,198,128              ; vblendvps     %ymm8,%ymm14,%ymm0,%ymm0
   DB  196,195,117,74,206,128              ; vblendvps     %ymm8,%ymm14,%ymm1,%ymm1
-  DB  196,98,125,24,5,164,76,0,0          ; vbroadcastss  0x4ca4(%rip),%ymm8        # 652c <_sk_callback_avx+0x20a>
+  DB  196,98,125,24,5,163,76,0,0          ; vbroadcastss  0x4ca3(%rip),%ymm8        # 6548 <_sk_callback_avx+0x20a>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -6464,7 +6477,7 @@ _sk_hsl_to_rgb_avx LABEL PROC
   DB  197,252,17,28,36                    ; vmovups       %ymm3,(%rsp)
   DB  197,252,40,225                      ; vmovaps       %ymm1,%ymm4
   DB  197,252,40,216                      ; vmovaps       %ymm0,%ymm3
-  DB  196,98,125,24,5,107,76,0,0          ; vbroadcastss  0x4c6b(%rip),%ymm8        # 6534 <_sk_callback_avx+0x212>
+  DB  196,98,125,24,5,106,76,0,0          ; vbroadcastss  0x4c6a(%rip),%ymm8        # 6550 <_sk_callback_avx+0x212>
   DB  197,60,194,202,2                    ; vcmpleps      %ymm2,%ymm8,%ymm9
   DB  197,92,89,210                       ; vmulps        %ymm2,%ymm4,%ymm10
   DB  196,65,92,92,218                    ; vsubps        %ymm10,%ymm4,%ymm11
@@ -6472,23 +6485,23 @@ _sk_hsl_to_rgb_avx LABEL PROC
   DB  197,52,88,210                       ; vaddps        %ymm2,%ymm9,%ymm10
   DB  197,108,88,202                      ; vaddps        %ymm2,%ymm2,%ymm9
   DB  196,65,52,92,202                    ; vsubps        %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,29,69,76,0,0          ; vbroadcastss  0x4c45(%rip),%ymm11        # 6538 <_sk_callback_avx+0x216>
+  DB  196,98,125,24,29,68,76,0,0          ; vbroadcastss  0x4c44(%rip),%ymm11        # 6554 <_sk_callback_avx+0x216>
   DB  196,65,100,88,219                   ; vaddps        %ymm11,%ymm3,%ymm11
   DB  196,67,125,8,227,1                  ; vroundps      $0x1,%ymm11,%ymm12
   DB  196,65,36,92,252                    ; vsubps        %ymm12,%ymm11,%ymm15
   DB  196,65,44,92,217                    ; vsubps        %ymm9,%ymm10,%ymm11
-  DB  196,98,125,24,37,47,76,0,0          ; vbroadcastss  0x4c2f(%rip),%ymm12        # 6540 <_sk_callback_avx+0x21e>
+  DB  196,98,125,24,37,46,76,0,0          ; vbroadcastss  0x4c2e(%rip),%ymm12        # 655c <_sk_callback_avx+0x21e>
   DB  196,193,4,89,196                    ; vmulps        %ymm12,%ymm15,%ymm0
-  DB  196,98,125,24,45,37,76,0,0          ; vbroadcastss  0x4c25(%rip),%ymm13        # 6544 <_sk_callback_avx+0x222>
+  DB  196,98,125,24,45,36,76,0,0          ; vbroadcastss  0x4c24(%rip),%ymm13        # 6560 <_sk_callback_avx+0x222>
   DB  197,20,92,240                       ; vsubps        %ymm0,%ymm13,%ymm14
   DB  196,65,36,89,246                    ; vmulps        %ymm14,%ymm11,%ymm14
   DB  196,65,52,88,246                    ; vaddps        %ymm14,%ymm9,%ymm14
-  DB  196,226,125,24,13,6,76,0,0          ; vbroadcastss  0x4c06(%rip),%ymm1        # 653c <_sk_callback_avx+0x21a>
+  DB  196,226,125,24,13,5,76,0,0          ; vbroadcastss  0x4c05(%rip),%ymm1        # 6558 <_sk_callback_avx+0x21a>
   DB  196,193,116,194,255,2               ; vcmpleps      %ymm15,%ymm1,%ymm7
   DB  196,195,13,74,249,112               ; vblendvps     %ymm7,%ymm9,%ymm14,%ymm7
   DB  196,65,60,194,247,2                 ; vcmpleps      %ymm15,%ymm8,%ymm14
   DB  196,227,45,74,255,224               ; vblendvps     %ymm14,%ymm7,%ymm10,%ymm7
-  DB  196,98,125,24,53,241,75,0,0         ; vbroadcastss  0x4bf1(%rip),%ymm14        # 6548 <_sk_callback_avx+0x226>
+  DB  196,98,125,24,53,240,75,0,0         ; vbroadcastss  0x4bf0(%rip),%ymm14        # 6564 <_sk_callback_avx+0x226>
   DB  196,65,12,194,255,2                 ; vcmpleps      %ymm15,%ymm14,%ymm15
   DB  196,193,124,89,195                  ; vmulps        %ymm11,%ymm0,%ymm0
   DB  197,180,88,192                      ; vaddps        %ymm0,%ymm9,%ymm0
@@ -6507,7 +6520,7 @@ _sk_hsl_to_rgb_avx LABEL PROC
   DB  197,164,89,247                      ; vmulps        %ymm7,%ymm11,%ymm6
   DB  197,180,88,246                      ; vaddps        %ymm6,%ymm9,%ymm6
   DB  196,227,77,74,237,0                 ; vblendvps     %ymm0,%ymm5,%ymm6,%ymm5
-  DB  196,226,125,24,5,147,75,0,0         ; vbroadcastss  0x4b93(%rip),%ymm0        # 654c <_sk_callback_avx+0x22a>
+  DB  196,226,125,24,5,146,75,0,0         ; vbroadcastss  0x4b92(%rip),%ymm0        # 6568 <_sk_callback_avx+0x22a>
   DB  197,228,88,192                      ; vaddps        %ymm0,%ymm3,%ymm0
   DB  196,227,125,8,216,1                 ; vroundps      $0x1,%ymm0,%ymm3
   DB  197,252,92,195                      ; vsubps        %ymm3,%ymm0,%ymm0
@@ -6555,14 +6568,14 @@ _sk_scale_u8_avx LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,68                              ; jne           1ab6 <_sk_scale_u8_avx+0x54>
+  DB  117,68                              ; jne           1ad3 <_sk_scale_u8_avx+0x54>
   DB  197,122,126,0                       ; vmovq         (%rax),%xmm8
   DB  196,66,121,49,200                   ; vpmovzxbd     %xmm8,%xmm9
   DB  196,67,121,4,192,229                ; vpermilps     $0xe5,%xmm8,%xmm8
   DB  196,66,121,49,192                   ; vpmovzxbd     %xmm8,%xmm8
   DB  196,67,53,24,192,1                  ; vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,182,74,0,0         ; vbroadcastss  0x4ab6(%rip),%ymm9        # 6550 <_sk_callback_avx+0x22e>
+  DB  196,98,125,24,13,181,74,0,0         ; vbroadcastss  0x4ab5(%rip),%ymm9        # 656c <_sk_callback_avx+0x22e>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
@@ -6580,9 +6593,9 @@ _sk_scale_u8_avx LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           1abe <_sk_scale_u8_avx+0x5c>
+  DB  117,234                             ; jne           1adb <_sk_scale_u8_avx+0x5c>
   DB  196,65,249,110,193                  ; vmovq         %r9,%xmm8
-  DB  235,155                             ; jmp           1a76 <_sk_scale_u8_avx+0x14>
+  DB  235,155                             ; jmp           1a93 <_sk_scale_u8_avx+0x14>
 
 PUBLIC _sk_lerp_1_float_avx
 _sk_lerp_1_float_avx LABEL PROC
@@ -6610,14 +6623,14 @@ _sk_lerp_u8_avx LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,104                             ; jne           1b92 <_sk_lerp_u8_avx+0x78>
+  DB  117,104                             ; jne           1baf <_sk_lerp_u8_avx+0x78>
   DB  197,122,126,0                       ; vmovq         (%rax),%xmm8
   DB  196,66,121,49,200                   ; vpmovzxbd     %xmm8,%xmm9
   DB  196,67,121,4,192,229                ; vpermilps     $0xe5,%xmm8,%xmm8
   DB  196,66,121,49,192                   ; vpmovzxbd     %xmm8,%xmm8
   DB  196,67,53,24,192,1                  ; vinsertf128   $0x1,%xmm8,%ymm9,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,13,2,74,0,0           ; vbroadcastss  0x4a02(%rip),%ymm9        # 6554 <_sk_callback_avx+0x232>
+  DB  196,98,125,24,13,1,74,0,0           ; vbroadcastss  0x4a01(%rip),%ymm9        # 6570 <_sk_callback_avx+0x232>
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
@@ -6643,35 +6656,35 @@ _sk_lerp_u8_avx LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           1b9a <_sk_lerp_u8_avx+0x80>
+  DB  117,234                             ; jne           1bb7 <_sk_lerp_u8_avx+0x80>
   DB  196,65,249,110,193                  ; vmovq         %r9,%xmm8
-  DB  233,116,255,255,255                 ; jmpq          1b2e <_sk_lerp_u8_avx+0x14>
+  DB  233,116,255,255,255                 ; jmpq          1b4b <_sk_lerp_u8_avx+0x14>
 
 PUBLIC _sk_lerp_565_avx
 _sk_lerp_565_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,208,0,0,0                    ; jne           1c98 <_sk_lerp_565_avx+0xde>
+  DB  15,133,208,0,0,0                    ; jne           1cb5 <_sk_lerp_565_avx+0xde>
   DB  196,65,122,111,4,122                ; vmovdqu       (%r10,%rdi,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  196,65,57,105,201                   ; vpunpckhwd    %xmm9,%xmm8,%xmm9
   DB  196,66,121,51,192                   ; vpmovzxwd     %xmm8,%xmm8
   DB  196,67,61,24,193,1                  ; vinsertf128   $0x1,%xmm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,108,73,0,0         ; vbroadcastss  0x496c(%rip),%ymm9        # 6558 <_sk_callback_avx+0x236>
+  DB  196,98,125,24,13,107,73,0,0         ; vbroadcastss  0x496b(%rip),%ymm9        # 6574 <_sk_callback_avx+0x236>
   DB  196,65,60,84,201                    ; vandps        %ymm9,%ymm8,%ymm9
   DB  196,65,124,91,201                   ; vcvtdq2ps     %ymm9,%ymm9
-  DB  196,98,125,24,21,93,73,0,0          ; vbroadcastss  0x495d(%rip),%ymm10        # 655c <_sk_callback_avx+0x23a>
+  DB  196,98,125,24,21,92,73,0,0          ; vbroadcastss  0x495c(%rip),%ymm10        # 6578 <_sk_callback_avx+0x23a>
   DB  196,65,52,89,202                    ; vmulps        %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,21,83,73,0,0          ; vbroadcastss  0x4953(%rip),%ymm10        # 6560 <_sk_callback_avx+0x23e>
+  DB  196,98,125,24,21,82,73,0,0          ; vbroadcastss  0x4952(%rip),%ymm10        # 657c <_sk_callback_avx+0x23e>
   DB  196,65,60,84,210                    ; vandps        %ymm10,%ymm8,%ymm10
   DB  196,65,124,91,210                   ; vcvtdq2ps     %ymm10,%ymm10
-  DB  196,98,125,24,29,68,73,0,0          ; vbroadcastss  0x4944(%rip),%ymm11        # 6564 <_sk_callback_avx+0x242>
+  DB  196,98,125,24,29,67,73,0,0          ; vbroadcastss  0x4943(%rip),%ymm11        # 6580 <_sk_callback_avx+0x242>
   DB  196,65,44,89,211                    ; vmulps        %ymm11,%ymm10,%ymm10
-  DB  196,98,125,24,29,58,73,0,0          ; vbroadcastss  0x493a(%rip),%ymm11        # 6568 <_sk_callback_avx+0x246>
+  DB  196,98,125,24,29,57,73,0,0          ; vbroadcastss  0x4939(%rip),%ymm11        # 6584 <_sk_callback_avx+0x246>
   DB  196,65,60,84,195                    ; vandps        %ymm11,%ymm8,%ymm8
   DB  196,65,124,91,192                   ; vcvtdq2ps     %ymm8,%ymm8
-  DB  196,98,125,24,29,43,73,0,0          ; vbroadcastss  0x492b(%rip),%ymm11        # 656c <_sk_callback_avx+0x24a>
+  DB  196,98,125,24,29,42,73,0,0          ; vbroadcastss  0x492a(%rip),%ymm11        # 6588 <_sk_callback_avx+0x24a>
   DB  196,65,60,89,195                    ; vmulps        %ymm11,%ymm8,%ymm8
   DB  197,252,92,196                      ; vsubps        %ymm4,%ymm0,%ymm0
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
@@ -6698,9 +6711,9 @@ _sk_lerp_565_avx LABEL PROC
   DB  196,65,57,239,192                   ; vpxor         %xmm8,%xmm8,%xmm8
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,29,255,255,255               ; ja            1bce <_sk_lerp_565_avx+0x14>
+  DB  15,135,29,255,255,255               ; ja            1beb <_sk_lerp_565_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,76,0,0,0                  ; lea           0x4c(%rip),%r9        # 1d08 <_sk_lerp_565_avx+0x14e>
+  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 1d24 <_sk_lerp_565_avx+0x14d>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -6712,28 +6725,27 @@ _sk_lerp_565_avx LABEL PROC
   DB  196,65,57,196,68,122,4,2            ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,68,122,2,1            ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm8,%xmm8
   DB  196,65,57,196,4,122,0               ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm8,%xmm8
-  DB  233,200,254,255,255                 ; jmpq          1bce <_sk_lerp_565_avx+0x14>
-  DB  102,144                             ; xchg          %ax,%ax
-  DB  242,255                             ; repnz         (bad)
-  DB  255                                 ; (bad)
+  DB  233,200,254,255,255                 ; jmpq          1beb <_sk_lerp_565_avx+0x14>
+  DB  144                                 ; nop
+  DB  243,255                             ; repz          (bad)
   DB  255                                 ; (bad)
-  DB  234                                 ; (bad)
   DB  255                                 ; (bad)
+  DB  235,255                             ; jmp           1d29 <_sk_lerp_565_avx+0x152>
   DB  255                                 ; (bad)
-  DB  255,226                             ; jmpq          *%rdx
+  DB  255,227                             ; jmpq          *%rbx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  218,255                             ; (bad)
+  DB  219,255                             ; (bad)
   DB  255                                 ; (bad)
-  DB  255,210                             ; callq         *%rdx
+  DB  255,211                             ; callq         *%rbx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,202                             ; dec           %edx
+  DB  255,203                             ; dec           %ebx
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  189                                 ; .byte         0xbd
+  DB  190                                 ; .byte         0xbe
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
@@ -6743,7 +6755,7 @@ _sk_load_tables_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,26,2,0,0                     ; jne           1f4c <_sk_load_tables_avx+0x228>
+  DB  15,133,26,2,0,0                     ; jne           1f68 <_sk_load_tables_avx+0x228>
   DB  196,65,124,16,4,184                 ; vmovups       (%r8,%rdi,4),%ymm8
   DB  85                                  ; push          %rbp
   DB  65,87                               ; push          %r15
@@ -6751,7 +6763,7 @@ _sk_load_tables_avx LABEL PROC
   DB  65,85                               ; push          %r13
   DB  65,84                               ; push          %r12
   DB  83                                  ; push          %rbx
-  DB  197,124,40,13,22,75,0,0             ; vmovaps       0x4b16(%rip),%ymm9        # 6860 <_sk_callback_avx+0x53e>
+  DB  197,124,40,13,250,74,0,0            ; vmovaps       0x4afa(%rip),%ymm9        # 6860 <_sk_callback_avx+0x522>
   DB  196,193,60,84,193                   ; vandps        %ymm9,%ymm8,%ymm0
   DB  196,193,249,126,193                 ; vmovq         %xmm0,%r9
   DB  69,137,203                          ; mov           %r9d,%r11d
@@ -6843,7 +6855,7 @@ _sk_load_tables_avx LABEL PROC
   DB  196,193,97,114,210,24               ; vpsrld        $0x18,%xmm10,%xmm3
   DB  196,227,61,24,219,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,55,70,0,0           ; vbroadcastss  0x4637(%rip),%ymm8        # 6570 <_sk_callback_avx+0x24e>
+  DB  196,98,125,24,5,55,70,0,0           ; vbroadcastss  0x4637(%rip),%ymm8        # 658c <_sk_callback_avx+0x24e>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -6858,9 +6870,9 @@ _sk_load_tables_avx LABEL PROC
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  65,254,201                          ; dec           %r9b
   DB  65,128,249,6                        ; cmp           $0x6,%r9b
-  DB  15,135,211,253,255,255              ; ja            1d38 <_sk_load_tables_avx+0x14>
+  DB  15,135,211,253,255,255              ; ja            1d54 <_sk_load_tables_avx+0x14>
   DB  69,15,182,201                       ; movzbl        %r9b,%r9d
-  DB  76,141,21,140,0,0,0                 ; lea           0x8c(%rip),%r10        # 1ffc <_sk_load_tables_avx+0x2d8>
+  DB  76,141,21,140,0,0,0                 ; lea           0x8c(%rip),%r10        # 2018 <_sk_load_tables_avx+0x2d8>
   DB  79,99,12,138                        ; movslq        (%r10,%r9,4),%r9
   DB  77,1,209                            ; add           %r10,%r9
   DB  65,255,225                          ; jmpq          *%r9
@@ -6883,7 +6895,7 @@ _sk_load_tables_avx LABEL PROC
   DB  196,99,61,12,192,15                 ; vblendps      $0xf,%ymm0,%ymm8,%ymm8
   DB  196,195,57,34,4,184,0               ; vpinsrd       $0x0,(%r8,%rdi,4),%xmm8,%xmm0
   DB  196,99,61,12,192,15                 ; vblendps      $0xf,%ymm0,%ymm8,%ymm8
-  DB  233,62,253,255,255                  ; jmpq          1d38 <_sk_load_tables_avx+0x14>
+  DB  233,62,253,255,255                  ; jmpq          1d54 <_sk_load_tables_avx+0x14>
   DB  102,144                             ; xchg          %ax,%ax
   DB  236                                 ; in            (%dx),%al
   DB  255                                 ; (bad)
@@ -6901,7 +6913,7 @@ _sk_load_tables_avx LABEL PROC
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  126,255                             ; jle           2015 <_sk_load_tables_avx+0x2f1>
+  DB  126,255                             ; jle           2031 <_sk_load_tables_avx+0x2f1>
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
 
@@ -6911,7 +6923,7 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,113,2,0,0                    ; jne           229f <_sk_load_tables_u16_be_avx+0x287>
+  DB  15,133,113,2,0,0                    ; jne           22bb <_sk_load_tables_u16_be_avx+0x287>
   DB  196,1,121,16,4,72                   ; vmovupd       (%r8,%r9,2),%xmm8
   DB  196,129,121,16,84,72,16             ; vmovupd       0x10(%r8,%r9,2),%xmm2
   DB  196,129,121,16,92,72,32             ; vmovupd       0x20(%r8,%r9,2),%xmm3
@@ -6933,7 +6945,7 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  197,177,108,208                     ; vpunpcklqdq   %xmm0,%xmm9,%xmm2
   DB  197,177,109,200                     ; vpunpckhqdq   %xmm0,%xmm9,%xmm1
   DB  196,65,57,108,212                   ; vpunpcklqdq   %xmm12,%xmm8,%xmm10
-  DB  197,121,111,29,86,72,0,0            ; vmovdqa       0x4856(%rip),%xmm11        # 68e0 <_sk_callback_avx+0x5be>
+  DB  197,121,111,29,58,72,0,0            ; vmovdqa       0x483a(%rip),%xmm11        # 68e0 <_sk_callback_avx+0x5a2>
   DB  196,193,105,219,195                 ; vpand         %xmm11,%xmm2,%xmm0
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  196,193,121,105,209                 ; vpunpckhwd    %xmm9,%xmm0,%xmm2
@@ -7032,7 +7044,7 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  196,226,121,51,219                  ; vpmovzxwd     %xmm3,%xmm3
   DB  196,195,101,24,216,1                ; vinsertf128   $0x1,%xmm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,232,66,0,0          ; vbroadcastss  0x42e8(%rip),%ymm8        # 6574 <_sk_callback_avx+0x252>
+  DB  196,98,125,24,5,232,66,0,0          ; vbroadcastss  0x42e8(%rip),%ymm8        # 6590 <_sk_callback_avx+0x252>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -7045,29 +7057,29 @@ _sk_load_tables_u16_be_avx LABEL PROC
   DB  196,1,123,16,4,72                   ; vmovsd        (%r8,%r9,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            2305 <_sk_load_tables_u16_be_avx+0x2ed>
+  DB  116,85                              ; je            2321 <_sk_load_tables_u16_be_avx+0x2ed>
   DB  196,1,57,22,68,72,8                 ; vmovhpd       0x8(%r8,%r9,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            2305 <_sk_load_tables_u16_be_avx+0x2ed>
+  DB  114,72                              ; jb            2321 <_sk_load_tables_u16_be_avx+0x2ed>
   DB  196,129,123,16,84,72,16             ; vmovsd        0x10(%r8,%r9,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            2312 <_sk_load_tables_u16_be_avx+0x2fa>
+  DB  116,72                              ; je            232e <_sk_load_tables_u16_be_avx+0x2fa>
   DB  196,129,105,22,84,72,24             ; vmovhpd       0x18(%r8,%r9,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            2312 <_sk_load_tables_u16_be_avx+0x2fa>
+  DB  114,59                              ; jb            232e <_sk_load_tables_u16_be_avx+0x2fa>
   DB  196,129,123,16,92,72,32             ; vmovsd        0x20(%r8,%r9,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,97,253,255,255               ; je            2049 <_sk_load_tables_u16_be_avx+0x31>
+  DB  15,132,97,253,255,255               ; je            2065 <_sk_load_tables_u16_be_avx+0x31>
   DB  196,129,97,22,92,72,40              ; vmovhpd       0x28(%r8,%r9,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,80,253,255,255               ; jb            2049 <_sk_load_tables_u16_be_avx+0x31>
+  DB  15,130,80,253,255,255               ; jb            2065 <_sk_load_tables_u16_be_avx+0x31>
   DB  196,1,122,126,76,72,48              ; vmovq         0x30(%r8,%r9,2),%xmm9
-  DB  233,68,253,255,255                  ; jmpq          2049 <_sk_load_tables_u16_be_avx+0x31>
+  DB  233,68,253,255,255                  ; jmpq          2065 <_sk_load_tables_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,55,253,255,255                  ; jmpq          2049 <_sk_load_tables_u16_be_avx+0x31>
+  DB  233,55,253,255,255                  ; jmpq          2065 <_sk_load_tables_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,46,253,255,255                  ; jmpq          2049 <_sk_load_tables_u16_be_avx+0x31>
+  DB  233,46,253,255,255                  ; jmpq          2065 <_sk_load_tables_u16_be_avx+0x31>
 
 PUBLIC _sk_load_tables_rgb_u16_be_avx
 _sk_load_tables_rgb_u16_be_avx LABEL PROC
@@ -7075,7 +7087,7 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,127                       ; lea           (%rdi,%rdi,2),%r9
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,93,2,0,0                     ; jne           258a <_sk_load_tables_rgb_u16_be_avx+0x26f>
+  DB  15,133,93,2,0,0                     ; jne           25a6 <_sk_load_tables_rgb_u16_be_avx+0x26f>
   DB  196,129,122,111,4,72                ; vmovdqu       (%r8,%r9,2),%xmm0
   DB  196,129,122,111,84,72,12            ; vmovdqu       0xc(%r8,%r9,2),%xmm2
   DB  196,129,122,111,76,72,24            ; vmovdqu       0x18(%r8,%r9,2),%xmm1
@@ -7102,7 +7114,7 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  197,185,108,202                     ; vpunpcklqdq   %xmm2,%xmm8,%xmm1
   DB  197,185,109,210                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm2
   DB  197,121,108,195                     ; vpunpcklqdq   %xmm3,%xmm0,%xmm8
-  DB  197,121,111,13,79,69,0,0            ; vmovdqa       0x454f(%rip),%xmm9        # 68f0 <_sk_callback_avx+0x5ce>
+  DB  197,121,111,13,51,69,0,0            ; vmovdqa       0x4533(%rip),%xmm9        # 68f0 <_sk_callback_avx+0x5b2>
   DB  196,193,113,219,193                 ; vpand         %xmm9,%xmm1,%xmm0
   DB  196,65,41,239,210                   ; vpxor         %xmm10,%xmm10,%xmm10
   DB  196,193,121,105,202                 ; vpunpckhwd    %xmm10,%xmm0,%xmm1
@@ -7194,7 +7206,7 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  196,227,105,33,211,48               ; vinsertps     $0x30,%xmm3,%xmm2,%xmm2
   DB  196,195,109,24,208,1                ; vinsertf128   $0x1,%xmm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,250,63,0,0        ; vbroadcastss  0x3ffa(%rip),%ymm3        # 6578 <_sk_callback_avx+0x256>
+  DB  196,226,125,24,29,250,63,0,0        ; vbroadcastss  0x3ffa(%rip),%ymm3        # 6594 <_sk_callback_avx+0x256>
   DB  91                                  ; pop           %rbx
   DB  65,92                               ; pop           %r12
   DB  65,93                               ; pop           %r13
@@ -7205,36 +7217,36 @@ _sk_load_tables_rgb_u16_be_avx LABEL PROC
   DB  196,129,121,110,4,72                ; vmovd         (%r8,%r9,2),%xmm0
   DB  196,129,121,196,68,72,4,2           ; vpinsrw       $0x2,0x4(%r8,%r9,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           25a3 <_sk_load_tables_rgb_u16_be_avx+0x288>
-  DB  233,190,253,255,255                 ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  117,5                               ; jne           25bf <_sk_load_tables_rgb_u16_be_avx+0x288>
+  DB  233,190,253,255,255                 ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
   DB  196,129,121,110,76,72,6             ; vmovd         0x6(%r8,%r9,2),%xmm1
   DB  196,1,113,196,68,72,10,2            ; vpinsrw       $0x2,0xa(%r8,%r9,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            25d2 <_sk_load_tables_rgb_u16_be_avx+0x2b7>
+  DB  114,26                              ; jb            25ee <_sk_load_tables_rgb_u16_be_avx+0x2b7>
   DB  196,129,121,110,76,72,12            ; vmovd         0xc(%r8,%r9,2),%xmm1
   DB  196,129,113,196,84,72,16,2          ; vpinsrw       $0x2,0x10(%r8,%r9,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           25d7 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
-  DB  233,143,253,255,255                 ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  DB  233,138,253,255,255                 ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           25f3 <_sk_load_tables_rgb_u16_be_avx+0x2bc>
+  DB  233,143,253,255,255                 ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,138,253,255,255                 ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
   DB  196,129,121,110,76,72,18            ; vmovd         0x12(%r8,%r9,2),%xmm1
   DB  196,1,113,196,76,72,22,2            ; vpinsrw       $0x2,0x16(%r8,%r9,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            2606 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
+  DB  114,26                              ; jb            2622 <_sk_load_tables_rgb_u16_be_avx+0x2eb>
   DB  196,129,121,110,76,72,24            ; vmovd         0x18(%r8,%r9,2),%xmm1
   DB  196,129,113,196,76,72,28,2          ; vpinsrw       $0x2,0x1c(%r8,%r9,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           260b <_sk_load_tables_rgb_u16_be_avx+0x2f0>
-  DB  233,91,253,255,255                  ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  DB  233,86,253,255,255                  ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           2627 <_sk_load_tables_rgb_u16_be_avx+0x2f0>
+  DB  233,91,253,255,255                  ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,86,253,255,255                  ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
   DB  196,129,121,110,92,72,30            ; vmovd         0x1e(%r8,%r9,2),%xmm3
   DB  196,1,97,196,92,72,34,2             ; vpinsrw       $0x2,0x22(%r8,%r9,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            2634 <_sk_load_tables_rgb_u16_be_avx+0x319>
+  DB  114,20                              ; jb            2650 <_sk_load_tables_rgb_u16_be_avx+0x319>
   DB  196,129,121,110,92,72,36            ; vmovd         0x24(%r8,%r9,2),%xmm3
   DB  196,129,97,196,92,72,40,2           ; vpinsrw       $0x2,0x28(%r8,%r9,2),%xmm3,%xmm3
-  DB  233,45,253,255,255                  ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
-  DB  233,40,253,255,255                  ; jmpq          2361 <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,45,253,255,255                  ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
+  DB  233,40,253,255,255                  ; jmpq          237d <_sk_load_tables_rgb_u16_be_avx+0x46>
 
 PUBLIC _sk_byte_tables_avx
 _sk_byte_tables_avx LABEL PROC
@@ -7245,7 +7257,7 @@ _sk_byte_tables_avx LABEL PROC
   DB  65,84                               ; push          %r12
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,46,63,0,0           ; vbroadcastss  0x3f2e(%rip),%ymm8        # 657c <_sk_callback_avx+0x25a>
+  DB  196,98,125,24,5,46,63,0,0           ; vbroadcastss  0x3f2e(%rip),%ymm8        # 6598 <_sk_callback_avx+0x25a>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,195,249,22,192,1                ; vpextrq       $0x1,%xmm0,%r8
@@ -7282,7 +7294,7 @@ _sk_byte_tables_avx LABEL PROC
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,53,24,192,1                 ; vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,124,62,0,0         ; vbroadcastss  0x3e7c(%rip),%ymm9        # 6580 <_sk_callback_avx+0x25e>
+  DB  196,98,125,24,13,124,62,0,0         ; vbroadcastss  0x3e7c(%rip),%ymm9        # 659c <_sk_callback_avx+0x25e>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -7442,7 +7454,7 @@ _sk_byte_tables_rgb_avx LABEL PROC
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,53,24,192,1                 ; vinsertf128   $0x1,%xmm0,%ymm9,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,162,59,0,0         ; vbroadcastss  0x3ba2(%rip),%ymm9        # 6584 <_sk_callback_avx+0x262>
+  DB  196,98,125,24,13,162,59,0,0         ; vbroadcastss  0x3ba2(%rip),%ymm9        # 65a0 <_sk_callback_avx+0x262>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  197,188,89,201                      ; vmulps        %ymm1,%ymm8,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
@@ -7729,36 +7741,36 @@ _sk_parametric_r_avx LABEL PROC
   DB  196,193,124,88,195                  ; vaddps        %ymm11,%ymm0,%ymm0
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,216                      ; vcvtdq2ps     %ymm0,%ymm11
-  DB  196,98,125,24,37,0,55,0,0           ; vbroadcastss  0x3700(%rip),%ymm12        # 6588 <_sk_callback_avx+0x266>
+  DB  196,98,125,24,37,0,55,0,0           ; vbroadcastss  0x3700(%rip),%ymm12        # 65a4 <_sk_callback_avx+0x266>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,246,54,0,0         ; vbroadcastss  0x36f6(%rip),%ymm12        # 658c <_sk_callback_avx+0x26a>
+  DB  196,98,125,24,37,246,54,0,0         ; vbroadcastss  0x36f6(%rip),%ymm12        # 65a8 <_sk_callback_avx+0x26a>
   DB  196,193,124,84,196                  ; vandps        %ymm12,%ymm0,%ymm0
-  DB  196,98,125,24,37,236,54,0,0         ; vbroadcastss  0x36ec(%rip),%ymm12        # 6590 <_sk_callback_avx+0x26e>
+  DB  196,98,125,24,37,236,54,0,0         ; vbroadcastss  0x36ec(%rip),%ymm12        # 65ac <_sk_callback_avx+0x26e>
   DB  196,193,124,86,196                  ; vorps         %ymm12,%ymm0,%ymm0
-  DB  196,98,125,24,37,226,54,0,0         ; vbroadcastss  0x36e2(%rip),%ymm12        # 6594 <_sk_callback_avx+0x272>
+  DB  196,98,125,24,37,226,54,0,0         ; vbroadcastss  0x36e2(%rip),%ymm12        # 65b0 <_sk_callback_avx+0x272>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,216,54,0,0         ; vbroadcastss  0x36d8(%rip),%ymm12        # 6598 <_sk_callback_avx+0x276>
+  DB  196,98,125,24,37,216,54,0,0         ; vbroadcastss  0x36d8(%rip),%ymm12        # 65b4 <_sk_callback_avx+0x276>
   DB  196,65,124,89,228                   ; vmulps        %ymm12,%ymm0,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,201,54,0,0         ; vbroadcastss  0x36c9(%rip),%ymm12        # 659c <_sk_callback_avx+0x27a>
+  DB  196,98,125,24,37,201,54,0,0         ; vbroadcastss  0x36c9(%rip),%ymm12        # 65b8 <_sk_callback_avx+0x27a>
   DB  196,193,124,88,196                  ; vaddps        %ymm12,%ymm0,%ymm0
-  DB  196,98,125,24,37,191,54,0,0         ; vbroadcastss  0x36bf(%rip),%ymm12        # 65a0 <_sk_callback_avx+0x27e>
+  DB  196,98,125,24,37,191,54,0,0         ; vbroadcastss  0x36bf(%rip),%ymm12        # 65bc <_sk_callback_avx+0x27e>
   DB  197,156,94,192                      ; vdivps        %ymm0,%ymm12,%ymm0
   DB  197,164,92,192                      ; vsubps        %ymm0,%ymm11,%ymm0
   DB  197,172,89,192                      ; vmulps        %ymm0,%ymm10,%ymm0
   DB  196,99,125,8,208,1                  ; vroundps      $0x1,%ymm0,%ymm10
   DB  196,65,124,92,210                   ; vsubps        %ymm10,%ymm0,%ymm10
-  DB  196,98,125,24,29,163,54,0,0         ; vbroadcastss  0x36a3(%rip),%ymm11        # 65a4 <_sk_callback_avx+0x282>
+  DB  196,98,125,24,29,163,54,0,0         ; vbroadcastss  0x36a3(%rip),%ymm11        # 65c0 <_sk_callback_avx+0x282>
   DB  196,193,124,88,195                  ; vaddps        %ymm11,%ymm0,%ymm0
-  DB  196,98,125,24,29,153,54,0,0         ; vbroadcastss  0x3699(%rip),%ymm11        # 65a8 <_sk_callback_avx+0x286>
+  DB  196,98,125,24,29,153,54,0,0         ; vbroadcastss  0x3699(%rip),%ymm11        # 65c4 <_sk_callback_avx+0x286>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,124,92,195                  ; vsubps        %ymm11,%ymm0,%ymm0
-  DB  196,98,125,24,29,138,54,0,0         ; vbroadcastss  0x368a(%rip),%ymm11        # 65ac <_sk_callback_avx+0x28a>
+  DB  196,98,125,24,29,138,54,0,0         ; vbroadcastss  0x368a(%rip),%ymm11        # 65c8 <_sk_callback_avx+0x28a>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,128,54,0,0         ; vbroadcastss  0x3680(%rip),%ymm11        # 65b0 <_sk_callback_avx+0x28e>
+  DB  196,98,125,24,29,128,54,0,0         ; vbroadcastss  0x3680(%rip),%ymm11        # 65cc <_sk_callback_avx+0x28e>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,124,88,194                  ; vaddps        %ymm10,%ymm0,%ymm0
-  DB  196,98,125,24,21,113,54,0,0         ; vbroadcastss  0x3671(%rip),%ymm10        # 65b4 <_sk_callback_avx+0x292>
+  DB  196,98,125,24,21,113,54,0,0         ; vbroadcastss  0x3671(%rip),%ymm10        # 65d0 <_sk_callback_avx+0x292>
   DB  196,193,124,89,194                  ; vmulps        %ymm10,%ymm0,%ymm0
   DB  197,253,91,192                      ; vcvtps2dq     %ymm0,%ymm0
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7766,7 +7778,7 @@ _sk_parametric_r_avx LABEL PROC
   DB  196,195,125,74,193,128              ; vblendvps     %ymm8,%ymm9,%ymm0,%ymm0
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,124,95,192                  ; vmaxps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,72,54,0,0           ; vbroadcastss  0x3648(%rip),%ymm8        # 65b8 <_sk_callback_avx+0x296>
+  DB  196,98,125,24,5,72,54,0,0           ; vbroadcastss  0x3648(%rip),%ymm8        # 65d4 <_sk_callback_avx+0x296>
   DB  196,193,124,93,192                  ; vminps        %ymm8,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7786,36 +7798,36 @@ _sk_parametric_g_avx LABEL PROC
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,217                      ; vcvtdq2ps     %ymm1,%ymm11
-  DB  196,98,125,24,37,249,53,0,0         ; vbroadcastss  0x35f9(%rip),%ymm12        # 65bc <_sk_callback_avx+0x29a>
+  DB  196,98,125,24,37,249,53,0,0         ; vbroadcastss  0x35f9(%rip),%ymm12        # 65d8 <_sk_callback_avx+0x29a>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,239,53,0,0         ; vbroadcastss  0x35ef(%rip),%ymm12        # 65c0 <_sk_callback_avx+0x29e>
+  DB  196,98,125,24,37,239,53,0,0         ; vbroadcastss  0x35ef(%rip),%ymm12        # 65dc <_sk_callback_avx+0x29e>
   DB  196,193,116,84,204                  ; vandps        %ymm12,%ymm1,%ymm1
-  DB  196,98,125,24,37,229,53,0,0         ; vbroadcastss  0x35e5(%rip),%ymm12        # 65c4 <_sk_callback_avx+0x2a2>
+  DB  196,98,125,24,37,229,53,0,0         ; vbroadcastss  0x35e5(%rip),%ymm12        # 65e0 <_sk_callback_avx+0x2a2>
   DB  196,193,116,86,204                  ; vorps         %ymm12,%ymm1,%ymm1
-  DB  196,98,125,24,37,219,53,0,0         ; vbroadcastss  0x35db(%rip),%ymm12        # 65c8 <_sk_callback_avx+0x2a6>
+  DB  196,98,125,24,37,219,53,0,0         ; vbroadcastss  0x35db(%rip),%ymm12        # 65e4 <_sk_callback_avx+0x2a6>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,209,53,0,0         ; vbroadcastss  0x35d1(%rip),%ymm12        # 65cc <_sk_callback_avx+0x2aa>
+  DB  196,98,125,24,37,209,53,0,0         ; vbroadcastss  0x35d1(%rip),%ymm12        # 65e8 <_sk_callback_avx+0x2aa>
   DB  196,65,116,89,228                   ; vmulps        %ymm12,%ymm1,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,194,53,0,0         ; vbroadcastss  0x35c2(%rip),%ymm12        # 65d0 <_sk_callback_avx+0x2ae>
+  DB  196,98,125,24,37,194,53,0,0         ; vbroadcastss  0x35c2(%rip),%ymm12        # 65ec <_sk_callback_avx+0x2ae>
   DB  196,193,116,88,204                  ; vaddps        %ymm12,%ymm1,%ymm1
-  DB  196,98,125,24,37,184,53,0,0         ; vbroadcastss  0x35b8(%rip),%ymm12        # 65d4 <_sk_callback_avx+0x2b2>
+  DB  196,98,125,24,37,184,53,0,0         ; vbroadcastss  0x35b8(%rip),%ymm12        # 65f0 <_sk_callback_avx+0x2b2>
   DB  197,156,94,201                      ; vdivps        %ymm1,%ymm12,%ymm1
   DB  197,164,92,201                      ; vsubps        %ymm1,%ymm11,%ymm1
   DB  197,172,89,201                      ; vmulps        %ymm1,%ymm10,%ymm1
   DB  196,99,125,8,209,1                  ; vroundps      $0x1,%ymm1,%ymm10
   DB  196,65,116,92,210                   ; vsubps        %ymm10,%ymm1,%ymm10
-  DB  196,98,125,24,29,156,53,0,0         ; vbroadcastss  0x359c(%rip),%ymm11        # 65d8 <_sk_callback_avx+0x2b6>
+  DB  196,98,125,24,29,156,53,0,0         ; vbroadcastss  0x359c(%rip),%ymm11        # 65f4 <_sk_callback_avx+0x2b6>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,146,53,0,0         ; vbroadcastss  0x3592(%rip),%ymm11        # 65dc <_sk_callback_avx+0x2ba>
+  DB  196,98,125,24,29,146,53,0,0         ; vbroadcastss  0x3592(%rip),%ymm11        # 65f8 <_sk_callback_avx+0x2ba>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,116,92,203                  ; vsubps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,29,131,53,0,0         ; vbroadcastss  0x3583(%rip),%ymm11        # 65e0 <_sk_callback_avx+0x2be>
+  DB  196,98,125,24,29,131,53,0,0         ; vbroadcastss  0x3583(%rip),%ymm11        # 65fc <_sk_callback_avx+0x2be>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,121,53,0,0         ; vbroadcastss  0x3579(%rip),%ymm11        # 65e4 <_sk_callback_avx+0x2c2>
+  DB  196,98,125,24,29,121,53,0,0         ; vbroadcastss  0x3579(%rip),%ymm11        # 6600 <_sk_callback_avx+0x2c2>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,116,88,202                  ; vaddps        %ymm10,%ymm1,%ymm1
-  DB  196,98,125,24,21,106,53,0,0         ; vbroadcastss  0x356a(%rip),%ymm10        # 65e8 <_sk_callback_avx+0x2c6>
+  DB  196,98,125,24,21,106,53,0,0         ; vbroadcastss  0x356a(%rip),%ymm10        # 6604 <_sk_callback_avx+0x2c6>
   DB  196,193,116,89,202                  ; vmulps        %ymm10,%ymm1,%ymm1
   DB  197,253,91,201                      ; vcvtps2dq     %ymm1,%ymm1
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7823,7 +7835,7 @@ _sk_parametric_g_avx LABEL PROC
   DB  196,195,117,74,201,128              ; vblendvps     %ymm8,%ymm9,%ymm1,%ymm1
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,116,95,200                  ; vmaxps        %ymm8,%ymm1,%ymm1
-  DB  196,98,125,24,5,65,53,0,0           ; vbroadcastss  0x3541(%rip),%ymm8        # 65ec <_sk_callback_avx+0x2ca>
+  DB  196,98,125,24,5,65,53,0,0           ; vbroadcastss  0x3541(%rip),%ymm8        # 6608 <_sk_callback_avx+0x2ca>
   DB  196,193,116,93,200                  ; vminps        %ymm8,%ymm1,%ymm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7843,36 +7855,36 @@ _sk_parametric_b_avx LABEL PROC
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,218                      ; vcvtdq2ps     %ymm2,%ymm11
-  DB  196,98,125,24,37,242,52,0,0         ; vbroadcastss  0x34f2(%rip),%ymm12        # 65f0 <_sk_callback_avx+0x2ce>
+  DB  196,98,125,24,37,242,52,0,0         ; vbroadcastss  0x34f2(%rip),%ymm12        # 660c <_sk_callback_avx+0x2ce>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,232,52,0,0         ; vbroadcastss  0x34e8(%rip),%ymm12        # 65f4 <_sk_callback_avx+0x2d2>
+  DB  196,98,125,24,37,232,52,0,0         ; vbroadcastss  0x34e8(%rip),%ymm12        # 6610 <_sk_callback_avx+0x2d2>
   DB  196,193,108,84,212                  ; vandps        %ymm12,%ymm2,%ymm2
-  DB  196,98,125,24,37,222,52,0,0         ; vbroadcastss  0x34de(%rip),%ymm12        # 65f8 <_sk_callback_avx+0x2d6>
+  DB  196,98,125,24,37,222,52,0,0         ; vbroadcastss  0x34de(%rip),%ymm12        # 6614 <_sk_callback_avx+0x2d6>
   DB  196,193,108,86,212                  ; vorps         %ymm12,%ymm2,%ymm2
-  DB  196,98,125,24,37,212,52,0,0         ; vbroadcastss  0x34d4(%rip),%ymm12        # 65fc <_sk_callback_avx+0x2da>
+  DB  196,98,125,24,37,212,52,0,0         ; vbroadcastss  0x34d4(%rip),%ymm12        # 6618 <_sk_callback_avx+0x2da>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,202,52,0,0         ; vbroadcastss  0x34ca(%rip),%ymm12        # 6600 <_sk_callback_avx+0x2de>
+  DB  196,98,125,24,37,202,52,0,0         ; vbroadcastss  0x34ca(%rip),%ymm12        # 661c <_sk_callback_avx+0x2de>
   DB  196,65,108,89,228                   ; vmulps        %ymm12,%ymm2,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,187,52,0,0         ; vbroadcastss  0x34bb(%rip),%ymm12        # 6604 <_sk_callback_avx+0x2e2>
+  DB  196,98,125,24,37,187,52,0,0         ; vbroadcastss  0x34bb(%rip),%ymm12        # 6620 <_sk_callback_avx+0x2e2>
   DB  196,193,108,88,212                  ; vaddps        %ymm12,%ymm2,%ymm2
-  DB  196,98,125,24,37,177,52,0,0         ; vbroadcastss  0x34b1(%rip),%ymm12        # 6608 <_sk_callback_avx+0x2e6>
+  DB  196,98,125,24,37,177,52,0,0         ; vbroadcastss  0x34b1(%rip),%ymm12        # 6624 <_sk_callback_avx+0x2e6>
   DB  197,156,94,210                      ; vdivps        %ymm2,%ymm12,%ymm2
   DB  197,164,92,210                      ; vsubps        %ymm2,%ymm11,%ymm2
   DB  197,172,89,210                      ; vmulps        %ymm2,%ymm10,%ymm2
   DB  196,99,125,8,210,1                  ; vroundps      $0x1,%ymm2,%ymm10
   DB  196,65,108,92,210                   ; vsubps        %ymm10,%ymm2,%ymm10
-  DB  196,98,125,24,29,149,52,0,0         ; vbroadcastss  0x3495(%rip),%ymm11        # 660c <_sk_callback_avx+0x2ea>
+  DB  196,98,125,24,29,149,52,0,0         ; vbroadcastss  0x3495(%rip),%ymm11        # 6628 <_sk_callback_avx+0x2ea>
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
-  DB  196,98,125,24,29,139,52,0,0         ; vbroadcastss  0x348b(%rip),%ymm11        # 6610 <_sk_callback_avx+0x2ee>
+  DB  196,98,125,24,29,139,52,0,0         ; vbroadcastss  0x348b(%rip),%ymm11        # 662c <_sk_callback_avx+0x2ee>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,108,92,211                  ; vsubps        %ymm11,%ymm2,%ymm2
-  DB  196,98,125,24,29,124,52,0,0         ; vbroadcastss  0x347c(%rip),%ymm11        # 6614 <_sk_callback_avx+0x2f2>
+  DB  196,98,125,24,29,124,52,0,0         ; vbroadcastss  0x347c(%rip),%ymm11        # 6630 <_sk_callback_avx+0x2f2>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,114,52,0,0         ; vbroadcastss  0x3472(%rip),%ymm11        # 6618 <_sk_callback_avx+0x2f6>
+  DB  196,98,125,24,29,114,52,0,0         ; vbroadcastss  0x3472(%rip),%ymm11        # 6634 <_sk_callback_avx+0x2f6>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,108,88,210                  ; vaddps        %ymm10,%ymm2,%ymm2
-  DB  196,98,125,24,21,99,52,0,0          ; vbroadcastss  0x3463(%rip),%ymm10        # 661c <_sk_callback_avx+0x2fa>
+  DB  196,98,125,24,21,99,52,0,0          ; vbroadcastss  0x3463(%rip),%ymm10        # 6638 <_sk_callback_avx+0x2fa>
   DB  196,193,108,89,210                  ; vmulps        %ymm10,%ymm2,%ymm2
   DB  197,253,91,210                      ; vcvtps2dq     %ymm2,%ymm2
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7880,7 +7892,7 @@ _sk_parametric_b_avx LABEL PROC
   DB  196,195,109,74,209,128              ; vblendvps     %ymm8,%ymm9,%ymm2,%ymm2
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,108,95,208                  ; vmaxps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,58,52,0,0           ; vbroadcastss  0x343a(%rip),%ymm8        # 6620 <_sk_callback_avx+0x2fe>
+  DB  196,98,125,24,5,58,52,0,0           ; vbroadcastss  0x343a(%rip),%ymm8        # 663c <_sk_callback_avx+0x2fe>
   DB  196,193,108,93,208                  ; vminps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7900,36 +7912,36 @@ _sk_parametric_a_avx LABEL PROC
   DB  196,193,100,88,219                  ; vaddps        %ymm11,%ymm3,%ymm3
   DB  196,98,125,24,16                    ; vbroadcastss  (%rax),%ymm10
   DB  197,124,91,219                      ; vcvtdq2ps     %ymm3,%ymm11
-  DB  196,98,125,24,37,235,51,0,0         ; vbroadcastss  0x33eb(%rip),%ymm12        # 6624 <_sk_callback_avx+0x302>
+  DB  196,98,125,24,37,235,51,0,0         ; vbroadcastss  0x33eb(%rip),%ymm12        # 6640 <_sk_callback_avx+0x302>
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,225,51,0,0         ; vbroadcastss  0x33e1(%rip),%ymm12        # 6628 <_sk_callback_avx+0x306>
+  DB  196,98,125,24,37,225,51,0,0         ; vbroadcastss  0x33e1(%rip),%ymm12        # 6644 <_sk_callback_avx+0x306>
   DB  196,193,100,84,220                  ; vandps        %ymm12,%ymm3,%ymm3
-  DB  196,98,125,24,37,215,51,0,0         ; vbroadcastss  0x33d7(%rip),%ymm12        # 662c <_sk_callback_avx+0x30a>
+  DB  196,98,125,24,37,215,51,0,0         ; vbroadcastss  0x33d7(%rip),%ymm12        # 6648 <_sk_callback_avx+0x30a>
   DB  196,193,100,86,220                  ; vorps         %ymm12,%ymm3,%ymm3
-  DB  196,98,125,24,37,205,51,0,0         ; vbroadcastss  0x33cd(%rip),%ymm12        # 6630 <_sk_callback_avx+0x30e>
+  DB  196,98,125,24,37,205,51,0,0         ; vbroadcastss  0x33cd(%rip),%ymm12        # 664c <_sk_callback_avx+0x30e>
   DB  196,65,36,88,220                    ; vaddps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,195,51,0,0         ; vbroadcastss  0x33c3(%rip),%ymm12        # 6634 <_sk_callback_avx+0x312>
+  DB  196,98,125,24,37,195,51,0,0         ; vbroadcastss  0x33c3(%rip),%ymm12        # 6650 <_sk_callback_avx+0x312>
   DB  196,65,100,89,228                   ; vmulps        %ymm12,%ymm3,%ymm12
   DB  196,65,36,92,220                    ; vsubps        %ymm12,%ymm11,%ymm11
-  DB  196,98,125,24,37,180,51,0,0         ; vbroadcastss  0x33b4(%rip),%ymm12        # 6638 <_sk_callback_avx+0x316>
+  DB  196,98,125,24,37,180,51,0,0         ; vbroadcastss  0x33b4(%rip),%ymm12        # 6654 <_sk_callback_avx+0x316>
   DB  196,193,100,88,220                  ; vaddps        %ymm12,%ymm3,%ymm3
-  DB  196,98,125,24,37,170,51,0,0         ; vbroadcastss  0x33aa(%rip),%ymm12        # 663c <_sk_callback_avx+0x31a>
+  DB  196,98,125,24,37,170,51,0,0         ; vbroadcastss  0x33aa(%rip),%ymm12        # 6658 <_sk_callback_avx+0x31a>
   DB  197,156,94,219                      ; vdivps        %ymm3,%ymm12,%ymm3
   DB  197,164,92,219                      ; vsubps        %ymm3,%ymm11,%ymm3
   DB  197,172,89,219                      ; vmulps        %ymm3,%ymm10,%ymm3
   DB  196,99,125,8,211,1                  ; vroundps      $0x1,%ymm3,%ymm10
   DB  196,65,100,92,210                   ; vsubps        %ymm10,%ymm3,%ymm10
-  DB  196,98,125,24,29,142,51,0,0         ; vbroadcastss  0x338e(%rip),%ymm11        # 6640 <_sk_callback_avx+0x31e>
+  DB  196,98,125,24,29,142,51,0,0         ; vbroadcastss  0x338e(%rip),%ymm11        # 665c <_sk_callback_avx+0x31e>
   DB  196,193,100,88,219                  ; vaddps        %ymm11,%ymm3,%ymm3
-  DB  196,98,125,24,29,132,51,0,0         ; vbroadcastss  0x3384(%rip),%ymm11        # 6644 <_sk_callback_avx+0x322>
+  DB  196,98,125,24,29,132,51,0,0         ; vbroadcastss  0x3384(%rip),%ymm11        # 6660 <_sk_callback_avx+0x322>
   DB  196,65,44,89,219                    ; vmulps        %ymm11,%ymm10,%ymm11
   DB  196,193,100,92,219                  ; vsubps        %ymm11,%ymm3,%ymm3
-  DB  196,98,125,24,29,117,51,0,0         ; vbroadcastss  0x3375(%rip),%ymm11        # 6648 <_sk_callback_avx+0x326>
+  DB  196,98,125,24,29,117,51,0,0         ; vbroadcastss  0x3375(%rip),%ymm11        # 6664 <_sk_callback_avx+0x326>
   DB  196,65,36,92,210                    ; vsubps        %ymm10,%ymm11,%ymm10
-  DB  196,98,125,24,29,107,51,0,0         ; vbroadcastss  0x336b(%rip),%ymm11        # 664c <_sk_callback_avx+0x32a>
+  DB  196,98,125,24,29,107,51,0,0         ; vbroadcastss  0x336b(%rip),%ymm11        # 6668 <_sk_callback_avx+0x32a>
   DB  196,65,36,94,210                    ; vdivps        %ymm10,%ymm11,%ymm10
   DB  196,193,100,88,218                  ; vaddps        %ymm10,%ymm3,%ymm3
-  DB  196,98,125,24,21,92,51,0,0          ; vbroadcastss  0x335c(%rip),%ymm10        # 6650 <_sk_callback_avx+0x32e>
+  DB  196,98,125,24,21,92,51,0,0          ; vbroadcastss  0x335c(%rip),%ymm10        # 666c <_sk_callback_avx+0x32e>
   DB  196,193,100,89,218                  ; vmulps        %ymm10,%ymm3,%ymm3
   DB  197,253,91,219                      ; vcvtps2dq     %ymm3,%ymm3
   DB  196,98,125,24,80,20                 ; vbroadcastss  0x14(%rax),%ymm10
@@ -7937,38 +7949,38 @@ _sk_parametric_a_avx LABEL PROC
   DB  196,195,101,74,217,128              ; vblendvps     %ymm8,%ymm9,%ymm3,%ymm3
   DB  196,65,60,87,192                    ; vxorps        %ymm8,%ymm8,%ymm8
   DB  196,193,100,95,216                  ; vmaxps        %ymm8,%ymm3,%ymm3
-  DB  196,98,125,24,5,51,51,0,0           ; vbroadcastss  0x3333(%rip),%ymm8        # 6654 <_sk_callback_avx+0x332>
+  DB  196,98,125,24,5,51,51,0,0           ; vbroadcastss  0x3333(%rip),%ymm8        # 6670 <_sk_callback_avx+0x332>
   DB  196,193,100,93,216                  ; vminps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_lab_to_xyz_avx
 _sk_lab_to_xyz_avx LABEL PROC
-  DB  196,98,125,24,5,37,51,0,0           ; vbroadcastss  0x3325(%rip),%ymm8        # 6658 <_sk_callback_avx+0x336>
+  DB  196,98,125,24,5,37,51,0,0           ; vbroadcastss  0x3325(%rip),%ymm8        # 6674 <_sk_callback_avx+0x336>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,27,51,0,0           ; vbroadcastss  0x331b(%rip),%ymm8        # 665c <_sk_callback_avx+0x33a>
+  DB  196,98,125,24,5,27,51,0,0           ; vbroadcastss  0x331b(%rip),%ymm8        # 6678 <_sk_callback_avx+0x33a>
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
-  DB  196,98,125,24,13,17,51,0,0          ; vbroadcastss  0x3311(%rip),%ymm9        # 6660 <_sk_callback_avx+0x33e>
+  DB  196,98,125,24,13,17,51,0,0          ; vbroadcastss  0x3311(%rip),%ymm9        # 667c <_sk_callback_avx+0x33e>
   DB  196,193,116,88,201                  ; vaddps        %ymm9,%ymm1,%ymm1
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  196,193,108,88,209                  ; vaddps        %ymm9,%ymm2,%ymm2
-  DB  196,98,125,24,5,253,50,0,0          ; vbroadcastss  0x32fd(%rip),%ymm8        # 6664 <_sk_callback_avx+0x342>
+  DB  196,98,125,24,5,253,50,0,0          ; vbroadcastss  0x32fd(%rip),%ymm8        # 6680 <_sk_callback_avx+0x342>
   DB  196,193,124,88,192                  ; vaddps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,243,50,0,0          ; vbroadcastss  0x32f3(%rip),%ymm8        # 6668 <_sk_callback_avx+0x346>
+  DB  196,98,125,24,5,243,50,0,0          ; vbroadcastss  0x32f3(%rip),%ymm8        # 6684 <_sk_callback_avx+0x346>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,5,233,50,0,0          ; vbroadcastss  0x32e9(%rip),%ymm8        # 666c <_sk_callback_avx+0x34a>
+  DB  196,98,125,24,5,233,50,0,0          ; vbroadcastss  0x32e9(%rip),%ymm8        # 6688 <_sk_callback_avx+0x34a>
   DB  196,193,116,89,200                  ; vmulps        %ymm8,%ymm1,%ymm1
   DB  197,252,88,201                      ; vaddps        %ymm1,%ymm0,%ymm1
-  DB  196,98,125,24,5,219,50,0,0          ; vbroadcastss  0x32db(%rip),%ymm8        # 6670 <_sk_callback_avx+0x34e>
+  DB  196,98,125,24,5,219,50,0,0          ; vbroadcastss  0x32db(%rip),%ymm8        # 668c <_sk_callback_avx+0x34e>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  197,252,92,210                      ; vsubps        %ymm2,%ymm0,%ymm2
   DB  197,116,89,193                      ; vmulps        %ymm1,%ymm1,%ymm8
   DB  196,65,116,89,192                   ; vmulps        %ymm8,%ymm1,%ymm8
-  DB  196,98,125,24,13,196,50,0,0         ; vbroadcastss  0x32c4(%rip),%ymm9        # 6674 <_sk_callback_avx+0x352>
+  DB  196,98,125,24,13,196,50,0,0         ; vbroadcastss  0x32c4(%rip),%ymm9        # 6690 <_sk_callback_avx+0x352>
   DB  196,65,52,194,208,1                 ; vcmpltps      %ymm8,%ymm9,%ymm10
-  DB  196,98,125,24,29,185,50,0,0         ; vbroadcastss  0x32b9(%rip),%ymm11        # 6678 <_sk_callback_avx+0x356>
+  DB  196,98,125,24,29,185,50,0,0         ; vbroadcastss  0x32b9(%rip),%ymm11        # 6694 <_sk_callback_avx+0x356>
   DB  196,193,116,88,203                  ; vaddps        %ymm11,%ymm1,%ymm1
-  DB  196,98,125,24,37,175,50,0,0         ; vbroadcastss  0x32af(%rip),%ymm12        # 667c <_sk_callback_avx+0x35a>
+  DB  196,98,125,24,37,175,50,0,0         ; vbroadcastss  0x32af(%rip),%ymm12        # 6698 <_sk_callback_avx+0x35a>
   DB  196,193,116,89,204                  ; vmulps        %ymm12,%ymm1,%ymm1
   DB  196,67,117,74,192,160               ; vblendvps     %ymm10,%ymm8,%ymm1,%ymm8
   DB  197,252,89,200                      ; vmulps        %ymm0,%ymm0,%ymm1
@@ -7983,9 +7995,9 @@ _sk_lab_to_xyz_avx LABEL PROC
   DB  196,193,108,88,211                  ; vaddps        %ymm11,%ymm2,%ymm2
   DB  196,193,108,89,212                  ; vmulps        %ymm12,%ymm2,%ymm2
   DB  196,227,109,74,208,144              ; vblendvps     %ymm9,%ymm0,%ymm2,%ymm2
-  DB  196,226,125,24,5,101,50,0,0         ; vbroadcastss  0x3265(%rip),%ymm0        # 6680 <_sk_callback_avx+0x35e>
+  DB  196,226,125,24,5,101,50,0,0         ; vbroadcastss  0x3265(%rip),%ymm0        # 669c <_sk_callback_avx+0x35e>
   DB  197,188,89,192                      ; vmulps        %ymm0,%ymm8,%ymm0
-  DB  196,98,125,24,5,92,50,0,0           ; vbroadcastss  0x325c(%rip),%ymm8        # 6684 <_sk_callback_avx+0x362>
+  DB  196,98,125,24,5,92,50,0,0           ; vbroadcastss  0x325c(%rip),%ymm8        # 66a0 <_sk_callback_avx+0x362>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -7997,14 +8009,14 @@ _sk_load_a8_avx LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,62                              ; jne           347f <_sk_load_a8_avx+0x4e>
+  DB  117,62                              ; jne           349b <_sk_load_a8_avx+0x4e>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,121,49,200                  ; vpmovzxbd     %xmm0,%xmm1
   DB  196,227,121,4,192,229               ; vpermilps     $0xe5,%xmm0,%xmm0
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,117,24,192,1                ; vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,32,50,0,0         ; vbroadcastss  0x3220(%rip),%ymm1        # 6688 <_sk_callback_avx+0x366>
+  DB  196,226,125,24,13,32,50,0,0         ; vbroadcastss  0x3220(%rip),%ymm1        # 66a4 <_sk_callback_avx+0x366>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -8021,9 +8033,9 @@ _sk_load_a8_avx LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           3487 <_sk_load_a8_avx+0x56>
+  DB  117,234                             ; jne           34a3 <_sk_load_a8_avx+0x56>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,161                             ; jmp           3445 <_sk_load_a8_avx+0x14>
+  DB  235,161                             ; jmp           3461 <_sk_load_a8_avx+0x14>
 
 PUBLIC _sk_gather_a8_avx
 _sk_gather_a8_avx LABEL PROC
@@ -8071,7 +8083,7 @@ _sk_gather_a8_avx LABEL PROC
   DB  196,226,121,49,201                  ; vpmovzxbd     %xmm1,%xmm1
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,21,49,0,0         ; vbroadcastss  0x3115(%rip),%ymm1        # 668c <_sk_callback_avx+0x36a>
+  DB  196,226,125,24,13,21,49,0,0         ; vbroadcastss  0x3115(%rip),%ymm1        # 66a8 <_sk_callback_avx+0x36a>
   DB  197,252,89,217                      ; vmulps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  197,252,87,192                      ; vxorps        %ymm0,%ymm0,%ymm0
@@ -8087,14 +8099,14 @@ PUBLIC _sk_store_a8_avx
 _sk_store_a8_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,240,48,0,0          ; vbroadcastss  0x30f0(%rip),%ymm8        # 6690 <_sk_callback_avx+0x36e>
+  DB  196,98,125,24,5,240,48,0,0          ; vbroadcastss  0x30f0(%rip),%ymm8        # 66ac <_sk_callback_avx+0x36e>
   DB  196,65,100,89,192                   ; vmulps        %ymm8,%ymm3,%ymm8
   DB  196,65,125,91,192                   ; vcvtps2dq     %ymm8,%ymm8
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  196,65,57,103,192                   ; vpackuswb     %xmm8,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           35c9 <_sk_store_a8_avx+0x37>
+  DB  117,10                              ; jne           35e5 <_sk_store_a8_avx+0x37>
   DB  196,65,123,17,4,58                  ; vmovsd        %xmm8,(%r10,%rdi,1)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8102,10 +8114,10 @@ _sk_store_a8_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            35c5 <_sk_store_a8_avx+0x33>
+  DB  119,236                             ; ja            35e1 <_sk_store_a8_avx+0x33>
   DB  196,66,121,48,192                   ; vpmovzxbw     %xmm8,%xmm8
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 362c <_sk_store_a8_avx+0x9a>
+  DB  76,141,13,67,0,0,0                  ; lea           0x43(%rip),%r9        # 3648 <_sk_store_a8_avx+0x9a>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8116,7 +8128,7 @@ _sk_store_a8_avx LABEL PROC
   DB  196,67,121,20,68,58,2,4             ; vpextrb       $0x4,%xmm8,0x2(%r10,%rdi,1)
   DB  196,67,121,20,68,58,1,2             ; vpextrb       $0x2,%xmm8,0x1(%r10,%rdi,1)
   DB  196,67,121,20,4,58,0                ; vpextrb       $0x0,%xmm8,(%r10,%rdi,1)
-  DB  235,154                             ; jmp           35c5 <_sk_store_a8_avx+0x33>
+  DB  235,154                             ; jmp           35e1 <_sk_store_a8_avx+0x33>
   DB  144                                 ; nop
   DB  246,255                             ; idiv          %bh
   DB  255                                 ; (bad)
@@ -8148,17 +8160,17 @@ _sk_load_g8_avx LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,1,248                            ; add           %rdi,%rax
   DB  77,133,192                          ; test          %r8,%r8
-  DB  117,67                              ; jne           369b <_sk_load_g8_avx+0x53>
+  DB  117,67                              ; jne           36b7 <_sk_load_g8_avx+0x53>
   DB  197,250,126,0                       ; vmovq         (%rax),%xmm0
   DB  196,226,121,49,200                  ; vpmovzxbd     %xmm0,%xmm1
   DB  196,227,121,4,192,229               ; vpermilps     $0xe5,%xmm0,%xmm0
   DB  196,226,121,49,192                  ; vpmovzxbd     %xmm0,%xmm0
   DB  196,227,117,24,192,1                ; vinsertf128   $0x1,%xmm0,%ymm1,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,21,48,0,0         ; vbroadcastss  0x3015(%rip),%ymm1        # 6694 <_sk_callback_avx+0x372>
+  DB  196,226,125,24,13,21,48,0,0         ; vbroadcastss  0x3015(%rip),%ymm1        # 66b0 <_sk_callback_avx+0x372>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,10,48,0,0         ; vbroadcastss  0x300a(%rip),%ymm3        # 6698 <_sk_callback_avx+0x376>
+  DB  196,226,125,24,29,10,48,0,0         ; vbroadcastss  0x300a(%rip),%ymm3        # 66b4 <_sk_callback_avx+0x376>
   DB  76,137,193                          ; mov           %r8,%rcx
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
@@ -8172,9 +8184,9 @@ _sk_load_g8_avx LABEL PROC
   DB  77,9,217                            ; or            %r11,%r9
   DB  72,131,193,8                        ; add           $0x8,%rcx
   DB  73,255,202                          ; dec           %r10
-  DB  117,234                             ; jne           36a3 <_sk_load_g8_avx+0x5b>
+  DB  117,234                             ; jne           36bf <_sk_load_g8_avx+0x5b>
   DB  196,193,249,110,193                 ; vmovq         %r9,%xmm0
-  DB  235,156                             ; jmp           365c <_sk_load_g8_avx+0x14>
+  DB  235,156                             ; jmp           3678 <_sk_load_g8_avx+0x14>
 
 PUBLIC _sk_gather_g8_avx
 _sk_gather_g8_avx LABEL PROC
@@ -8222,10 +8234,10 @@ _sk_gather_g8_avx LABEL PROC
   DB  196,226,121,49,201                  ; vpmovzxbd     %xmm1,%xmm1
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,9,47,0,0          ; vbroadcastss  0x2f09(%rip),%ymm1        # 669c <_sk_callback_avx+0x37a>
+  DB  196,226,125,24,13,9,47,0,0          ; vbroadcastss  0x2f09(%rip),%ymm1        # 66b8 <_sk_callback_avx+0x37a>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,254,46,0,0        ; vbroadcastss  0x2efe(%rip),%ymm3        # 66a0 <_sk_callback_avx+0x37e>
+  DB  196,226,125,24,29,254,46,0,0        ; vbroadcastss  0x2efe(%rip),%ymm3        # 66bc <_sk_callback_avx+0x37e>
   DB  197,252,40,200                      ; vmovaps       %ymm0,%ymm1
   DB  197,252,40,208                      ; vmovaps       %ymm0,%ymm2
   DB  91                                  ; pop           %rbx
@@ -8239,9 +8251,9 @@ _sk_gather_i8_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            37c2 <_sk_gather_i8_avx+0xf>
+  DB  116,5                               ; je            37de <_sk_gather_i8_avx+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           37c4 <_sk_gather_i8_avx+0x11>
+  DB  235,2                               ; jmp           37e0 <_sk_gather_i8_avx+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,87                               ; push          %r15
   DB  65,86                               ; push          %r14
@@ -8303,10 +8315,10 @@ _sk_gather_i8_avx LABEL PROC
   DB  196,163,121,34,4,163,2              ; vpinsrd       $0x2,(%rbx,%r12,4),%xmm0,%xmm0
   DB  196,163,121,34,28,19,3              ; vpinsrd       $0x3,(%rbx,%r10,1),%xmm0,%xmm3
   DB  196,227,61,24,195,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  DB  197,124,40,21,146,47,0,0            ; vmovaps       0x2f92(%rip),%ymm10        # 6880 <_sk_callback_avx+0x55e>
+  DB  197,124,40,21,118,47,0,0            ; vmovaps       0x2f76(%rip),%ymm10        # 6880 <_sk_callback_avx+0x542>
   DB  196,193,124,84,194                  ; vandps        %ymm10,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,164,45,0,0         ; vbroadcastss  0x2da4(%rip),%ymm9        # 66a4 <_sk_callback_avx+0x382>
+  DB  196,98,125,24,13,164,45,0,0         ; vbroadcastss  0x2da4(%rip),%ymm9        # 66c0 <_sk_callback_avx+0x382>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,113,114,208,8               ; vpsrld        $0x8,%xmm8,%xmm1
   DB  197,233,114,211,8                   ; vpsrld        $0x8,%xmm3,%xmm2
@@ -8338,38 +8350,38 @@ _sk_load_565_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,128,0,0,0                    ; jne           39f8 <_sk_load_565_avx+0x8e>
+  DB  15,133,128,0,0,0                    ; jne           3a14 <_sk_load_565_avx+0x8e>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  197,241,239,201                     ; vpxor         %xmm1,%xmm1,%xmm1
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,209,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  DB  196,226,125,24,5,14,45,0,0          ; vbroadcastss  0x2d0e(%rip),%ymm0        # 66a8 <_sk_callback_avx+0x386>
+  DB  196,226,125,24,5,14,45,0,0          ; vbroadcastss  0x2d0e(%rip),%ymm0        # 66c4 <_sk_callback_avx+0x386>
   DB  197,236,84,192                      ; vandps        %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,1,45,0,0          ; vbroadcastss  0x2d01(%rip),%ymm1        # 66ac <_sk_callback_avx+0x38a>
+  DB  196,226,125,24,13,1,45,0,0          ; vbroadcastss  0x2d01(%rip),%ymm1        # 66c8 <_sk_callback_avx+0x38a>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,248,44,0,0        ; vbroadcastss  0x2cf8(%rip),%ymm1        # 66b0 <_sk_callback_avx+0x38e>
+  DB  196,226,125,24,13,248,44,0,0        ; vbroadcastss  0x2cf8(%rip),%ymm1        # 66cc <_sk_callback_avx+0x38e>
   DB  197,236,84,201                      ; vandps        %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,235,44,0,0        ; vbroadcastss  0x2ceb(%rip),%ymm3        # 66b4 <_sk_callback_avx+0x392>
+  DB  196,226,125,24,29,235,44,0,0        ; vbroadcastss  0x2ceb(%rip),%ymm3        # 66d0 <_sk_callback_avx+0x392>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,24,29,226,44,0,0        ; vbroadcastss  0x2ce2(%rip),%ymm3        # 66b8 <_sk_callback_avx+0x396>
+  DB  196,226,125,24,29,226,44,0,0        ; vbroadcastss  0x2ce2(%rip),%ymm3        # 66d4 <_sk_callback_avx+0x396>
   DB  197,236,84,211                      ; vandps        %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,213,44,0,0        ; vbroadcastss  0x2cd5(%rip),%ymm3        # 66bc <_sk_callback_avx+0x39a>
+  DB  196,226,125,24,29,213,44,0,0        ; vbroadcastss  0x2cd5(%rip),%ymm3        # 66d8 <_sk_callback_avx+0x39a>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,202,44,0,0        ; vbroadcastss  0x2cca(%rip),%ymm3        # 66c0 <_sk_callback_avx+0x39e>
+  DB  196,226,125,24,29,202,44,0,0        ; vbroadcastss  0x2cca(%rip),%ymm3        # 66dc <_sk_callback_avx+0x39e>
   DB  255,224                             ; jmpq          *%rax
   DB  65,137,200                          ; mov           %ecx,%r8d
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,110,255,255,255              ; ja            397e <_sk_load_565_avx+0x14>
+  DB  15,135,110,255,255,255              ; ja            399a <_sk_load_565_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 3a64 <_sk_load_565_avx+0xfa>
+  DB  76,141,13,73,0,0,0                  ; lea           0x49(%rip),%r9        # 3a80 <_sk_load_565_avx+0xfa>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8381,7 +8393,7 @@ _sk_load_565_avx LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,26,255,255,255                  ; jmpq          397e <_sk_load_565_avx+0x14>
+  DB  233,26,255,255,255                  ; jmpq          399a <_sk_load_565_avx+0x14>
   DB  244                                 ; hlt
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -8457,23 +8469,23 @@ _sk_gather_565_avx LABEL PROC
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,209,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm2
-  DB  196,226,125,24,5,106,43,0,0         ; vbroadcastss  0x2b6a(%rip),%ymm0        # 66c4 <_sk_callback_avx+0x3a2>
+  DB  196,226,125,24,5,106,43,0,0         ; vbroadcastss  0x2b6a(%rip),%ymm0        # 66e0 <_sk_callback_avx+0x3a2>
   DB  197,236,84,192                      ; vandps        %ymm0,%ymm2,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,93,43,0,0         ; vbroadcastss  0x2b5d(%rip),%ymm1        # 66c8 <_sk_callback_avx+0x3a6>
+  DB  196,226,125,24,13,93,43,0,0         ; vbroadcastss  0x2b5d(%rip),%ymm1        # 66e4 <_sk_callback_avx+0x3a6>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,84,43,0,0         ; vbroadcastss  0x2b54(%rip),%ymm1        # 66cc <_sk_callback_avx+0x3aa>
+  DB  196,226,125,24,13,84,43,0,0         ; vbroadcastss  0x2b54(%rip),%ymm1        # 66e8 <_sk_callback_avx+0x3aa>
   DB  197,236,84,201                      ; vandps        %ymm1,%ymm2,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,29,71,43,0,0         ; vbroadcastss  0x2b47(%rip),%ymm3        # 66d0 <_sk_callback_avx+0x3ae>
+  DB  196,226,125,24,29,71,43,0,0         ; vbroadcastss  0x2b47(%rip),%ymm3        # 66ec <_sk_callback_avx+0x3ae>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
-  DB  196,226,125,24,29,62,43,0,0         ; vbroadcastss  0x2b3e(%rip),%ymm3        # 66d4 <_sk_callback_avx+0x3b2>
+  DB  196,226,125,24,29,62,43,0,0         ; vbroadcastss  0x2b3e(%rip),%ymm3        # 66f0 <_sk_callback_avx+0x3b2>
   DB  197,236,84,211                      ; vandps        %ymm3,%ymm2,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,226,125,24,29,49,43,0,0         ; vbroadcastss  0x2b31(%rip),%ymm3        # 66d8 <_sk_callback_avx+0x3b6>
+  DB  196,226,125,24,29,49,43,0,0         ; vbroadcastss  0x2b31(%rip),%ymm3        # 66f4 <_sk_callback_avx+0x3b6>
   DB  197,236,89,211                      ; vmulps        %ymm3,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,38,43,0,0         ; vbroadcastss  0x2b26(%rip),%ymm3        # 66dc <_sk_callback_avx+0x3ba>
+  DB  196,226,125,24,29,38,43,0,0         ; vbroadcastss  0x2b26(%rip),%ymm3        # 66f8 <_sk_callback_avx+0x3ba>
   DB  91                                  ; pop           %rbx
   DB  65,92                               ; pop           %r12
   DB  65,94                               ; pop           %r14
@@ -8485,14 +8497,14 @@ PUBLIC _sk_store_565_avx
 _sk_store_565_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,18,43,0,0           ; vbroadcastss  0x2b12(%rip),%ymm8        # 66e0 <_sk_callback_avx+0x3be>
+  DB  196,98,125,24,5,18,43,0,0           ; vbroadcastss  0x2b12(%rip),%ymm8        # 66fc <_sk_callback_avx+0x3be>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,41,114,241,11               ; vpslld        $0xb,%xmm9,%xmm10
   DB  196,67,125,25,201,1                 ; vextractf128  $0x1,%ymm9,%xmm9
   DB  196,193,49,114,241,11               ; vpslld        $0xb,%xmm9,%xmm9
   DB  196,67,45,24,201,1                  ; vinsertf128   $0x1,%xmm9,%ymm10,%ymm9
-  DB  196,98,125,24,21,235,42,0,0         ; vbroadcastss  0x2aeb(%rip),%ymm10        # 66e4 <_sk_callback_avx+0x3c2>
+  DB  196,98,125,24,21,235,42,0,0         ; vbroadcastss  0x2aeb(%rip),%ymm10        # 6700 <_sk_callback_avx+0x3c2>
   DB  196,65,116,89,210                   ; vmulps        %ymm10,%ymm1,%ymm10
   DB  196,65,125,91,210                   ; vcvtps2dq     %ymm10,%ymm10
   DB  196,193,33,114,242,5                ; vpslld        $0x5,%xmm10,%xmm11
@@ -8506,7 +8518,7 @@ _sk_store_565_avx LABEL PROC
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           3c49 <_sk_store_565_avx+0x89>
+  DB  117,10                              ; jne           3c65 <_sk_store_565_avx+0x89>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8514,9 +8526,9 @@ _sk_store_565_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            3c45 <_sk_store_565_avx+0x85>
+  DB  119,236                             ; ja            3c61 <_sk_store_565_avx+0x85>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3ca8 <_sk_store_565_avx+0xe8>
+  DB  76,141,13,68,0,0,0                  ; lea           0x44(%rip),%r9        # 3cc4 <_sk_store_565_avx+0xe8>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8527,7 +8539,7 @@ _sk_store_565_avx LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           3c45 <_sk_store_565_avx+0x85>
+  DB  235,159                             ; jmp           3c61 <_sk_store_565_avx+0x85>
   DB  102,144                             ; xchg          %ax,%ax
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -8558,31 +8570,31 @@ _sk_load_4444_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,152,0,0,0                    ; jne           3d6a <_sk_load_4444_avx+0xa6>
+  DB  15,133,152,0,0,0                    ; jne           3d86 <_sk_load_4444_avx+0xa6>
   DB  196,193,122,111,4,122               ; vmovdqu       (%r10,%rdi,2),%xmm0
   DB  197,241,239,201                     ; vpxor         %xmm1,%xmm1,%xmm1
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,217,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  DB  196,226,125,24,5,244,41,0,0         ; vbroadcastss  0x29f4(%rip),%ymm0        # 66e8 <_sk_callback_avx+0x3c6>
+  DB  196,226,125,24,5,244,41,0,0         ; vbroadcastss  0x29f4(%rip),%ymm0        # 6704 <_sk_callback_avx+0x3c6>
   DB  197,228,84,192                      ; vandps        %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,231,41,0,0        ; vbroadcastss  0x29e7(%rip),%ymm1        # 66ec <_sk_callback_avx+0x3ca>
+  DB  196,226,125,24,13,231,41,0,0        ; vbroadcastss  0x29e7(%rip),%ymm1        # 6708 <_sk_callback_avx+0x3ca>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,222,41,0,0        ; vbroadcastss  0x29de(%rip),%ymm1        # 66f0 <_sk_callback_avx+0x3ce>
+  DB  196,226,125,24,13,222,41,0,0        ; vbroadcastss  0x29de(%rip),%ymm1        # 670c <_sk_callback_avx+0x3ce>
   DB  197,228,84,201                      ; vandps        %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,209,41,0,0        ; vbroadcastss  0x29d1(%rip),%ymm2        # 66f4 <_sk_callback_avx+0x3d2>
+  DB  196,226,125,24,21,209,41,0,0        ; vbroadcastss  0x29d1(%rip),%ymm2        # 6710 <_sk_callback_avx+0x3d2>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,24,21,200,41,0,0        ; vbroadcastss  0x29c8(%rip),%ymm2        # 66f8 <_sk_callback_avx+0x3d6>
+  DB  196,226,125,24,21,200,41,0,0        ; vbroadcastss  0x29c8(%rip),%ymm2        # 6714 <_sk_callback_avx+0x3d6>
   DB  197,228,84,210                      ; vandps        %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,187,41,0,0          ; vbroadcastss  0x29bb(%rip),%ymm8        # 66fc <_sk_callback_avx+0x3da>
+  DB  196,98,125,24,5,187,41,0,0          ; vbroadcastss  0x29bb(%rip),%ymm8        # 6718 <_sk_callback_avx+0x3da>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,177,41,0,0          ; vbroadcastss  0x29b1(%rip),%ymm8        # 6700 <_sk_callback_avx+0x3de>
+  DB  196,98,125,24,5,177,41,0,0          ; vbroadcastss  0x29b1(%rip),%ymm8        # 671c <_sk_callback_avx+0x3de>
   DB  196,193,100,84,216                  ; vandps        %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,163,41,0,0          ; vbroadcastss  0x29a3(%rip),%ymm8        # 6704 <_sk_callback_avx+0x3e2>
+  DB  196,98,125,24,5,163,41,0,0          ; vbroadcastss  0x29a3(%rip),%ymm8        # 6720 <_sk_callback_avx+0x3e2>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8591,9 +8603,9 @@ _sk_load_4444_avx LABEL PROC
   DB  197,249,239,192                     ; vpxor         %xmm0,%xmm0,%xmm0
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,86,255,255,255               ; ja            3cd8 <_sk_load_4444_avx+0x14>
+  DB  15,135,86,255,255,255               ; ja            3cf4 <_sk_load_4444_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 3dd8 <_sk_load_4444_avx+0x114>
+  DB  76,141,13,75,0,0,0                  ; lea           0x4b(%rip),%r9        # 3df4 <_sk_load_4444_avx+0x114>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8605,7 +8617,7 @@ _sk_load_4444_avx LABEL PROC
   DB  196,193,121,196,68,122,4,2          ; vpinsrw       $0x2,0x4(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,68,122,2,1          ; vpinsrw       $0x1,0x2(%r10,%rdi,2),%xmm0,%xmm0
   DB  196,193,121,196,4,122,0             ; vpinsrw       $0x0,(%r10,%rdi,2),%xmm0,%xmm0
-  DB  233,2,255,255,255                   ; jmpq          3cd8 <_sk_load_4444_avx+0x14>
+  DB  233,2,255,255,255                   ; jmpq          3cf4 <_sk_load_4444_avx+0x14>
   DB  102,144                             ; xchg          %ax,%ax
   DB  242,255                             ; repnz         (bad)
   DB  255                                 ; (bad)
@@ -8682,25 +8694,25 @@ _sk_gather_4444_avx LABEL PROC
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,217,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm3
-  DB  196,226,125,24,5,58,40,0,0          ; vbroadcastss  0x283a(%rip),%ymm0        # 6708 <_sk_callback_avx+0x3e6>
+  DB  196,226,125,24,5,58,40,0,0          ; vbroadcastss  0x283a(%rip),%ymm0        # 6724 <_sk_callback_avx+0x3e6>
   DB  197,228,84,192                      ; vandps        %ymm0,%ymm3,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,226,125,24,13,45,40,0,0         ; vbroadcastss  0x282d(%rip),%ymm1        # 670c <_sk_callback_avx+0x3ea>
+  DB  196,226,125,24,13,45,40,0,0         ; vbroadcastss  0x282d(%rip),%ymm1        # 6728 <_sk_callback_avx+0x3ea>
   DB  197,252,89,193                      ; vmulps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,36,40,0,0         ; vbroadcastss  0x2824(%rip),%ymm1        # 6710 <_sk_callback_avx+0x3ee>
+  DB  196,226,125,24,13,36,40,0,0         ; vbroadcastss  0x2824(%rip),%ymm1        # 672c <_sk_callback_avx+0x3ee>
   DB  197,228,84,201                      ; vandps        %ymm1,%ymm3,%ymm1
   DB  197,252,91,201                      ; vcvtdq2ps     %ymm1,%ymm1
-  DB  196,226,125,24,21,23,40,0,0         ; vbroadcastss  0x2817(%rip),%ymm2        # 6714 <_sk_callback_avx+0x3f2>
+  DB  196,226,125,24,21,23,40,0,0         ; vbroadcastss  0x2817(%rip),%ymm2        # 6730 <_sk_callback_avx+0x3f2>
   DB  197,244,89,202                      ; vmulps        %ymm2,%ymm1,%ymm1
-  DB  196,226,125,24,21,14,40,0,0         ; vbroadcastss  0x280e(%rip),%ymm2        # 6718 <_sk_callback_avx+0x3f6>
+  DB  196,226,125,24,21,14,40,0,0         ; vbroadcastss  0x280e(%rip),%ymm2        # 6734 <_sk_callback_avx+0x3f6>
   DB  197,228,84,210                      ; vandps        %ymm2,%ymm3,%ymm2
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
-  DB  196,98,125,24,5,1,40,0,0            ; vbroadcastss  0x2801(%rip),%ymm8        # 671c <_sk_callback_avx+0x3fa>
+  DB  196,98,125,24,5,1,40,0,0            ; vbroadcastss  0x2801(%rip),%ymm8        # 6738 <_sk_callback_avx+0x3fa>
   DB  196,193,108,89,208                  ; vmulps        %ymm8,%ymm2,%ymm2
-  DB  196,98,125,24,5,247,39,0,0          ; vbroadcastss  0x27f7(%rip),%ymm8        # 6720 <_sk_callback_avx+0x3fe>
+  DB  196,98,125,24,5,247,39,0,0          ; vbroadcastss  0x27f7(%rip),%ymm8        # 673c <_sk_callback_avx+0x3fe>
   DB  196,193,100,84,216                  ; vandps        %ymm8,%ymm3,%ymm3
   DB  197,252,91,219                      ; vcvtdq2ps     %ymm3,%ymm3
-  DB  196,98,125,24,5,233,39,0,0          ; vbroadcastss  0x27e9(%rip),%ymm8        # 6724 <_sk_callback_avx+0x402>
+  DB  196,98,125,24,5,233,39,0,0          ; vbroadcastss  0x27e9(%rip),%ymm8        # 6740 <_sk_callback_avx+0x402>
   DB  196,193,100,89,216                  ; vmulps        %ymm8,%ymm3,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  91                                  ; pop           %rbx
@@ -8714,7 +8726,7 @@ PUBLIC _sk_store_4444_avx
 _sk_store_4444_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,206,39,0,0          ; vbroadcastss  0x27ce(%rip),%ymm8        # 6728 <_sk_callback_avx+0x406>
+  DB  196,98,125,24,5,206,39,0,0          ; vbroadcastss  0x27ce(%rip),%ymm8        # 6744 <_sk_callback_avx+0x406>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,193,41,114,241,12               ; vpslld        $0xc,%xmm9,%xmm10
@@ -8741,7 +8753,7 @@ _sk_store_4444_avx LABEL PROC
   DB  196,67,125,25,193,1                 ; vextractf128  $0x1,%ymm8,%xmm9
   DB  196,66,57,43,193                    ; vpackusdw     %xmm9,%xmm8,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           3ff3 <_sk_store_4444_avx+0xa7>
+  DB  117,10                              ; jne           400f <_sk_store_4444_avx+0xa7>
   DB  196,65,122,127,4,122                ; vmovdqu       %xmm8,(%r10,%rdi,2)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8749,9 +8761,9 @@ _sk_store_4444_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            3fef <_sk_store_4444_avx+0xa3>
+  DB  119,236                             ; ja            400b <_sk_store_4444_avx+0xa3>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,66,0,0,0                  ; lea           0x42(%rip),%r9        # 4050 <_sk_store_4444_avx+0x104>
+  DB  76,141,13,66,0,0,0                  ; lea           0x42(%rip),%r9        # 406c <_sk_store_4444_avx+0x104>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8762,7 +8774,7 @@ _sk_store_4444_avx LABEL PROC
   DB  196,67,121,21,68,122,4,2            ; vpextrw       $0x2,%xmm8,0x4(%r10,%rdi,2)
   DB  196,67,121,21,68,122,2,1            ; vpextrw       $0x1,%xmm8,0x2(%r10,%rdi,2)
   DB  196,67,121,21,4,122,0               ; vpextrw       $0x0,%xmm8,(%r10,%rdi,2)
-  DB  235,159                             ; jmp           3fef <_sk_store_4444_avx+0xa3>
+  DB  235,159                             ; jmp           400b <_sk_store_4444_avx+0xa3>
   DB  247,255                             ; idiv          %edi
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
@@ -8791,12 +8803,12 @@ _sk_load_8888_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,135,0,0,0                    ; jne           4101 <_sk_load_8888_avx+0x95>
+  DB  15,133,135,0,0,0                    ; jne           411d <_sk_load_8888_avx+0x95>
   DB  196,65,124,16,12,186                ; vmovups       (%r10,%rdi,4),%ymm9
-  DB  197,124,40,21,24,40,0,0             ; vmovaps       0x2818(%rip),%ymm10        # 68a0 <_sk_callback_avx+0x57e>
+  DB  197,124,40,21,252,39,0,0            ; vmovaps       0x27fc(%rip),%ymm10        # 68a0 <_sk_callback_avx+0x562>
   DB  196,193,52,84,194                   ; vandps        %ymm10,%ymm9,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,5,146,38,0,0          ; vbroadcastss  0x2692(%rip),%ymm8        # 672c <_sk_callback_avx+0x40a>
+  DB  196,98,125,24,5,146,38,0,0          ; vbroadcastss  0x2692(%rip),%ymm8        # 6748 <_sk_callback_avx+0x40a>
   DB  196,193,124,89,192                  ; vmulps        %ymm8,%ymm0,%ymm0
   DB  196,193,113,114,209,8               ; vpsrld        $0x8,%xmm9,%xmm1
   DB  196,99,125,25,203,1                 ; vextractf128  $0x1,%ymm9,%xmm3
@@ -8823,9 +8835,9 @@ _sk_load_8888_avx LABEL PROC
   DB  196,65,52,87,201                    ; vxorps        %ymm9,%ymm9,%ymm9
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  15,135,102,255,255,255              ; ja            4080 <_sk_load_8888_avx+0x14>
+  DB  15,135,102,255,255,255              ; ja            409c <_sk_load_8888_avx+0x14>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,139,0,0,0                 ; lea           0x8b(%rip),%r9        # 41b0 <_sk_load_8888_avx+0x144>
+  DB  76,141,13,139,0,0,0                 ; lea           0x8b(%rip),%r9        # 41cc <_sk_load_8888_avx+0x144>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8848,7 +8860,7 @@ _sk_load_8888_avx LABEL PROC
   DB  196,99,53,12,200,15                 ; vblendps      $0xf,%ymm0,%ymm9,%ymm9
   DB  196,195,49,34,4,186,0               ; vpinsrd       $0x0,(%r10,%rdi,4),%xmm9,%xmm0
   DB  196,99,53,12,200,15                 ; vblendps      $0xf,%ymm0,%ymm9,%ymm9
-  DB  233,210,254,255,255                 ; jmpq          4080 <_sk_load_8888_avx+0x14>
+  DB  233,210,254,255,255                 ; jmpq          409c <_sk_load_8888_avx+0x14>
   DB  102,144                             ; xchg          %ax,%ax
   DB  236                                 ; in            (%dx),%al
   DB  255                                 ; (bad)
@@ -8866,7 +8878,7 @@ _sk_load_8888_avx LABEL PROC
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  126,255                             ; jle           41c9 <_sk_load_8888_avx+0x15d>
+  DB  126,255                             ; jle           41e5 <_sk_load_8888_avx+0x15d>
   DB  255                                 ; (bad)
   DB  255                                 ; .byte         0xff
 
@@ -8909,10 +8921,10 @@ _sk_gather_8888_avx LABEL PROC
   DB  196,131,121,34,4,152,2              ; vpinsrd       $0x2,(%r8,%r11,4),%xmm0,%xmm0
   DB  196,131,121,34,28,144,3             ; vpinsrd       $0x3,(%r8,%r10,4),%xmm0,%xmm3
   DB  196,227,61,24,195,1                 ; vinsertf128   $0x1,%xmm3,%ymm8,%ymm0
-  DB  197,124,40,21,66,38,0,0             ; vmovaps       0x2642(%rip),%ymm10        # 68c0 <_sk_callback_avx+0x59e>
+  DB  197,124,40,21,38,38,0,0             ; vmovaps       0x2626(%rip),%ymm10        # 68c0 <_sk_callback_avx+0x582>
   DB  196,193,124,84,194                  ; vandps        %ymm10,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,13,160,36,0,0         ; vbroadcastss  0x24a0(%rip),%ymm9        # 6730 <_sk_callback_avx+0x40e>
+  DB  196,98,125,24,13,160,36,0,0         ; vbroadcastss  0x24a0(%rip),%ymm9        # 674c <_sk_callback_avx+0x40e>
   DB  196,193,124,89,193                  ; vmulps        %ymm9,%ymm0,%ymm0
   DB  196,193,113,114,208,8               ; vpsrld        $0x8,%xmm8,%xmm1
   DB  197,233,114,211,8                   ; vpsrld        $0x8,%xmm3,%xmm2
@@ -8942,7 +8954,7 @@ PUBLIC _sk_store_8888_avx
 _sk_store_8888_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
-  DB  196,98,125,24,5,46,36,0,0           ; vbroadcastss  0x242e(%rip),%ymm8        # 6734 <_sk_callback_avx+0x412>
+  DB  196,98,125,24,5,46,36,0,0           ; vbroadcastss  0x242e(%rip),%ymm8        # 6750 <_sk_callback_avx+0x412>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,65,116,89,208                   ; vmulps        %ymm8,%ymm1,%ymm10
@@ -8967,7 +8979,7 @@ _sk_store_8888_avx LABEL PROC
   DB  196,65,45,86,192                    ; vorpd         %ymm8,%ymm10,%ymm8
   DB  196,65,53,86,192                    ; vorpd         %ymm8,%ymm9,%ymm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,10                              ; jne           4394 <_sk_store_8888_avx+0x9c>
+  DB  117,10                              ; jne           43b0 <_sk_store_8888_avx+0x9c>
   DB  196,65,124,17,4,186                 ; vmovups       %ymm8,(%r10,%rdi,4)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8975,9 +8987,9 @@ _sk_store_8888_avx LABEL PROC
   DB  65,128,224,7                        ; and           $0x7,%r8b
   DB  65,254,200                          ; dec           %r8b
   DB  65,128,248,6                        ; cmp           $0x6,%r8b
-  DB  119,236                             ; ja            4390 <_sk_store_8888_avx+0x98>
+  DB  119,236                             ; ja            43ac <_sk_store_8888_avx+0x98>
   DB  69,15,182,192                       ; movzbl        %r8b,%r8d
-  DB  76,141,13,85,0,0,0                  ; lea           0x55(%rip),%r9        # 4404 <_sk_store_8888_avx+0x10c>
+  DB  76,141,13,85,0,0,0                  ; lea           0x55(%rip),%r9        # 4420 <_sk_store_8888_avx+0x10c>
   DB  75,99,4,129                         ; movslq        (%r9,%r8,4),%rax
   DB  76,1,200                            ; add           %r9,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -8991,7 +9003,7 @@ _sk_store_8888_avx LABEL PROC
   DB  196,67,121,22,68,186,8,2            ; vpextrd       $0x2,%xmm8,0x8(%r10,%rdi,4)
   DB  196,67,121,22,68,186,4,1            ; vpextrd       $0x1,%xmm8,0x4(%r10,%rdi,4)
   DB  196,65,121,126,4,186                ; vmovd         %xmm8,(%r10,%rdi,4)
-  DB  235,143                             ; jmp           4390 <_sk_store_8888_avx+0x98>
+  DB  235,143                             ; jmp           43ac <_sk_store_8888_avx+0x98>
   DB  15,31,0                             ; nopl          (%rax)
   DB  245                                 ; cmc
   DB  255                                 ; (bad)
@@ -9027,7 +9039,7 @@ _sk_load_f16_avx LABEL PROC
   DB  197,252,17,116,36,64                ; vmovups       %ymm6,0x40(%rsp)
   DB  197,252,17,108,36,32                ; vmovups       %ymm5,0x20(%rsp)
   DB  197,254,127,36,36                   ; vmovdqu       %ymm4,(%rsp)
-  DB  15,133,143,2,0,0                    ; jne           46db <_sk_load_f16_avx+0x2bb>
+  DB  15,133,143,2,0,0                    ; jne           46f7 <_sk_load_f16_avx+0x2bb>
   DB  197,121,16,4,248                    ; vmovupd       (%rax,%rdi,8),%xmm8
   DB  197,249,16,84,248,16                ; vmovupd       0x10(%rax,%rdi,8),%xmm2
   DB  197,249,16,76,248,32                ; vmovupd       0x20(%rax,%rdi,8),%xmm1
@@ -9045,13 +9057,13 @@ _sk_load_f16_avx LABEL PROC
   DB  197,249,105,201                     ; vpunpckhwd    %xmm1,%xmm0,%xmm1
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
-  DB  196,98,125,24,37,147,34,0,0         ; vbroadcastss  0x2293(%rip),%ymm12        # 6738 <_sk_callback_avx+0x416>
+  DB  196,98,125,24,37,147,34,0,0         ; vbroadcastss  0x2293(%rip),%ymm12        # 6754 <_sk_callback_avx+0x416>
   DB  196,193,124,84,204                  ; vandps        %ymm12,%ymm0,%ymm1
   DB  197,252,87,193                      ; vxorps        %ymm1,%ymm0,%ymm0
   DB  196,195,125,25,198,1                ; vextractf128  $0x1,%ymm0,%xmm14
-  DB  196,98,121,24,29,127,34,0,0         ; vbroadcastss  0x227f(%rip),%xmm11        # 673c <_sk_callback_avx+0x41a>
+  DB  196,98,121,24,29,127,34,0,0         ; vbroadcastss  0x227f(%rip),%xmm11        # 6758 <_sk_callback_avx+0x41a>
   DB  196,193,8,87,219                    ; vxorps        %xmm11,%xmm14,%xmm3
-  DB  196,98,121,24,45,117,34,0,0         ; vbroadcastss  0x2275(%rip),%xmm13        # 6740 <_sk_callback_avx+0x41e>
+  DB  196,98,121,24,45,117,34,0,0         ; vbroadcastss  0x2275(%rip),%xmm13        # 675c <_sk_callback_avx+0x41e>
   DB  197,145,102,219                     ; vpcmpgtd      %xmm3,%xmm13,%xmm3
   DB  196,65,120,87,211                   ; vxorps        %xmm11,%xmm0,%xmm10
   DB  196,65,17,102,210                   ; vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -9065,7 +9077,7 @@ _sk_load_f16_avx LABEL PROC
   DB  196,227,125,24,195,1                ; vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   DB  197,252,86,193                      ; vorps         %ymm1,%ymm0,%ymm0
   DB  196,227,125,25,193,1                ; vextractf128  $0x1,%ymm0,%xmm1
-  DB  196,226,121,24,29,43,34,0,0         ; vbroadcastss  0x222b(%rip),%xmm3        # 6744 <_sk_callback_avx+0x422>
+  DB  196,226,121,24,29,43,34,0,0         ; vbroadcastss  0x222b(%rip),%xmm3        # 6760 <_sk_callback_avx+0x422>
   DB  197,241,254,203                     ; vpaddd        %xmm3,%xmm1,%xmm1
   DB  197,249,254,195                     ; vpaddd        %xmm3,%xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
@@ -9158,29 +9170,29 @@ _sk_load_f16_avx LABEL PROC
   DB  197,123,16,4,248                    ; vmovsd        (%rax,%rdi,8),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,79                              ; je            473a <_sk_load_f16_avx+0x31a>
+  DB  116,79                              ; je            4756 <_sk_load_f16_avx+0x31a>
   DB  197,57,22,68,248,8                  ; vmovhpd       0x8(%rax,%rdi,8),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,67                              ; jb            473a <_sk_load_f16_avx+0x31a>
+  DB  114,67                              ; jb            4756 <_sk_load_f16_avx+0x31a>
   DB  197,251,16,84,248,16                ; vmovsd        0x10(%rax,%rdi,8),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,68                              ; je            4747 <_sk_load_f16_avx+0x327>
+  DB  116,68                              ; je            4763 <_sk_load_f16_avx+0x327>
   DB  197,233,22,84,248,24                ; vmovhpd       0x18(%rax,%rdi,8),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,56                              ; jb            4747 <_sk_load_f16_avx+0x327>
+  DB  114,56                              ; jb            4763 <_sk_load_f16_avx+0x327>
   DB  197,251,16,76,248,32                ; vmovsd        0x20(%rax,%rdi,8),%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,68,253,255,255               ; je            4463 <_sk_load_f16_avx+0x43>
+  DB  15,132,68,253,255,255               ; je            447f <_sk_load_f16_avx+0x43>
   DB  197,241,22,76,248,40                ; vmovhpd       0x28(%rax,%rdi,8),%xmm1,%xmm1
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,52,253,255,255               ; jb            4463 <_sk_load_f16_avx+0x43>
+  DB  15,130,52,253,255,255               ; jb            447f <_sk_load_f16_avx+0x43>
   DB  197,122,126,76,248,48               ; vmovq         0x30(%rax,%rdi,8),%xmm9
-  DB  233,41,253,255,255                  ; jmpq          4463 <_sk_load_f16_avx+0x43>
+  DB  233,41,253,255,255                  ; jmpq          447f <_sk_load_f16_avx+0x43>
   DB  197,241,87,201                      ; vxorpd        %xmm1,%xmm1,%xmm1
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,28,253,255,255                  ; jmpq          4463 <_sk_load_f16_avx+0x43>
+  DB  233,28,253,255,255                  ; jmpq          447f <_sk_load_f16_avx+0x43>
   DB  197,241,87,201                      ; vxorpd        %xmm1,%xmm1,%xmm1
-  DB  233,19,253,255,255                  ; jmpq          4463 <_sk_load_f16_avx+0x43>
+  DB  233,19,253,255,255                  ; jmpq          447f <_sk_load_f16_avx+0x43>
 
 PUBLIC _sk_gather_f16_avx
 _sk_gather_f16_avx LABEL PROC
@@ -9242,13 +9254,13 @@ _sk_gather_f16_avx LABEL PROC
   DB  197,249,105,210                     ; vpunpckhwd    %xmm2,%xmm0,%xmm2
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,194,1                ; vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
-  DB  196,98,125,24,37,235,30,0,0         ; vbroadcastss  0x1eeb(%rip),%ymm12        # 6748 <_sk_callback_avx+0x426>
+  DB  196,98,125,24,37,235,30,0,0         ; vbroadcastss  0x1eeb(%rip),%ymm12        # 6764 <_sk_callback_avx+0x426>
   DB  196,193,124,84,212                  ; vandps        %ymm12,%ymm0,%ymm2
   DB  197,252,87,194                      ; vxorps        %ymm2,%ymm0,%ymm0
   DB  196,195,125,25,198,1                ; vextractf128  $0x1,%ymm0,%xmm14
-  DB  196,98,121,24,29,215,30,0,0         ; vbroadcastss  0x1ed7(%rip),%xmm11        # 674c <_sk_callback_avx+0x42a>
+  DB  196,98,121,24,29,215,30,0,0         ; vbroadcastss  0x1ed7(%rip),%xmm11        # 6768 <_sk_callback_avx+0x42a>
   DB  196,193,8,87,219                    ; vxorps        %xmm11,%xmm14,%xmm3
-  DB  196,98,121,24,45,205,30,0,0         ; vbroadcastss  0x1ecd(%rip),%xmm13        # 6750 <_sk_callback_avx+0x42e>
+  DB  196,98,121,24,45,205,30,0,0         ; vbroadcastss  0x1ecd(%rip),%xmm13        # 676c <_sk_callback_avx+0x42e>
   DB  197,145,102,219                     ; vpcmpgtd      %xmm3,%xmm13,%xmm3
   DB  196,65,120,87,211                   ; vxorps        %xmm11,%xmm0,%xmm10
   DB  196,65,17,102,210                   ; vpcmpgtd      %xmm10,%xmm13,%xmm10
@@ -9262,7 +9274,7 @@ _sk_gather_f16_avx LABEL PROC
   DB  196,227,125,24,195,1                ; vinsertf128   $0x1,%xmm3,%ymm0,%ymm0
   DB  197,252,86,194                      ; vorps         %ymm2,%ymm0,%ymm0
   DB  196,227,125,25,194,1                ; vextractf128  $0x1,%ymm0,%xmm2
-  DB  196,226,121,24,29,131,30,0,0        ; vbroadcastss  0x1e83(%rip),%xmm3        # 6754 <_sk_callback_avx+0x432>
+  DB  196,226,121,24,29,131,30,0,0        ; vbroadcastss  0x1e83(%rip),%xmm3        # 6770 <_sk_callback_avx+0x432>
   DB  197,233,254,211                     ; vpaddd        %xmm3,%xmm2,%xmm2
   DB  197,249,254,195                     ; vpaddd        %xmm3,%xmm0,%xmm0
   DB  196,227,125,24,194,1                ; vinsertf128   $0x1,%xmm2,%ymm0,%ymm0
@@ -9364,12 +9376,12 @@ _sk_store_f16_avx LABEL PROC
   DB  197,252,17,180,36,128,0,0,0         ; vmovups       %ymm6,0x80(%rsp)
   DB  197,252,17,108,36,96                ; vmovups       %ymm5,0x60(%rsp)
   DB  197,252,17,100,36,64                ; vmovups       %ymm4,0x40(%rsp)
-  DB  196,98,125,24,13,144,28,0,0         ; vbroadcastss  0x1c90(%rip),%ymm9        # 6758 <_sk_callback_avx+0x436>
+  DB  196,98,125,24,13,144,28,0,0         ; vbroadcastss  0x1c90(%rip),%ymm9        # 6774 <_sk_callback_avx+0x436>
   DB  196,65,124,84,209                   ; vandps        %ymm9,%ymm0,%ymm10
   DB  197,252,17,4,36                     ; vmovups       %ymm0,(%rsp)
   DB  196,65,124,87,218                   ; vxorps        %ymm10,%ymm0,%ymm11
   DB  196,67,125,25,220,1                 ; vextractf128  $0x1,%ymm11,%xmm12
-  DB  196,98,121,24,5,118,28,0,0          ; vbroadcastss  0x1c76(%rip),%xmm8        # 675c <_sk_callback_avx+0x43a>
+  DB  196,98,121,24,5,118,28,0,0          ; vbroadcastss  0x1c76(%rip),%xmm8        # 6778 <_sk_callback_avx+0x43a>
   DB  196,65,57,102,236                   ; vpcmpgtd      %xmm12,%xmm8,%xmm13
   DB  196,65,57,102,243                   ; vpcmpgtd      %xmm11,%xmm8,%xmm14
   DB  196,67,13,24,237,1                  ; vinsertf128   $0x1,%xmm13,%ymm14,%ymm13
@@ -9379,7 +9391,7 @@ _sk_store_f16_avx LABEL PROC
   DB  196,67,13,24,242,1                  ; vinsertf128   $0x1,%xmm10,%ymm14,%ymm14
   DB  196,193,33,114,211,13               ; vpsrld        $0xd,%xmm11,%xmm11
   DB  196,193,25,114,212,13               ; vpsrld        $0xd,%xmm12,%xmm12
-  DB  196,98,125,24,21,61,28,0,0          ; vbroadcastss  0x1c3d(%rip),%ymm10        # 6760 <_sk_callback_avx+0x43e>
+  DB  196,98,125,24,21,61,28,0,0          ; vbroadcastss  0x1c3d(%rip),%ymm10        # 677c <_sk_callback_avx+0x43e>
   DB  196,65,12,86,242                    ; vorps         %ymm10,%ymm14,%ymm14
   DB  196,67,125,25,247,1                 ; vextractf128  $0x1,%ymm14,%xmm15
   DB  196,65,1,254,228                    ; vpaddd        %xmm12,%xmm15,%xmm12
@@ -9461,7 +9473,7 @@ _sk_store_f16_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,75                              ; jne           4d0a <_sk_store_f16_avx+0x270>
+  DB  117,75                              ; jne           4d26 <_sk_store_f16_avx+0x270>
   DB  197,120,17,28,248                   ; vmovups       %xmm11,(%rax,%rdi,8)
   DB  197,120,17,84,248,16                ; vmovups       %xmm10,0x10(%rax,%rdi,8)
   DB  197,120,17,76,248,32                ; vmovups       %xmm9,0x20(%rax,%rdi,8)
@@ -9477,22 +9489,22 @@ _sk_store_f16_avx LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  197,121,214,28,248                  ; vmovq         %xmm11,(%rax,%rdi,8)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,193                             ; je            4cd6 <_sk_store_f16_avx+0x23c>
+  DB  116,193                             ; je            4cf2 <_sk_store_f16_avx+0x23c>
   DB  197,121,23,92,248,8                 ; vmovhpd       %xmm11,0x8(%rax,%rdi,8)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,181                             ; jb            4cd6 <_sk_store_f16_avx+0x23c>
+  DB  114,181                             ; jb            4cf2 <_sk_store_f16_avx+0x23c>
   DB  197,121,214,84,248,16               ; vmovq         %xmm10,0x10(%rax,%rdi,8)
-  DB  116,173                             ; je            4cd6 <_sk_store_f16_avx+0x23c>
+  DB  116,173                             ; je            4cf2 <_sk_store_f16_avx+0x23c>
   DB  197,121,23,84,248,24                ; vmovhpd       %xmm10,0x18(%rax,%rdi,8)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,161                             ; jb            4cd6 <_sk_store_f16_avx+0x23c>
+  DB  114,161                             ; jb            4cf2 <_sk_store_f16_avx+0x23c>
   DB  197,121,214,76,248,32               ; vmovq         %xmm9,0x20(%rax,%rdi,8)
-  DB  116,153                             ; je            4cd6 <_sk_store_f16_avx+0x23c>
+  DB  116,153                             ; je            4cf2 <_sk_store_f16_avx+0x23c>
   DB  197,121,23,76,248,40                ; vmovhpd       %xmm9,0x28(%rax,%rdi,8)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,141                             ; jb            4cd6 <_sk_store_f16_avx+0x23c>
+  DB  114,141                             ; jb            4cf2 <_sk_store_f16_avx+0x23c>
   DB  197,121,214,68,248,48               ; vmovq         %xmm8,0x30(%rax,%rdi,8)
-  DB  235,133                             ; jmp           4cd6 <_sk_store_f16_avx+0x23c>
+  DB  235,133                             ; jmp           4cf2 <_sk_store_f16_avx+0x23c>
 
 PUBLIC _sk_load_u16_be_avx
 _sk_load_u16_be_avx LABEL PROC
@@ -9500,7 +9512,7 @@ _sk_load_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,253,0,0,0                    ; jne           4e64 <_sk_load_u16_be_avx+0x113>
+  DB  15,133,253,0,0,0                    ; jne           4e80 <_sk_load_u16_be_avx+0x113>
   DB  196,65,121,16,4,64                  ; vmovupd       (%r8,%rax,2),%xmm8
   DB  196,193,121,16,84,64,16             ; vmovupd       0x10(%r8,%rax,2),%xmm2
   DB  196,193,121,16,92,64,32             ; vmovupd       0x20(%r8,%rax,2),%xmm3
@@ -9522,7 +9534,7 @@ _sk_load_u16_be_avx LABEL PROC
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,29,140,25,0,0         ; vbroadcastss  0x198c(%rip),%ymm11        # 6764 <_sk_callback_avx+0x442>
+  DB  196,98,125,24,29,140,25,0,0         ; vbroadcastss  0x198c(%rip),%ymm11        # 6780 <_sk_callback_avx+0x442>
   DB  196,193,124,89,195                  ; vmulps        %ymm11,%ymm0,%ymm0
   DB  197,177,109,202                     ; vpunpckhqdq   %xmm2,%xmm9,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -9556,29 +9568,29 @@ _sk_load_u16_be_avx LABEL PROC
   DB  196,65,123,16,4,64                  ; vmovsd        (%r8,%rax,2),%xmm8
   DB  196,65,49,239,201                   ; vpxor         %xmm9,%xmm9,%xmm9
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,85                              ; je            4eca <_sk_load_u16_be_avx+0x179>
+  DB  116,85                              ; je            4ee6 <_sk_load_u16_be_avx+0x179>
   DB  196,65,57,22,68,64,8                ; vmovhpd       0x8(%r8,%rax,2),%xmm8,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,72                              ; jb            4eca <_sk_load_u16_be_avx+0x179>
+  DB  114,72                              ; jb            4ee6 <_sk_load_u16_be_avx+0x179>
   DB  196,193,123,16,84,64,16             ; vmovsd        0x10(%r8,%rax,2),%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  116,72                              ; je            4ed7 <_sk_load_u16_be_avx+0x186>
+  DB  116,72                              ; je            4ef3 <_sk_load_u16_be_avx+0x186>
   DB  196,193,105,22,84,64,24             ; vmovhpd       0x18(%r8,%rax,2),%xmm2,%xmm2
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,59                              ; jb            4ed7 <_sk_load_u16_be_avx+0x186>
+  DB  114,59                              ; jb            4ef3 <_sk_load_u16_be_avx+0x186>
   DB  196,193,123,16,92,64,32             ; vmovsd        0x20(%r8,%rax,2),%xmm3
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  15,132,213,254,255,255              ; je            4d82 <_sk_load_u16_be_avx+0x31>
+  DB  15,132,213,254,255,255              ; je            4d9e <_sk_load_u16_be_avx+0x31>
   DB  196,193,97,22,92,64,40              ; vmovhpd       0x28(%r8,%rax,2),%xmm3,%xmm3
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  15,130,196,254,255,255              ; jb            4d82 <_sk_load_u16_be_avx+0x31>
+  DB  15,130,196,254,255,255              ; jb            4d9e <_sk_load_u16_be_avx+0x31>
   DB  196,65,122,126,76,64,48             ; vmovq         0x30(%r8,%rax,2),%xmm9
-  DB  233,184,254,255,255                 ; jmpq          4d82 <_sk_load_u16_be_avx+0x31>
+  DB  233,184,254,255,255                 ; jmpq          4d9e <_sk_load_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
   DB  197,233,87,210                      ; vxorpd        %xmm2,%xmm2,%xmm2
-  DB  233,171,254,255,255                 ; jmpq          4d82 <_sk_load_u16_be_avx+0x31>
+  DB  233,171,254,255,255                 ; jmpq          4d9e <_sk_load_u16_be_avx+0x31>
   DB  197,225,87,219                      ; vxorpd        %xmm3,%xmm3,%xmm3
-  DB  233,162,254,255,255                 ; jmpq          4d82 <_sk_load_u16_be_avx+0x31>
+  DB  233,162,254,255,255                 ; jmpq          4d9e <_sk_load_u16_be_avx+0x31>
 
 PUBLIC _sk_load_rgb_u16_be_avx
 _sk_load_rgb_u16_be_avx LABEL PROC
@@ -9586,7 +9598,7 @@ _sk_load_rgb_u16_be_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,127                        ; lea           (%rdi,%rdi,2),%rax
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  15,133,243,0,0,0                    ; jne           4fe5 <_sk_load_rgb_u16_be_avx+0x105>
+  DB  15,133,243,0,0,0                    ; jne           5001 <_sk_load_rgb_u16_be_avx+0x105>
   DB  196,193,122,111,4,64                ; vmovdqu       (%r8,%rax,2),%xmm0
   DB  196,193,122,111,84,64,12            ; vmovdqu       0xc(%r8,%rax,2),%xmm2
   DB  196,193,122,111,76,64,24            ; vmovdqu       0x18(%r8,%rax,2),%xmm1
@@ -9613,7 +9625,7 @@ _sk_load_rgb_u16_be_avx LABEL PROC
   DB  196,226,121,51,192                  ; vpmovzxwd     %xmm0,%xmm0
   DB  196,227,125,24,193,1                ; vinsertf128   $0x1,%xmm1,%ymm0,%ymm0
   DB  197,252,91,192                      ; vcvtdq2ps     %ymm0,%ymm0
-  DB  196,98,125,24,29,236,23,0,0         ; vbroadcastss  0x17ec(%rip),%ymm11        # 6768 <_sk_callback_avx+0x446>
+  DB  196,98,125,24,29,236,23,0,0         ; vbroadcastss  0x17ec(%rip),%ymm11        # 6784 <_sk_callback_avx+0x446>
   DB  196,193,124,89,195                  ; vmulps        %ymm11,%ymm0,%ymm0
   DB  197,185,109,202                     ; vpunpckhqdq   %xmm2,%xmm8,%xmm1
   DB  197,233,113,241,8                   ; vpsllw        $0x8,%xmm1,%xmm2
@@ -9634,48 +9646,48 @@ _sk_load_rgb_u16_be_avx LABEL PROC
   DB  197,252,91,210                      ; vcvtdq2ps     %ymm2,%ymm2
   DB  196,193,108,89,211                  ; vmulps        %ymm11,%ymm2,%ymm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,29,137,23,0,0        ; vbroadcastss  0x1789(%rip),%ymm3        # 676c <_sk_callback_avx+0x44a>
+  DB  196,226,125,24,29,137,23,0,0        ; vbroadcastss  0x1789(%rip),%ymm3        # 6788 <_sk_callback_avx+0x44a>
   DB  255,224                             ; jmpq          *%rax
   DB  196,193,121,110,4,64                ; vmovd         (%r8,%rax,2),%xmm0
   DB  196,193,121,196,68,64,4,2           ; vpinsrw       $0x2,0x4(%r8,%rax,2),%xmm0,%xmm0
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  117,5                               ; jne           4ffe <_sk_load_rgb_u16_be_avx+0x11e>
-  DB  233,40,255,255,255                  ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  117,5                               ; jne           501a <_sk_load_rgb_u16_be_avx+0x11e>
+  DB  233,40,255,255,255                  ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
   DB  196,193,121,110,76,64,6             ; vmovd         0x6(%r8,%rax,2),%xmm1
   DB  196,65,113,196,68,64,10,2           ; vpinsrw       $0x2,0xa(%r8,%rax,2),%xmm1,%xmm8
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,26                              ; jb            502d <_sk_load_rgb_u16_be_avx+0x14d>
+  DB  114,26                              ; jb            5049 <_sk_load_rgb_u16_be_avx+0x14d>
   DB  196,193,121,110,76,64,12            ; vmovd         0xc(%r8,%rax,2),%xmm1
   DB  196,193,113,196,84,64,16,2          ; vpinsrw       $0x2,0x10(%r8,%rax,2),%xmm1,%xmm2
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  117,10                              ; jne           5032 <_sk_load_rgb_u16_be_avx+0x152>
-  DB  233,249,254,255,255                 ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
-  DB  233,244,254,255,255                 ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           504e <_sk_load_rgb_u16_be_avx+0x152>
+  DB  233,249,254,255,255                 ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,244,254,255,255                 ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
   DB  196,193,121,110,76,64,18            ; vmovd         0x12(%r8,%rax,2),%xmm1
   DB  196,65,113,196,76,64,22,2           ; vpinsrw       $0x2,0x16(%r8,%rax,2),%xmm1,%xmm9
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,26                              ; jb            5061 <_sk_load_rgb_u16_be_avx+0x181>
+  DB  114,26                              ; jb            507d <_sk_load_rgb_u16_be_avx+0x181>
   DB  196,193,121,110,76,64,24            ; vmovd         0x18(%r8,%rax,2),%xmm1
   DB  196,193,113,196,76,64,28,2          ; vpinsrw       $0x2,0x1c(%r8,%rax,2),%xmm1,%xmm1
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  117,10                              ; jne           5066 <_sk_load_rgb_u16_be_avx+0x186>
-  DB  233,197,254,255,255                 ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
-  DB  233,192,254,255,255                 ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  117,10                              ; jne           5082 <_sk_load_rgb_u16_be_avx+0x186>
+  DB  233,197,254,255,255                 ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,192,254,255,255                 ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
   DB  196,193,121,110,92,64,30            ; vmovd         0x1e(%r8,%rax,2),%xmm3
   DB  196,65,97,196,92,64,34,2            ; vpinsrw       $0x2,0x22(%r8,%rax,2),%xmm3,%xmm11
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,20                              ; jb            508f <_sk_load_rgb_u16_be_avx+0x1af>
+  DB  114,20                              ; jb            50ab <_sk_load_rgb_u16_be_avx+0x1af>
   DB  196,193,121,110,92,64,36            ; vmovd         0x24(%r8,%rax,2),%xmm3
   DB  196,193,97,196,92,64,40,2           ; vpinsrw       $0x2,0x28(%r8,%rax,2),%xmm3,%xmm3
-  DB  233,151,254,255,255                 ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
-  DB  233,146,254,255,255                 ; jmpq          4f26 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,151,254,255,255                 ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
+  DB  233,146,254,255,255                 ; jmpq          4f42 <_sk_load_rgb_u16_be_avx+0x46>
 
 PUBLIC _sk_store_u16_be_avx
 _sk_store_u16_be_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  72,141,4,189,0,0,0,0                ; lea           0x0(,%rdi,4),%rax
-  DB  196,98,125,24,5,198,22,0,0          ; vbroadcastss  0x16c6(%rip),%ymm8        # 6770 <_sk_callback_avx+0x44e>
+  DB  196,98,125,24,5,198,22,0,0          ; vbroadcastss  0x16c6(%rip),%ymm8        # 678c <_sk_callback_avx+0x44e>
   DB  196,65,124,89,200                   ; vmulps        %ymm8,%ymm0,%ymm9
   DB  196,65,125,91,201                   ; vcvtps2dq     %ymm9,%ymm9
   DB  196,67,125,25,202,1                 ; vextractf128  $0x1,%ymm9,%xmm10
@@ -9713,7 +9725,7 @@ _sk_store_u16_be_avx LABEL PROC
   DB  196,65,17,98,200                    ; vpunpckldq    %xmm8,%xmm13,%xmm9
   DB  196,65,17,106,192                   ; vpunpckhdq    %xmm8,%xmm13,%xmm8
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,31                              ; jne           518e <_sk_store_u16_be_avx+0xfa>
+  DB  117,31                              ; jne           51aa <_sk_store_u16_be_avx+0xfa>
   DB  196,65,120,17,28,64                 ; vmovups       %xmm11,(%r8,%rax,2)
   DB  196,65,120,17,84,64,16              ; vmovups       %xmm10,0x10(%r8,%rax,2)
   DB  196,65,120,17,76,64,32              ; vmovups       %xmm9,0x20(%r8,%rax,2)
@@ -9722,31 +9734,31 @@ _sk_store_u16_be_avx LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,214,28,64                ; vmovq         %xmm11,(%r8,%rax,2)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            518a <_sk_store_u16_be_avx+0xf6>
+  DB  116,240                             ; je            51a6 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,23,92,64,8               ; vmovhpd       %xmm11,0x8(%r8,%rax,2)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            518a <_sk_store_u16_be_avx+0xf6>
+  DB  114,227                             ; jb            51a6 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,214,84,64,16             ; vmovq         %xmm10,0x10(%r8,%rax,2)
-  DB  116,218                             ; je            518a <_sk_store_u16_be_avx+0xf6>
+  DB  116,218                             ; je            51a6 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,23,84,64,24              ; vmovhpd       %xmm10,0x18(%r8,%rax,2)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            518a <_sk_store_u16_be_avx+0xf6>
+  DB  114,205                             ; jb            51a6 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,214,76,64,32             ; vmovq         %xmm9,0x20(%r8,%rax,2)
-  DB  116,196                             ; je            518a <_sk_store_u16_be_avx+0xf6>
+  DB  116,196                             ; je            51a6 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,23,76,64,40              ; vmovhpd       %xmm9,0x28(%r8,%rax,2)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,183                             ; jb            518a <_sk_store_u16_be_avx+0xf6>
+  DB  114,183                             ; jb            51a6 <_sk_store_u16_be_avx+0xf6>
   DB  196,65,121,214,68,64,48             ; vmovq         %xmm8,0x30(%r8,%rax,2)
-  DB  235,174                             ; jmp           518a <_sk_store_u16_be_avx+0xf6>
+  DB  235,174                             ; jmp           51a6 <_sk_store_u16_be_avx+0xf6>
 
 PUBLIC _sk_load_f32_avx
 _sk_load_f32_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  119,110                             ; ja            5252 <_sk_load_f32_avx+0x76>
+  DB  119,110                             ; ja            526e <_sk_load_f32_avx+0x76>
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,141,12,189,0,0,0,0               ; lea           0x0(,%rdi,4),%r9
-  DB  76,141,21,134,0,0,0                 ; lea           0x86(%rip),%r10        # 527c <_sk_load_f32_avx+0xa0>
+  DB  76,141,21,134,0,0,0                 ; lea           0x86(%rip),%r10        # 5298 <_sk_load_f32_avx+0xa0>
   DB  73,99,4,138                         ; movslq        (%r10,%rcx,4),%rax
   DB  76,1,208                            ; add           %r10,%rax
   DB  255,224                             ; jmpq          *%rax
@@ -9803,7 +9815,7 @@ _sk_store_f32_avx LABEL PROC
   DB  196,65,37,20,196                    ; vunpcklpd     %ymm12,%ymm11,%ymm8
   DB  196,65,37,21,220                    ; vunpckhpd     %ymm12,%ymm11,%ymm11
   DB  72,133,201                          ; test          %rcx,%rcx
-  DB  117,55                              ; jne           5309 <_sk_store_f32_avx+0x6d>
+  DB  117,55                              ; jne           5325 <_sk_store_f32_avx+0x6d>
   DB  196,67,45,24,225,1                  ; vinsertf128   $0x1,%xmm9,%ymm10,%ymm12
   DB  196,67,61,24,235,1                  ; vinsertf128   $0x1,%xmm11,%ymm8,%ymm13
   DB  196,67,45,6,201,49                  ; vperm2f128    $0x31,%ymm9,%ymm10,%ymm9
@@ -9816,22 +9828,22 @@ _sk_store_f32_avx LABEL PROC
   DB  255,224                             ; jmpq          *%rax
   DB  196,65,121,17,20,128                ; vmovupd       %xmm10,(%r8,%rax,4)
   DB  72,131,249,1                        ; cmp           $0x1,%rcx
-  DB  116,240                             ; je            5305 <_sk_store_f32_avx+0x69>
+  DB  116,240                             ; je            5321 <_sk_store_f32_avx+0x69>
   DB  196,65,121,17,76,128,16             ; vmovupd       %xmm9,0x10(%r8,%rax,4)
   DB  72,131,249,3                        ; cmp           $0x3,%rcx
-  DB  114,227                             ; jb            5305 <_sk_store_f32_avx+0x69>
+  DB  114,227                             ; jb            5321 <_sk_store_f32_avx+0x69>
   DB  196,65,121,17,68,128,32             ; vmovupd       %xmm8,0x20(%r8,%rax,4)
-  DB  116,218                             ; je            5305 <_sk_store_f32_avx+0x69>
+  DB  116,218                             ; je            5321 <_sk_store_f32_avx+0x69>
   DB  196,65,121,17,92,128,48             ; vmovupd       %xmm11,0x30(%r8,%rax,4)
   DB  72,131,249,5                        ; cmp           $0x5,%rcx
-  DB  114,205                             ; jb            5305 <_sk_store_f32_avx+0x69>
+  DB  114,205                             ; jb            5321 <_sk_store_f32_avx+0x69>
   DB  196,67,125,25,84,128,64,1           ; vextractf128  $0x1,%ymm10,0x40(%r8,%rax,4)
-  DB  116,195                             ; je            5305 <_sk_store_f32_avx+0x69>
+  DB  116,195                             ; je            5321 <_sk_store_f32_avx+0x69>
   DB  196,67,125,25,76,128,80,1           ; vextractf128  $0x1,%ymm9,0x50(%r8,%rax,4)
   DB  72,131,249,7                        ; cmp           $0x7,%rcx
-  DB  114,181                             ; jb            5305 <_sk_store_f32_avx+0x69>
+  DB  114,181                             ; jb            5321 <_sk_store_f32_avx+0x69>
   DB  196,67,125,25,68,128,96,1           ; vextractf128  $0x1,%ymm8,0x60(%r8,%rax,4)
-  DB  235,171                             ; jmp           5305 <_sk_store_f32_avx+0x69>
+  DB  235,171                             ; jmp           5321 <_sk_store_f32_avx+0x69>
 
 PUBLIC _sk_clamp_x_avx
 _sk_clamp_x_avx LABEL PROC
@@ -9923,12 +9935,12 @@ _sk_mirror_y_avx LABEL PROC
 
 PUBLIC _sk_luminance_to_alpha_avx
 _sk_luminance_to_alpha_avx LABEL PROC
-  DB  196,226,125,24,29,235,18,0,0        ; vbroadcastss  0x12eb(%rip),%ymm3        # 6774 <_sk_callback_avx+0x452>
+  DB  196,226,125,24,29,235,18,0,0        ; vbroadcastss  0x12eb(%rip),%ymm3        # 6790 <_sk_callback_avx+0x452>
   DB  197,252,89,195                      ; vmulps        %ymm3,%ymm0,%ymm0
-  DB  196,226,125,24,29,226,18,0,0        ; vbroadcastss  0x12e2(%rip),%ymm3        # 6778 <_sk_callback_avx+0x456>
+  DB  196,226,125,24,29,226,18,0,0        ; vbroadcastss  0x12e2(%rip),%ymm3        # 6794 <_sk_callback_avx+0x456>
   DB  197,244,89,203                      ; vmulps        %ymm3,%ymm1,%ymm1
   DB  197,252,88,193                      ; vaddps        %ymm1,%ymm0,%ymm0
-  DB  196,226,125,24,13,213,18,0,0        ; vbroadcastss  0x12d5(%rip),%ymm1        # 677c <_sk_callback_avx+0x45a>
+  DB  196,226,125,24,13,213,18,0,0        ; vbroadcastss  0x12d5(%rip),%ymm1        # 6798 <_sk_callback_avx+0x45a>
   DB  197,236,89,201                      ; vmulps        %ymm1,%ymm2,%ymm1
   DB  197,252,88,217                      ; vaddps        %ymm1,%ymm0,%ymm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10099,9 +10111,9 @@ _sk_evenly_spaced_gradient_avx LABEL PROC
   DB  72,139,24                           ; mov           (%rax),%rbx
   DB  72,139,104,8                        ; mov           0x8(%rax),%rbp
   DB  72,255,203                          ; dec           %rbx
-  DB  120,7                               ; js            5764 <_sk_evenly_spaced_gradient_avx+0x1f>
+  DB  120,7                               ; js            5780 <_sk_evenly_spaced_gradient_avx+0x1f>
   DB  196,225,242,42,203                  ; vcvtsi2ss     %rbx,%xmm1,%xmm1
-  DB  235,21                              ; jmp           5779 <_sk_evenly_spaced_gradient_avx+0x34>
+  DB  235,21                              ; jmp           5795 <_sk_evenly_spaced_gradient_avx+0x34>
   DB  73,137,216                          ; mov           %rbx,%r8
   DB  73,209,232                          ; shr           %r8
   DB  131,227,1                           ; and           $0x1,%ebx
@@ -10266,12 +10278,12 @@ _sk_gradient_avx LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  197,244,87,201                      ; vxorps        %ymm1,%ymm1,%ymm1
   DB  73,131,248,2                        ; cmp           $0x2,%r8
-  DB  114,80                              ; jb            5b07 <_sk_gradient_avx+0x69>
+  DB  114,80                              ; jb            5b23 <_sk_gradient_avx+0x69>
   DB  72,139,88,72                        ; mov           0x48(%rax),%rbx
   DB  73,255,200                          ; dec           %r8
   DB  72,131,195,4                        ; add           $0x4,%rbx
   DB  196,65,52,87,201                    ; vxorps        %ymm9,%ymm9,%ymm9
-  DB  196,98,125,24,21,176,12,0,0         ; vbroadcastss  0xcb0(%rip),%ymm10        # 6780 <_sk_callback_avx+0x45e>
+  DB  196,98,125,24,21,176,12,0,0         ; vbroadcastss  0xcb0(%rip),%ymm10        # 679c <_sk_callback_avx+0x45e>
   DB  197,244,87,201                      ; vxorps        %ymm1,%ymm1,%ymm1
   DB  196,98,125,24,3                     ; vbroadcastss  (%rbx),%ymm8
   DB  197,60,194,192,2                    ; vcmpleps      %ymm0,%ymm8,%ymm8
@@ -10283,7 +10295,7 @@ _sk_gradient_avx LABEL PROC
   DB  196,227,117,24,202,1                ; vinsertf128   $0x1,%xmm2,%ymm1,%ymm1
   DB  72,131,195,4                        ; add           $0x4,%rbx
   DB  73,255,200                          ; dec           %r8
-  DB  117,205                             ; jne           5ad4 <_sk_gradient_avx+0x36>
+  DB  117,205                             ; jne           5af0 <_sk_gradient_avx+0x36>
   DB  196,195,249,22,200,1                ; vpextrq       $0x1,%xmm1,%r8
   DB  69,137,193                          ; mov           %r8d,%r9d
   DB  73,193,232,32                       ; shr           $0x20,%r8
@@ -10461,27 +10473,27 @@ _sk_xy_to_unit_angle_avx LABEL PROC
   DB  196,65,52,95,226                    ; vmaxps        %ymm10,%ymm9,%ymm12
   DB  196,65,36,94,220                    ; vdivps        %ymm12,%ymm11,%ymm11
   DB  196,65,36,89,227                    ; vmulps        %ymm11,%ymm11,%ymm12
-  DB  196,98,125,24,45,212,8,0,0          ; vbroadcastss  0x8d4(%rip),%ymm13        # 6784 <_sk_callback_avx+0x462>
+  DB  196,98,125,24,45,212,8,0,0          ; vbroadcastss  0x8d4(%rip),%ymm13        # 67a0 <_sk_callback_avx+0x462>
   DB  196,65,28,89,237                    ; vmulps        %ymm13,%ymm12,%ymm13
-  DB  196,98,125,24,53,202,8,0,0          ; vbroadcastss  0x8ca(%rip),%ymm14        # 6788 <_sk_callback_avx+0x466>
+  DB  196,98,125,24,53,202,8,0,0          ; vbroadcastss  0x8ca(%rip),%ymm14        # 67a4 <_sk_callback_avx+0x466>
   DB  196,65,20,88,238                    ; vaddps        %ymm14,%ymm13,%ymm13
   DB  196,65,28,89,237                    ; vmulps        %ymm13,%ymm12,%ymm13
-  DB  196,98,125,24,53,187,8,0,0          ; vbroadcastss  0x8bb(%rip),%ymm14        # 678c <_sk_callback_avx+0x46a>
+  DB  196,98,125,24,53,187,8,0,0          ; vbroadcastss  0x8bb(%rip),%ymm14        # 67a8 <_sk_callback_avx+0x46a>
   DB  196,65,20,88,238                    ; vaddps        %ymm14,%ymm13,%ymm13
   DB  196,65,28,89,229                    ; vmulps        %ymm13,%ymm12,%ymm12
-  DB  196,98,125,24,45,172,8,0,0          ; vbroadcastss  0x8ac(%rip),%ymm13        # 6790 <_sk_callback_avx+0x46e>
+  DB  196,98,125,24,45,172,8,0,0          ; vbroadcastss  0x8ac(%rip),%ymm13        # 67ac <_sk_callback_avx+0x46e>
   DB  196,65,28,88,229                    ; vaddps        %ymm13,%ymm12,%ymm12
   DB  196,65,36,89,220                    ; vmulps        %ymm12,%ymm11,%ymm11
   DB  196,65,52,194,202,1                 ; vcmpltps      %ymm10,%ymm9,%ymm9
-  DB  196,98,125,24,21,151,8,0,0          ; vbroadcastss  0x897(%rip),%ymm10        # 6794 <_sk_callback_avx+0x472>
+  DB  196,98,125,24,21,151,8,0,0          ; vbroadcastss  0x897(%rip),%ymm10        # 67b0 <_sk_callback_avx+0x472>
   DB  196,65,44,92,211                    ; vsubps        %ymm11,%ymm10,%ymm10
   DB  196,67,37,74,202,144                ; vblendvps     %ymm9,%ymm10,%ymm11,%ymm9
   DB  196,193,124,194,192,1               ; vcmpltps      %ymm8,%ymm0,%ymm0
-  DB  196,98,125,24,21,129,8,0,0          ; vbroadcastss  0x881(%rip),%ymm10        # 6798 <_sk_callback_avx+0x476>
+  DB  196,98,125,24,21,129,8,0,0          ; vbroadcastss  0x881(%rip),%ymm10        # 67b4 <_sk_callback_avx+0x476>
   DB  196,65,44,92,209                    ; vsubps        %ymm9,%ymm10,%ymm10
   DB  196,195,53,74,194,0                 ; vblendvps     %ymm0,%ymm10,%ymm9,%ymm0
   DB  196,65,116,194,200,1                ; vcmpltps      %ymm8,%ymm1,%ymm9
-  DB  196,98,125,24,21,107,8,0,0          ; vbroadcastss  0x86b(%rip),%ymm10        # 679c <_sk_callback_avx+0x47a>
+  DB  196,98,125,24,21,107,8,0,0          ; vbroadcastss  0x86b(%rip),%ymm10        # 67b8 <_sk_callback_avx+0x47a>
   DB  197,44,92,208                       ; vsubps        %ymm0,%ymm10,%ymm10
   DB  196,195,125,74,194,144              ; vblendvps     %ymm9,%ymm10,%ymm0,%ymm0
   DB  196,65,124,194,200,3                ; vcmpunordps   %ymm8,%ymm0,%ymm9
@@ -10501,7 +10513,7 @@ _sk_xy_to_radius_avx LABEL PROC
 PUBLIC _sk_save_xy_avx
 _sk_save_xy_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,53,8,0,0            ; vbroadcastss  0x835(%rip),%ymm8        # 67a0 <_sk_callback_avx+0x47e>
+  DB  196,98,125,24,5,53,8,0,0            ; vbroadcastss  0x835(%rip),%ymm8        # 67bc <_sk_callback_avx+0x47e>
   DB  196,65,124,88,200                   ; vaddps        %ymm8,%ymm0,%ymm9
   DB  196,67,125,8,209,1                  ; vroundps      $0x1,%ymm9,%ymm10
   DB  196,65,52,92,202                    ; vsubps        %ymm10,%ymm9,%ymm9
@@ -10534,9 +10546,9 @@ _sk_accumulate_avx LABEL PROC
 PUBLIC _sk_bilinear_nx_avx
 _sk_bilinear_nx_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,193,7,0,0          ; vbroadcastss  0x7c1(%rip),%ymm0        # 67a4 <_sk_callback_avx+0x482>
+  DB  196,226,125,24,5,193,7,0,0          ; vbroadcastss  0x7c1(%rip),%ymm0        # 67c0 <_sk_callback_avx+0x482>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,184,7,0,0           ; vbroadcastss  0x7b8(%rip),%ymm8        # 67a8 <_sk_callback_avx+0x486>
+  DB  196,98,125,24,5,184,7,0,0           ; vbroadcastss  0x7b8(%rip),%ymm8        # 67c4 <_sk_callback_avx+0x486>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10545,7 +10557,7 @@ _sk_bilinear_nx_avx LABEL PROC
 PUBLIC _sk_bilinear_px_avx
 _sk_bilinear_px_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,160,7,0,0          ; vbroadcastss  0x7a0(%rip),%ymm0        # 67ac <_sk_callback_avx+0x48a>
+  DB  196,226,125,24,5,160,7,0,0          ; vbroadcastss  0x7a0(%rip),%ymm0        # 67c8 <_sk_callback_avx+0x48a>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -10555,9 +10567,9 @@ _sk_bilinear_px_avx LABEL PROC
 PUBLIC _sk_bilinear_ny_avx
 _sk_bilinear_ny_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,132,7,0,0         ; vbroadcastss  0x784(%rip),%ymm1        # 67b0 <_sk_callback_avx+0x48e>
+  DB  196,226,125,24,13,132,7,0,0         ; vbroadcastss  0x784(%rip),%ymm1        # 67cc <_sk_callback_avx+0x48e>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,122,7,0,0           ; vbroadcastss  0x77a(%rip),%ymm8        # 67b4 <_sk_callback_avx+0x492>
+  DB  196,98,125,24,5,122,7,0,0           ; vbroadcastss  0x77a(%rip),%ymm8        # 67d0 <_sk_callback_avx+0x492>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10566,7 +10578,7 @@ _sk_bilinear_ny_avx LABEL PROC
 PUBLIC _sk_bilinear_py_avx
 _sk_bilinear_py_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,98,7,0,0          ; vbroadcastss  0x762(%rip),%ymm1        # 67b8 <_sk_callback_avx+0x496>
+  DB  196,226,125,24,13,98,7,0,0          ; vbroadcastss  0x762(%rip),%ymm1        # 67d4 <_sk_callback_avx+0x496>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -10576,14 +10588,14 @@ _sk_bilinear_py_avx LABEL PROC
 PUBLIC _sk_bicubic_n3x_avx
 _sk_bicubic_n3x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,69,7,0,0           ; vbroadcastss  0x745(%rip),%ymm0        # 67bc <_sk_callback_avx+0x49a>
+  DB  196,226,125,24,5,69,7,0,0           ; vbroadcastss  0x745(%rip),%ymm0        # 67d8 <_sk_callback_avx+0x49a>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,60,7,0,0            ; vbroadcastss  0x73c(%rip),%ymm8        # 67c0 <_sk_callback_avx+0x49e>
+  DB  196,98,125,24,5,60,7,0,0            ; vbroadcastss  0x73c(%rip),%ymm8        # 67dc <_sk_callback_avx+0x49e>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,45,7,0,0           ; vbroadcastss  0x72d(%rip),%ymm10        # 67c4 <_sk_callback_avx+0x4a2>
+  DB  196,98,125,24,21,45,7,0,0           ; vbroadcastss  0x72d(%rip),%ymm10        # 67e0 <_sk_callback_avx+0x4a2>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,35,7,0,0           ; vbroadcastss  0x723(%rip),%ymm10        # 67c8 <_sk_callback_avx+0x4a6>
+  DB  196,98,125,24,21,35,7,0,0           ; vbroadcastss  0x723(%rip),%ymm10        # 67e4 <_sk_callback_avx+0x4a6>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -10593,19 +10605,19 @@ _sk_bicubic_n3x_avx LABEL PROC
 PUBLIC _sk_bicubic_n1x_avx
 _sk_bicubic_n1x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,6,7,0,0            ; vbroadcastss  0x706(%rip),%ymm0        # 67cc <_sk_callback_avx+0x4aa>
+  DB  196,226,125,24,5,6,7,0,0            ; vbroadcastss  0x706(%rip),%ymm0        # 67e8 <_sk_callback_avx+0x4aa>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
-  DB  196,98,125,24,5,253,6,0,0           ; vbroadcastss  0x6fd(%rip),%ymm8        # 67d0 <_sk_callback_avx+0x4ae>
+  DB  196,98,125,24,5,253,6,0,0           ; vbroadcastss  0x6fd(%rip),%ymm8        # 67ec <_sk_callback_avx+0x4ae>
   DB  197,60,92,64,64                     ; vsubps        0x40(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,243,6,0,0          ; vbroadcastss  0x6f3(%rip),%ymm9        # 67d4 <_sk_callback_avx+0x4b2>
+  DB  196,98,125,24,13,243,6,0,0          ; vbroadcastss  0x6f3(%rip),%ymm9        # 67f0 <_sk_callback_avx+0x4b2>
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,233,6,0,0          ; vbroadcastss  0x6e9(%rip),%ymm10        # 67d8 <_sk_callback_avx+0x4b6>
+  DB  196,98,125,24,21,233,6,0,0          ; vbroadcastss  0x6e9(%rip),%ymm10        # 67f4 <_sk_callback_avx+0x4b6>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,218,6,0,0          ; vbroadcastss  0x6da(%rip),%ymm10        # 67dc <_sk_callback_avx+0x4ba>
+  DB  196,98,125,24,21,218,6,0,0          ; vbroadcastss  0x6da(%rip),%ymm10        # 67f8 <_sk_callback_avx+0x4ba>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,203,6,0,0          ; vbroadcastss  0x6cb(%rip),%ymm9        # 67e0 <_sk_callback_avx+0x4be>
+  DB  196,98,125,24,13,203,6,0,0          ; vbroadcastss  0x6cb(%rip),%ymm9        # 67fc <_sk_callback_avx+0x4be>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10614,17 +10626,17 @@ _sk_bicubic_n1x_avx LABEL PROC
 PUBLIC _sk_bicubic_p1x_avx
 _sk_bicubic_p1x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,179,6,0,0           ; vbroadcastss  0x6b3(%rip),%ymm8        # 67e4 <_sk_callback_avx+0x4c2>
+  DB  196,98,125,24,5,179,6,0,0           ; vbroadcastss  0x6b3(%rip),%ymm8        # 6800 <_sk_callback_avx+0x4c2>
   DB  197,188,88,0                        ; vaddps        (%rax),%ymm8,%ymm0
   DB  197,124,16,72,64                    ; vmovups       0x40(%rax),%ymm9
-  DB  196,98,125,24,21,165,6,0,0          ; vbroadcastss  0x6a5(%rip),%ymm10        # 67e8 <_sk_callback_avx+0x4c6>
+  DB  196,98,125,24,21,165,6,0,0          ; vbroadcastss  0x6a5(%rip),%ymm10        # 6804 <_sk_callback_avx+0x4c6>
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
-  DB  196,98,125,24,29,155,6,0,0          ; vbroadcastss  0x69b(%rip),%ymm11        # 67ec <_sk_callback_avx+0x4ca>
+  DB  196,98,125,24,29,155,6,0,0          ; vbroadcastss  0x69b(%rip),%ymm11        # 6808 <_sk_callback_avx+0x4ca>
   DB  196,65,44,88,211                    ; vaddps        %ymm11,%ymm10,%ymm10
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
   DB  196,65,44,88,192                    ; vaddps        %ymm8,%ymm10,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
-  DB  196,98,125,24,13,130,6,0,0          ; vbroadcastss  0x682(%rip),%ymm9        # 67f0 <_sk_callback_avx+0x4ce>
+  DB  196,98,125,24,13,130,6,0,0          ; vbroadcastss  0x682(%rip),%ymm9        # 680c <_sk_callback_avx+0x4ce>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10633,13 +10645,13 @@ _sk_bicubic_p1x_avx LABEL PROC
 PUBLIC _sk_bicubic_p3x_avx
 _sk_bicubic_p3x_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,5,106,6,0,0          ; vbroadcastss  0x66a(%rip),%ymm0        # 67f4 <_sk_callback_avx+0x4d2>
+  DB  196,226,125,24,5,106,6,0,0          ; vbroadcastss  0x66a(%rip),%ymm0        # 6810 <_sk_callback_avx+0x4d2>
   DB  197,252,88,0                        ; vaddps        (%rax),%ymm0,%ymm0
   DB  197,124,16,64,64                    ; vmovups       0x40(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,87,6,0,0           ; vbroadcastss  0x657(%rip),%ymm10        # 67f8 <_sk_callback_avx+0x4d6>
+  DB  196,98,125,24,21,87,6,0,0           ; vbroadcastss  0x657(%rip),%ymm10        # 6814 <_sk_callback_avx+0x4d6>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,77,6,0,0           ; vbroadcastss  0x64d(%rip),%ymm10        # 67fc <_sk_callback_avx+0x4da>
+  DB  196,98,125,24,21,77,6,0,0           ; vbroadcastss  0x64d(%rip),%ymm10        # 6818 <_sk_callback_avx+0x4da>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,128,0,0,0            ; vmovups       %ymm8,0x80(%rax)
@@ -10649,14 +10661,14 @@ _sk_bicubic_p3x_avx LABEL PROC
 PUBLIC _sk_bicubic_n3y_avx
 _sk_bicubic_n3y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,48,6,0,0          ; vbroadcastss  0x630(%rip),%ymm1        # 6800 <_sk_callback_avx+0x4de>
+  DB  196,226,125,24,13,48,6,0,0          ; vbroadcastss  0x630(%rip),%ymm1        # 681c <_sk_callback_avx+0x4de>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,38,6,0,0            ; vbroadcastss  0x626(%rip),%ymm8        # 6804 <_sk_callback_avx+0x4e2>
+  DB  196,98,125,24,5,38,6,0,0            ; vbroadcastss  0x626(%rip),%ymm8        # 6820 <_sk_callback_avx+0x4e2>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,23,6,0,0           ; vbroadcastss  0x617(%rip),%ymm10        # 6808 <_sk_callback_avx+0x4e6>
+  DB  196,98,125,24,21,23,6,0,0           ; vbroadcastss  0x617(%rip),%ymm10        # 6824 <_sk_callback_avx+0x4e6>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,13,6,0,0           ; vbroadcastss  0x60d(%rip),%ymm10        # 680c <_sk_callback_avx+0x4ea>
+  DB  196,98,125,24,21,13,6,0,0           ; vbroadcastss  0x60d(%rip),%ymm10        # 6828 <_sk_callback_avx+0x4ea>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -10666,19 +10678,19 @@ _sk_bicubic_n3y_avx LABEL PROC
 PUBLIC _sk_bicubic_n1y_avx
 _sk_bicubic_n1y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,240,5,0,0         ; vbroadcastss  0x5f0(%rip),%ymm1        # 6810 <_sk_callback_avx+0x4ee>
+  DB  196,226,125,24,13,240,5,0,0         ; vbroadcastss  0x5f0(%rip),%ymm1        # 682c <_sk_callback_avx+0x4ee>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
-  DB  196,98,125,24,5,230,5,0,0           ; vbroadcastss  0x5e6(%rip),%ymm8        # 6814 <_sk_callback_avx+0x4f2>
+  DB  196,98,125,24,5,230,5,0,0           ; vbroadcastss  0x5e6(%rip),%ymm8        # 6830 <_sk_callback_avx+0x4f2>
   DB  197,60,92,64,96                     ; vsubps        0x60(%rax),%ymm8,%ymm8
-  DB  196,98,125,24,13,220,5,0,0          ; vbroadcastss  0x5dc(%rip),%ymm9        # 6818 <_sk_callback_avx+0x4f6>
+  DB  196,98,125,24,13,220,5,0,0          ; vbroadcastss  0x5dc(%rip),%ymm9        # 6834 <_sk_callback_avx+0x4f6>
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,210,5,0,0          ; vbroadcastss  0x5d2(%rip),%ymm10        # 681c <_sk_callback_avx+0x4fa>
+  DB  196,98,125,24,21,210,5,0,0          ; vbroadcastss  0x5d2(%rip),%ymm10        # 6838 <_sk_callback_avx+0x4fa>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,201                    ; vmulps        %ymm9,%ymm8,%ymm9
-  DB  196,98,125,24,21,195,5,0,0          ; vbroadcastss  0x5c3(%rip),%ymm10        # 6820 <_sk_callback_avx+0x4fe>
+  DB  196,98,125,24,21,195,5,0,0          ; vbroadcastss  0x5c3(%rip),%ymm10        # 683c <_sk_callback_avx+0x4fe>
   DB  196,65,52,88,202                    ; vaddps        %ymm10,%ymm9,%ymm9
   DB  196,65,60,89,193                    ; vmulps        %ymm9,%ymm8,%ymm8
-  DB  196,98,125,24,13,180,5,0,0          ; vbroadcastss  0x5b4(%rip),%ymm9        # 6824 <_sk_callback_avx+0x502>
+  DB  196,98,125,24,13,180,5,0,0          ; vbroadcastss  0x5b4(%rip),%ymm9        # 6840 <_sk_callback_avx+0x502>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10687,17 +10699,17 @@ _sk_bicubic_n1y_avx LABEL PROC
 PUBLIC _sk_bicubic_p1y_avx
 _sk_bicubic_p1y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,98,125,24,5,156,5,0,0           ; vbroadcastss  0x59c(%rip),%ymm8        # 6828 <_sk_callback_avx+0x506>
+  DB  196,98,125,24,5,156,5,0,0           ; vbroadcastss  0x59c(%rip),%ymm8        # 6844 <_sk_callback_avx+0x506>
   DB  197,188,88,72,32                    ; vaddps        0x20(%rax),%ymm8,%ymm1
   DB  197,124,16,72,96                    ; vmovups       0x60(%rax),%ymm9
-  DB  196,98,125,24,21,141,5,0,0          ; vbroadcastss  0x58d(%rip),%ymm10        # 682c <_sk_callback_avx+0x50a>
+  DB  196,98,125,24,21,141,5,0,0          ; vbroadcastss  0x58d(%rip),%ymm10        # 6848 <_sk_callback_avx+0x50a>
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
-  DB  196,98,125,24,29,131,5,0,0          ; vbroadcastss  0x583(%rip),%ymm11        # 6830 <_sk_callback_avx+0x50e>
+  DB  196,98,125,24,29,131,5,0,0          ; vbroadcastss  0x583(%rip),%ymm11        # 684c <_sk_callback_avx+0x50e>
   DB  196,65,44,88,211                    ; vaddps        %ymm11,%ymm10,%ymm10
   DB  196,65,52,89,210                    ; vmulps        %ymm10,%ymm9,%ymm10
   DB  196,65,44,88,192                    ; vaddps        %ymm8,%ymm10,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
-  DB  196,98,125,24,13,106,5,0,0          ; vbroadcastss  0x56a(%rip),%ymm9        # 6834 <_sk_callback_avx+0x512>
+  DB  196,98,125,24,13,106,5,0,0          ; vbroadcastss  0x56a(%rip),%ymm9        # 6850 <_sk_callback_avx+0x512>
   DB  196,65,60,88,193                    ; vaddps        %ymm9,%ymm8,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -10706,13 +10718,13 @@ _sk_bicubic_p1y_avx LABEL PROC
 PUBLIC _sk_bicubic_p3y_avx
 _sk_bicubic_p3y_avx LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  196,226,125,24,13,82,5,0,0          ; vbroadcastss  0x552(%rip),%ymm1        # 6838 <_sk_callback_avx+0x516>
+  DB  196,226,125,24,13,82,5,0,0          ; vbroadcastss  0x552(%rip),%ymm1        # 6854 <_sk_callback_avx+0x516>
   DB  197,244,88,72,32                    ; vaddps        0x20(%rax),%ymm1,%ymm1
   DB  197,124,16,64,96                    ; vmovups       0x60(%rax),%ymm8
   DB  196,65,60,89,200                    ; vmulps        %ymm8,%ymm8,%ymm9
-  DB  196,98,125,24,21,62,5,0,0           ; vbroadcastss  0x53e(%rip),%ymm10        # 683c <_sk_callback_avx+0x51a>
+  DB  196,98,125,24,21,62,5,0,0           ; vbroadcastss  0x53e(%rip),%ymm10        # 6858 <_sk_callback_avx+0x51a>
   DB  196,65,60,89,194                    ; vmulps        %ymm10,%ymm8,%ymm8
-  DB  196,98,125,24,21,52,5,0,0           ; vbroadcastss  0x534(%rip),%ymm10        # 6840 <_sk_callback_avx+0x51e>
+  DB  196,98,125,24,21,52,5,0,0           ; vbroadcastss  0x534(%rip),%ymm10        # 685c <_sk_callback_avx+0x51e>
   DB  196,65,60,88,194                    ; vaddps        %ymm10,%ymm8,%ymm8
   DB  196,65,52,89,192                    ; vmulps        %ymm8,%ymm9,%ymm8
   DB  197,124,17,128,160,0,0,0            ; vmovups       %ymm8,0xa0(%rax)
@@ -10826,25 +10838,25 @@ ALIGN 4
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 64ed <.literal4+0xb1>
+  DB  71,225,61                           ; rex.RXB       loope 6509 <.literal4+0xb1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 64fd <.literal4+0xc1>
+  DB  71,225,61                           ; rex.RXB       loope 6519 <.literal4+0xc1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 650d <.literal4+0xd1>
+  DB  71,225,61                           ; rex.RXB       loope 6529 <.literal4+0xd1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,154                          ; cmpb          $0x9a,(%rdi)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
   DB  62,61,10,23,63,174                  ; ds            cmp $0xae3f170a,%eax
-  DB  71,225,61                           ; rex.RXB       loope 651d <.literal4+0xe1>
+  DB  71,225,61                           ; rex.RXB       loope 6539 <.literal4+0xe1>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -10894,7 +10906,7 @@ ALIGN 4
   DB  190,129,128,128,59                  ; mov           $0x3b808081,%esi
   DB  129,128,128,59,0,248,0,0,8,33       ; addl          $0x21080000,-0x7ffc480(%rax)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        6569 <.literal4+0x12d>
+  DB  224,7                               ; loopne        6585 <.literal4+0x12d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -10910,10 +10922,10 @@ ALIGN 4
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
   DB  0,52,255                            ; add           %dh,(%rdi,%rdi,8)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            6590 <.literal4+0x154>
+  DB  127,0                               ; jg            65ac <.literal4+0x154>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            6609 <.literal4+0x1cd>
+  DB  119,115                             ; ja            6625 <.literal4+0x1cd>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10927,10 +10939,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            65c4 <.literal4+0x188>
+  DB  127,0                               ; jg            65e0 <.literal4+0x188>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            663d <.literal4+0x201>
+  DB  119,115                             ; ja            6659 <.literal4+0x201>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10944,10 +10956,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            65f8 <.literal4+0x1bc>
+  DB  127,0                               ; jg            6614 <.literal4+0x1bc>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            6671 <.literal4+0x235>
+  DB  119,115                             ; ja            668d <.literal4+0x235>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10961,10 +10973,10 @@ ALIGN 4
   DB  0,128,63,0,0,0                      ; add           %al,0x3f(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            662c <.literal4+0x1f0>
+  DB  127,0                               ; jg            6648 <.literal4+0x1f0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            66a5 <.literal4+0x269>
+  DB  119,115                             ; ja            66c1 <.literal4+0x269>
   DB  248                                 ; clc
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,249,68,180                   ; mov           $0xb444f93f,%edi
@@ -10977,7 +10989,7 @@ ALIGN 4
   DB  0,75,0                              ; add           %cl,0x0(%rbx)
   DB  0,128,63,0,0,200                    ; add           %al,-0x37ffffc1(%rax)
   DB  66,0,0                              ; rex.X         add %al,(%rax)
-  DB  127,67                              ; jg            66a3 <.literal4+0x267>
+  DB  127,67                              ; jg            66bf <.literal4+0x267>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -10989,10 +11001,10 @@ ALIGN 4
   DB  190,80,128,3,62                     ; mov           $0x3e038050,%esi
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           66c3 <.literal4+0x287>
+  DB  118,63                              ; jbe           66df <.literal4+0x287>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            66d7 <.literal4+0x29b>
+  DB  127,67                              ; jg            66f3 <.literal4+0x29b>
   DB  129,128,128,59,0,0,128,63,129,128   ; addl          $0x80813f80,0x3b80(%rax)
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,128,63,129,128,128                ; add           %al,-0x7f7f7ec1(%rax)
@@ -11001,7 +11013,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        66b9 <.literal4+0x27d>
+  DB  224,7                               ; loopne        66d5 <.literal4+0x27d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -11013,7 +11025,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        66d5 <.literal4+0x299>
+  DB  224,7                               ; loopne        66f1 <.literal4+0x299>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -11024,7 +11036,7 @@ ALIGN 4
   DB  0,0                                 ; add           %al,(%rax)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            672a <.literal4+0x2ee>
+  DB  124,66                              ; jl            6746 <.literal4+0x2ee>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,55,0,15                 ; mov           %ecx,0xf003788(%rax)
@@ -11042,9 +11054,9 @@ ALIGN 4
   DB  137,136,136,59,15,0                 ; mov           %ecx,0xf3b88(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  137,136,136,61,0,0                  ; mov           %ecx,0x3d88(%rax)
-  DB  112,65                              ; jo            676d <.literal4+0x331>
+  DB  112,65                              ; jo            6789 <.literal4+0x331>
   DB  129,128,128,59,129,128,128,59,0,0   ; addl          $0x3b80,-0x7f7ec480(%rax)
-  DB  127,67                              ; jg            677b <.literal4+0x33f>
+  DB  127,67                              ; jg            6797 <.literal4+0x33f>
   DB  0,128,0,0,0,0                       ; add           %al,0x0(%rax)
   DB  0,128,0,4,0,128                     ; add           %al,-0x7ffffc00(%rax)
   DB  0,0                                 ; add           %al,(%rax)
@@ -11060,7 +11072,7 @@ ALIGN 4
   DB  0,128,55,0,0,128                    ; add           %al,-0x7fffffc9(%rax)
   DB  63                                  ; (bad)
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            67bb <.literal4+0x37f>
+  DB  127,71                              ; jg            67d7 <.literal4+0x37f>
   DB  208                                 ; (bad)
   DB  179,89                              ; mov           $0x59,%bl
   DB  62,89                               ; ds            pop %rcx
@@ -11311,7 +11323,7 @@ _sk_seed_shader_sse41 LABEL PROC
   DB  102,15,110,199                      ; movd          %edi,%xmm0
   DB  102,15,112,192,0                    ; pshufd        $0x0,%xmm0,%xmm0
   DB  15,91,200                           ; cvtdq2ps      %xmm0,%xmm1
-  DB  15,40,21,113,70,0,0                 ; movaps        0x4671(%rip),%xmm2        # 4780 <_sk_callback_sse41+0xb0>
+  DB  15,40,21,161,70,0,0                 ; movaps        0x46a1(%rip),%xmm2        # 47b0 <_sk_callback_sse41+0xb6>
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  15,16,2                             ; movups        (%rdx),%xmm0
   DB  15,88,193                           ; addps         %xmm1,%xmm0
@@ -11320,7 +11332,7 @@ _sk_seed_shader_sse41 LABEL PROC
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,21,96,70,0,0                  ; movaps        0x4660(%rip),%xmm2        # 4790 <_sk_callback_sse41+0xc0>
+  DB  15,40,21,144,70,0,0                 ; movaps        0x4690(%rip),%xmm2        # 47c0 <_sk_callback_sse41+0xc6>
   DB  15,87,219                           ; xorps         %xmm3,%xmm3
   DB  15,87,228                           ; xorps         %xmm4,%xmm4
   DB  15,87,237                           ; xorps         %xmm5,%xmm5
@@ -11341,14 +11353,14 @@ _sk_dither_sse41 LABEL PROC
   DB  102,68,15,110,1                     ; movd          (%rcx),%xmm8
   DB  102,69,15,112,192,0                 ; pshufd        $0x0,%xmm8,%xmm8
   DB  102,69,15,239,193                   ; pxor          %xmm9,%xmm8
-  DB  102,68,15,111,21,37,70,0,0          ; movdqa        0x4625(%rip),%xmm10        # 47a0 <_sk_callback_sse41+0xd0>
+  DB  102,68,15,111,21,85,70,0,0          ; movdqa        0x4655(%rip),%xmm10        # 47d0 <_sk_callback_sse41+0xd6>
   DB  102,69,15,111,216                   ; movdqa        %xmm8,%xmm11
   DB  102,69,15,219,218                   ; pand          %xmm10,%xmm11
   DB  102,65,15,114,243,5                 ; pslld         $0x5,%xmm11
   DB  102,69,15,219,209                   ; pand          %xmm9,%xmm10
   DB  102,65,15,114,242,4                 ; pslld         $0x4,%xmm10
-  DB  102,68,15,111,37,17,70,0,0          ; movdqa        0x4611(%rip),%xmm12        # 47b0 <_sk_callback_sse41+0xe0>
-  DB  102,68,15,111,45,24,70,0,0          ; movdqa        0x4618(%rip),%xmm13        # 47c0 <_sk_callback_sse41+0xf0>
+  DB  102,68,15,111,37,65,70,0,0          ; movdqa        0x4641(%rip),%xmm12        # 47e0 <_sk_callback_sse41+0xe6>
+  DB  102,68,15,111,45,72,70,0,0          ; movdqa        0x4648(%rip),%xmm13        # 47f0 <_sk_callback_sse41+0xf6>
   DB  102,69,15,111,240                   ; movdqa        %xmm8,%xmm14
   DB  102,69,15,219,245                   ; pand          %xmm13,%xmm14
   DB  102,65,15,114,246,2                 ; pslld         $0x2,%xmm14
@@ -11364,15 +11376,26 @@ _sk_dither_sse41 LABEL PROC
   DB  102,69,15,235,245                   ; por           %xmm13,%xmm14
   DB  102,69,15,235,240                   ; por           %xmm8,%xmm14
   DB  69,15,91,198                        ; cvtdq2ps      %xmm14,%xmm8
-  DB  68,15,89,5,211,69,0,0               ; mulps         0x45d3(%rip),%xmm8        # 47d0 <_sk_callback_sse41+0x100>
-  DB  68,15,88,5,219,69,0,0               ; addps         0x45db(%rip),%xmm8        # 47e0 <_sk_callback_sse41+0x110>
-  DB  243,68,15,16,72,8                   ; movss         0x8(%rax),%xmm9
-  DB  69,15,198,201,0                     ; shufps        $0x0,%xmm9,%xmm9
-  DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
-  DB  65,15,88,193                        ; addps         %xmm9,%xmm0
-  DB  65,15,88,201                        ; addps         %xmm9,%xmm1
-  DB  65,15,88,209                        ; addps         %xmm9,%xmm2
+  DB  68,15,89,5,3,70,0,0                 ; mulps         0x4603(%rip),%xmm8        # 4800 <_sk_callback_sse41+0x106>
+  DB  68,15,88,5,11,70,0,0                ; addps         0x460b(%rip),%xmm8        # 4810 <_sk_callback_sse41+0x116>
+  DB  243,68,15,16,80,8                   ; movss         0x8(%rax),%xmm10
+  DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
+  DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
+  DB  65,15,88,194                        ; addps         %xmm10,%xmm0
+  DB  65,15,88,202                        ; addps         %xmm10,%xmm1
+  DB  68,15,88,210                        ; addps         %xmm2,%xmm10
+  DB  15,93,195                           ; minps         %xmm3,%xmm0
+  DB  15,87,210                           ; xorps         %xmm2,%xmm2
+  DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
+  DB  68,15,95,192                        ; maxps         %xmm0,%xmm8
+  DB  15,93,203                           ; minps         %xmm3,%xmm1
+  DB  102,69,15,239,201                   ; pxor          %xmm9,%xmm9
+  DB  68,15,95,201                        ; maxps         %xmm1,%xmm9
+  DB  68,15,93,211                        ; minps         %xmm3,%xmm10
+  DB  65,15,95,210                        ; maxps         %xmm10,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
+  DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
+  DB  65,15,40,201                        ; movaps        %xmm9,%xmm1
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_constant_color_sse41
@@ -11421,7 +11444,7 @@ _sk_clear_sse41 LABEL PROC
 PUBLIC _sk_srcatop_sse41
 _sk_srcatop_sse41 LABEL PROC
   DB  15,89,199                           ; mulps         %xmm7,%xmm0
-  DB  68,15,40,5,94,69,0,0                ; movaps        0x455e(%rip),%xmm8        # 47f0 <_sk_callback_sse41+0x120>
+  DB  68,15,40,5,100,69,0,0               ; movaps        0x4564(%rip),%xmm8        # 4820 <_sk_callback_sse41+0x126>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -11444,7 +11467,7 @@ PUBLIC _sk_dstatop_sse41
 _sk_dstatop_sse41 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
   DB  68,15,89,196                        ; mulps         %xmm4,%xmm8
-  DB  68,15,40,13,33,69,0,0               ; movaps        0x4521(%rip),%xmm9        # 4800 <_sk_callback_sse41+0x130>
+  DB  68,15,40,13,39,69,0,0               ; movaps        0x4527(%rip),%xmm9        # 4830 <_sk_callback_sse41+0x136>
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
@@ -11485,7 +11508,7 @@ _sk_dstin_sse41 LABEL PROC
 
 PUBLIC _sk_srcout_sse41
 _sk_srcout_sse41 LABEL PROC
-  DB  68,15,40,5,197,68,0,0               ; movaps        0x44c5(%rip),%xmm8        # 4810 <_sk_callback_sse41+0x140>
+  DB  68,15,40,5,203,68,0,0               ; movaps        0x44cb(%rip),%xmm8        # 4840 <_sk_callback_sse41+0x146>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
@@ -11496,7 +11519,7 @@ _sk_srcout_sse41 LABEL PROC
 
 PUBLIC _sk_dstout_sse41
 _sk_dstout_sse41 LABEL PROC
-  DB  68,15,40,5,181,68,0,0               ; movaps        0x44b5(%rip),%xmm8        # 4820 <_sk_callback_sse41+0x150>
+  DB  68,15,40,5,187,68,0,0               ; movaps        0x44bb(%rip),%xmm8        # 4850 <_sk_callback_sse41+0x156>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  15,89,196                           ; mulps         %xmm4,%xmm0
@@ -11511,7 +11534,7 @@ _sk_dstout_sse41 LABEL PROC
 
 PUBLIC _sk_srcover_sse41
 _sk_srcover_sse41 LABEL PROC
-  DB  68,15,40,5,152,68,0,0               ; movaps        0x4498(%rip),%xmm8        # 4830 <_sk_callback_sse41+0x160>
+  DB  68,15,40,5,158,68,0,0               ; movaps        0x449e(%rip),%xmm8        # 4860 <_sk_callback_sse41+0x166>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -11529,7 +11552,7 @@ _sk_srcover_sse41 LABEL PROC
 
 PUBLIC _sk_dstover_sse41
 _sk_dstover_sse41 LABEL PROC
-  DB  68,15,40,5,108,68,0,0               ; movaps        0x446c(%rip),%xmm8        # 4840 <_sk_callback_sse41+0x170>
+  DB  68,15,40,5,114,68,0,0               ; movaps        0x4472(%rip),%xmm8        # 4870 <_sk_callback_sse41+0x176>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -11553,7 +11576,7 @@ _sk_modulate_sse41 LABEL PROC
 
 PUBLIC _sk_multiply_sse41
 _sk_multiply_sse41 LABEL PROC
-  DB  68,15,40,5,64,68,0,0                ; movaps        0x4440(%rip),%xmm8        # 4850 <_sk_callback_sse41+0x180>
+  DB  68,15,40,5,70,68,0,0                ; movaps        0x4446(%rip),%xmm8        # 4880 <_sk_callback_sse41+0x186>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
@@ -11623,7 +11646,7 @@ _sk_screen_sse41 LABEL PROC
 PUBLIC _sk_xor__sse41
 _sk_xor__sse41 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
-  DB  15,40,29,113,67,0,0                 ; movaps        0x4371(%rip),%xmm3        # 4860 <_sk_callback_sse41+0x190>
+  DB  15,40,29,119,67,0,0                 ; movaps        0x4377(%rip),%xmm3        # 4890 <_sk_callback_sse41+0x196>
   DB  68,15,40,203                        ; movaps        %xmm3,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
@@ -11669,7 +11692,7 @@ _sk_darken_sse41 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,95,209                        ; maxps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,220,66,0,0                 ; movaps        0x42dc(%rip),%xmm2        # 4870 <_sk_callback_sse41+0x1a0>
+  DB  15,40,21,226,66,0,0                 ; movaps        0x42e2(%rip),%xmm2        # 48a0 <_sk_callback_sse41+0x1a6>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -11701,7 +11724,7 @@ _sk_lighten_sse41 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,129,66,0,0                 ; movaps        0x4281(%rip),%xmm2        # 4880 <_sk_callback_sse41+0x1b0>
+  DB  15,40,21,135,66,0,0                 ; movaps        0x4287(%rip),%xmm2        # 48b0 <_sk_callback_sse41+0x1b6>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -11736,7 +11759,7 @@ _sk_difference_sse41 LABEL PROC
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,27,66,0,0                  ; movaps        0x421b(%rip),%xmm2        # 4890 <_sk_callback_sse41+0x1c0>
+  DB  15,40,21,33,66,0,0                  ; movaps        0x4221(%rip),%xmm2        # 48c0 <_sk_callback_sse41+0x1c6>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -11761,7 +11784,7 @@ _sk_exclusion_sse41 LABEL PROC
   DB  15,89,214                           ; mulps         %xmm6,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,202                        ; subps         %xmm2,%xmm9
-  DB  15,40,13,220,65,0,0                 ; movaps        0x41dc(%rip),%xmm1        # 48a0 <_sk_callback_sse41+0x1d0>
+  DB  15,40,13,226,65,0,0                 ; movaps        0x41e2(%rip),%xmm1        # 48d0 <_sk_callback_sse41+0x1d6>
   DB  15,92,203                           ; subps         %xmm3,%xmm1
   DB  15,89,207                           ; mulps         %xmm7,%xmm1
   DB  15,88,217                           ; addps         %xmm1,%xmm3
@@ -11773,7 +11796,7 @@ _sk_exclusion_sse41 LABEL PROC
 PUBLIC _sk_colorburn_sse41
 _sk_colorburn_sse41 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,40,21,203,65,0,0              ; movaps        0x41cb(%rip),%xmm10        # 48b0 <_sk_callback_sse41+0x1e0>
+  DB  68,15,40,21,209,65,0,0              ; movaps        0x41d1(%rip),%xmm10        # 48e0 <_sk_callback_sse41+0x1e6>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  69,15,40,203                        ; movaps        %xmm11,%xmm9
@@ -11853,7 +11876,7 @@ _sk_colorburn_sse41 LABEL PROC
 PUBLIC _sk_colordodge_sse41
 _sk_colordodge_sse41 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,40,21,169,64,0,0              ; movaps        0x40a9(%rip),%xmm10        # 48c0 <_sk_callback_sse41+0x1f0>
+  DB  68,15,40,21,175,64,0,0              ; movaps        0x40af(%rip),%xmm10        # 48f0 <_sk_callback_sse41+0x1f6>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  69,15,40,227                        ; movaps        %xmm11,%xmm12
@@ -11934,7 +11957,7 @@ _sk_hardlight_sse41 LABEL PROC
   DB  15,40,244                           ; movaps        %xmm4,%xmm6
   DB  15,40,227                           ; movaps        %xmm3,%xmm4
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
-  DB  68,15,40,21,127,63,0,0              ; movaps        0x3f7f(%rip),%xmm10        # 48d0 <_sk_callback_sse41+0x200>
+  DB  68,15,40,21,133,63,0,0              ; movaps        0x3f85(%rip),%xmm10        # 4900 <_sk_callback_sse41+0x206>
   DB  65,15,40,234                        ; movaps        %xmm10,%xmm5
   DB  15,92,239                           ; subps         %xmm7,%xmm5
   DB  15,40,197                           ; movaps        %xmm5,%xmm0
@@ -12016,7 +12039,7 @@ PUBLIC _sk_overlay_sse41
 _sk_overlay_sse41 LABEL PROC
   DB  68,15,40,201                        ; movaps        %xmm1,%xmm9
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
-  DB  68,15,40,21,97,62,0,0               ; movaps        0x3e61(%rip),%xmm10        # 48e0 <_sk_callback_sse41+0x210>
+  DB  68,15,40,21,103,62,0,0              ; movaps        0x3e67(%rip),%xmm10        # 4910 <_sk_callback_sse41+0x216>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
@@ -12100,7 +12123,7 @@ _sk_softlight_sse41 LABEL PROC
   DB  15,40,198                           ; movaps        %xmm6,%xmm0
   DB  15,94,199                           ; divps         %xmm7,%xmm0
   DB  65,15,84,193                        ; andps         %xmm9,%xmm0
-  DB  15,40,13,52,61,0,0                  ; movaps        0x3d34(%rip),%xmm1        # 48f0 <_sk_callback_sse41+0x220>
+  DB  15,40,13,58,61,0,0                  ; movaps        0x3d3a(%rip),%xmm1        # 4920 <_sk_callback_sse41+0x226>
   DB  68,15,40,209                        ; movaps        %xmm1,%xmm10
   DB  68,15,92,208                        ; subps         %xmm0,%xmm10
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
@@ -12113,10 +12136,10 @@ _sk_softlight_sse41 LABEL PROC
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  15,89,210                           ; mulps         %xmm2,%xmm2
   DB  15,88,208                           ; addps         %xmm0,%xmm2
-  DB  68,15,40,45,18,61,0,0               ; movaps        0x3d12(%rip),%xmm13        # 4900 <_sk_callback_sse41+0x230>
+  DB  68,15,40,45,24,61,0,0               ; movaps        0x3d18(%rip),%xmm13        # 4930 <_sk_callback_sse41+0x236>
   DB  69,15,88,245                        ; addps         %xmm13,%xmm14
   DB  68,15,89,242                        ; mulps         %xmm2,%xmm14
-  DB  68,15,40,37,18,61,0,0               ; movaps        0x3d12(%rip),%xmm12        # 4910 <_sk_callback_sse41+0x240>
+  DB  68,15,40,37,24,61,0,0               ; movaps        0x3d18(%rip),%xmm12        # 4940 <_sk_callback_sse41+0x246>
   DB  69,15,89,252                        ; mulps         %xmm12,%xmm15
   DB  69,15,88,254                        ; addps         %xmm14,%xmm15
   DB  15,40,198                           ; movaps        %xmm6,%xmm0
@@ -12302,12 +12325,12 @@ _sk_hue_sse41 LABEL PROC
   DB  68,15,84,208                        ; andps         %xmm0,%xmm10
   DB  15,84,200                           ; andps         %xmm0,%xmm1
   DB  68,15,84,232                        ; andps         %xmm0,%xmm13
-  DB  15,40,5,120,58,0,0                  ; movaps        0x3a78(%rip),%xmm0        # 4920 <_sk_callback_sse41+0x250>
+  DB  15,40,5,126,58,0,0                  ; movaps        0x3a7e(%rip),%xmm0        # 4950 <_sk_callback_sse41+0x256>
   DB  68,15,89,224                        ; mulps         %xmm0,%xmm12
-  DB  15,40,21,125,58,0,0                 ; movaps        0x3a7d(%rip),%xmm2        # 4930 <_sk_callback_sse41+0x260>
+  DB  15,40,21,131,58,0,0                 ; movaps        0x3a83(%rip),%xmm2        # 4960 <_sk_callback_sse41+0x266>
   DB  15,89,250                           ; mulps         %xmm2,%xmm7
   DB  65,15,88,252                        ; addps         %xmm12,%xmm7
-  DB  68,15,40,53,126,58,0,0              ; movaps        0x3a7e(%rip),%xmm14        # 4940 <_sk_callback_sse41+0x270>
+  DB  68,15,40,53,132,58,0,0              ; movaps        0x3a84(%rip),%xmm14        # 4970 <_sk_callback_sse41+0x276>
   DB  68,15,40,252                        ; movaps        %xmm4,%xmm15
   DB  69,15,89,254                        ; mulps         %xmm14,%xmm15
   DB  68,15,88,255                        ; addps         %xmm7,%xmm15
@@ -12390,7 +12413,7 @@ _sk_hue_sse41 LABEL PROC
   DB  65,15,88,214                        ; addps         %xmm14,%xmm2
   DB  15,40,196                           ; movaps        %xmm4,%xmm0
   DB  102,15,56,20,202                    ; blendvps      %xmm0,%xmm2,%xmm1
-  DB  68,15,40,13,67,57,0,0               ; movaps        0x3943(%rip),%xmm9        # 4950 <_sk_callback_sse41+0x280>
+  DB  68,15,40,13,73,57,0,0               ; movaps        0x3949(%rip),%xmm9        # 4980 <_sk_callback_sse41+0x286>
   DB  65,15,40,225                        ; movaps        %xmm9,%xmm4
   DB  15,92,229                           ; subps         %xmm5,%xmm4
   DB  15,40,68,36,48                      ; movaps        0x30(%rsp),%xmm0
@@ -12484,14 +12507,14 @@ _sk_saturation_sse41 LABEL PROC
   DB  68,15,84,215                        ; andps         %xmm7,%xmm10
   DB  68,15,84,223                        ; andps         %xmm7,%xmm11
   DB  68,15,84,199                        ; andps         %xmm7,%xmm8
-  DB  15,40,21,246,55,0,0                 ; movaps        0x37f6(%rip),%xmm2        # 4960 <_sk_callback_sse41+0x290>
+  DB  15,40,21,252,55,0,0                 ; movaps        0x37fc(%rip),%xmm2        # 4990 <_sk_callback_sse41+0x296>
   DB  15,40,221                           ; movaps        %xmm5,%xmm3
   DB  15,89,218                           ; mulps         %xmm2,%xmm3
-  DB  15,40,13,249,55,0,0                 ; movaps        0x37f9(%rip),%xmm1        # 4970 <_sk_callback_sse41+0x2a0>
+  DB  15,40,13,255,55,0,0                 ; movaps        0x37ff(%rip),%xmm1        # 49a0 <_sk_callback_sse41+0x2a6>
   DB  15,40,254                           ; movaps        %xmm6,%xmm7
   DB  15,89,249                           ; mulps         %xmm1,%xmm7
   DB  15,88,251                           ; addps         %xmm3,%xmm7
-  DB  68,15,40,45,248,55,0,0              ; movaps        0x37f8(%rip),%xmm13        # 4980 <_sk_callback_sse41+0x2b0>
+  DB  68,15,40,45,254,55,0,0              ; movaps        0x37fe(%rip),%xmm13        # 49b0 <_sk_callback_sse41+0x2b6>
   DB  69,15,89,245                        ; mulps         %xmm13,%xmm14
   DB  68,15,88,247                        ; addps         %xmm7,%xmm14
   DB  65,15,40,218                        ; movaps        %xmm10,%xmm3
@@ -12572,7 +12595,7 @@ _sk_saturation_sse41 LABEL PROC
   DB  65,15,88,253                        ; addps         %xmm13,%xmm7
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  102,68,15,56,20,223                 ; blendvps      %xmm0,%xmm7,%xmm11
-  DB  68,15,40,13,190,54,0,0              ; movaps        0x36be(%rip),%xmm9        # 4990 <_sk_callback_sse41+0x2c0>
+  DB  68,15,40,13,196,54,0,0              ; movaps        0x36c4(%rip),%xmm9        # 49c0 <_sk_callback_sse41+0x2c6>
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  68,15,92,204                        ; subps         %xmm4,%xmm9
   DB  15,40,60,36                         ; movaps        (%rsp),%xmm7
@@ -12627,14 +12650,14 @@ _sk_color_sse41 LABEL PROC
   DB  15,40,231                           ; movaps        %xmm7,%xmm4
   DB  68,15,89,244                        ; mulps         %xmm4,%xmm14
   DB  15,89,204                           ; mulps         %xmm4,%xmm1
-  DB  68,15,40,13,3,54,0,0                ; movaps        0x3603(%rip),%xmm9        # 49a0 <_sk_callback_sse41+0x2d0>
+  DB  68,15,40,13,9,54,0,0                ; movaps        0x3609(%rip),%xmm9        # 49d0 <_sk_callback_sse41+0x2d6>
   DB  65,15,40,250                        ; movaps        %xmm10,%xmm7
   DB  65,15,89,249                        ; mulps         %xmm9,%xmm7
-  DB  68,15,40,21,3,54,0,0                ; movaps        0x3603(%rip),%xmm10        # 49b0 <_sk_callback_sse41+0x2e0>
+  DB  68,15,40,21,9,54,0,0                ; movaps        0x3609(%rip),%xmm10        # 49e0 <_sk_callback_sse41+0x2e6>
   DB  65,15,40,219                        ; movaps        %xmm11,%xmm3
   DB  65,15,89,218                        ; mulps         %xmm10,%xmm3
   DB  15,88,223                           ; addps         %xmm7,%xmm3
-  DB  68,15,40,29,0,54,0,0                ; movaps        0x3600(%rip),%xmm11        # 49c0 <_sk_callback_sse41+0x2f0>
+  DB  68,15,40,29,6,54,0,0                ; movaps        0x3606(%rip),%xmm11        # 49f0 <_sk_callback_sse41+0x2f6>
   DB  69,15,40,236                        ; movaps        %xmm12,%xmm13
   DB  69,15,89,235                        ; mulps         %xmm11,%xmm13
   DB  68,15,88,235                        ; addps         %xmm3,%xmm13
@@ -12719,7 +12742,7 @@ _sk_color_sse41 LABEL PROC
   DB  65,15,88,251                        ; addps         %xmm11,%xmm7
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  102,15,56,20,207                    ; blendvps      %xmm0,%xmm7,%xmm1
-  DB  68,15,40,13,188,52,0,0              ; movaps        0x34bc(%rip),%xmm9        # 49d0 <_sk_callback_sse41+0x300>
+  DB  68,15,40,13,194,52,0,0              ; movaps        0x34c2(%rip),%xmm9        # 4a00 <_sk_callback_sse41+0x306>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  68,15,89,192                        ; mulps         %xmm0,%xmm8
@@ -12771,13 +12794,13 @@ _sk_luminosity_sse41 LABEL PROC
   DB  69,15,89,216                        ; mulps         %xmm8,%xmm11
   DB  68,15,40,203                        ; movaps        %xmm3,%xmm9
   DB  68,15,89,205                        ; mulps         %xmm5,%xmm9
-  DB  68,15,40,5,14,52,0,0                ; movaps        0x340e(%rip),%xmm8        # 49e0 <_sk_callback_sse41+0x310>
+  DB  68,15,40,5,20,52,0,0                ; movaps        0x3414(%rip),%xmm8        # 4a10 <_sk_callback_sse41+0x316>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
-  DB  68,15,40,21,18,52,0,0               ; movaps        0x3412(%rip),%xmm10        # 49f0 <_sk_callback_sse41+0x320>
+  DB  68,15,40,21,24,52,0,0               ; movaps        0x3418(%rip),%xmm10        # 4a20 <_sk_callback_sse41+0x326>
   DB  15,40,233                           ; movaps        %xmm1,%xmm5
   DB  65,15,89,234                        ; mulps         %xmm10,%xmm5
   DB  15,88,232                           ; addps         %xmm0,%xmm5
-  DB  68,15,40,37,16,52,0,0               ; movaps        0x3410(%rip),%xmm12        # 4a00 <_sk_callback_sse41+0x330>
+  DB  68,15,40,37,22,52,0,0               ; movaps        0x3416(%rip),%xmm12        # 4a30 <_sk_callback_sse41+0x336>
   DB  68,15,40,242                        ; movaps        %xmm2,%xmm14
   DB  69,15,89,244                        ; mulps         %xmm12,%xmm14
   DB  68,15,88,245                        ; addps         %xmm5,%xmm14
@@ -12862,7 +12885,7 @@ _sk_luminosity_sse41 LABEL PROC
   DB  65,15,88,244                        ; addps         %xmm12,%xmm6
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  102,68,15,56,20,206                 ; blendvps      %xmm0,%xmm6,%xmm9
-  DB  15,40,5,198,50,0,0                  ; movaps        0x32c6(%rip),%xmm0        # 4a10 <_sk_callback_sse41+0x340>
+  DB  15,40,5,204,50,0,0                  ; movaps        0x32cc(%rip),%xmm0        # 4a40 <_sk_callback_sse41+0x346>
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  15,92,215                           ; subps         %xmm7,%xmm2
   DB  15,89,226                           ; mulps         %xmm2,%xmm4
@@ -12908,7 +12931,7 @@ _sk_clamp_0_sse41 LABEL PROC
 
 PUBLIC _sk_clamp_1_sse41
 _sk_clamp_1_sse41 LABEL PROC
-  DB  68,15,40,5,70,50,0,0                ; movaps        0x3246(%rip),%xmm8        # 4a20 <_sk_callback_sse41+0x350>
+  DB  68,15,40,5,76,50,0,0                ; movaps        0x324c(%rip),%xmm8        # 4a50 <_sk_callback_sse41+0x356>
   DB  65,15,93,192                        ; minps         %xmm8,%xmm0
   DB  65,15,93,200                        ; minps         %xmm8,%xmm1
   DB  65,15,93,208                        ; minps         %xmm8,%xmm2
@@ -12918,7 +12941,7 @@ _sk_clamp_1_sse41 LABEL PROC
 
 PUBLIC _sk_clamp_a_sse41
 _sk_clamp_a_sse41 LABEL PROC
-  DB  15,93,29,59,50,0,0                  ; minps         0x323b(%rip),%xmm3        # 4a30 <_sk_callback_sse41+0x360>
+  DB  15,93,29,65,50,0,0                  ; minps         0x3241(%rip),%xmm3        # 4a60 <_sk_callback_sse41+0x366>
   DB  15,93,195                           ; minps         %xmm3,%xmm0
   DB  15,93,203                           ; minps         %xmm3,%xmm1
   DB  15,93,211                           ; minps         %xmm3,%xmm2
@@ -12991,7 +13014,7 @@ _sk_premul_sse41 LABEL PROC
 PUBLIC _sk_unpremul_sse41
 _sk_unpremul_sse41 LABEL PROC
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
-  DB  68,15,40,13,166,49,0,0              ; movaps        0x31a6(%rip),%xmm9        # 4a40 <_sk_callback_sse41+0x370>
+  DB  68,15,40,13,172,49,0,0              ; movaps        0x31ac(%rip),%xmm9        # 4a70 <_sk_callback_sse41+0x376>
   DB  68,15,94,203                        ; divps         %xmm3,%xmm9
   DB  68,15,194,195,4                     ; cmpneqps      %xmm3,%xmm8
   DB  69,15,84,193                        ; andps         %xmm9,%xmm8
@@ -13003,20 +13026,20 @@ _sk_unpremul_sse41 LABEL PROC
 
 PUBLIC _sk_from_srgb_sse41
 _sk_from_srgb_sse41 LABEL PROC
-  DB  68,15,40,29,145,49,0,0              ; movaps        0x3191(%rip),%xmm11        # 4a50 <_sk_callback_sse41+0x380>
+  DB  68,15,40,29,151,49,0,0              ; movaps        0x3197(%rip),%xmm11        # 4a80 <_sk_callback_sse41+0x386>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
   DB  68,15,40,208                        ; movaps        %xmm0,%xmm10
   DB  69,15,89,210                        ; mulps         %xmm10,%xmm10
-  DB  68,15,40,37,137,49,0,0              ; movaps        0x3189(%rip),%xmm12        # 4a60 <_sk_callback_sse41+0x390>
+  DB  68,15,40,37,143,49,0,0              ; movaps        0x318f(%rip),%xmm12        # 4a90 <_sk_callback_sse41+0x396>
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,89,196                        ; mulps         %xmm12,%xmm8
-  DB  68,15,40,45,137,49,0,0              ; movaps        0x3189(%rip),%xmm13        # 4a70 <_sk_callback_sse41+0x3a0>
+  DB  68,15,40,45,143,49,0,0              ; movaps        0x318f(%rip),%xmm13        # 4aa0 <_sk_callback_sse41+0x3a6>
   DB  69,15,88,197                        ; addps         %xmm13,%xmm8
   DB  69,15,89,194                        ; mulps         %xmm10,%xmm8
-  DB  68,15,40,53,137,49,0,0              ; movaps        0x3189(%rip),%xmm14        # 4a80 <_sk_callback_sse41+0x3b0>
+  DB  68,15,40,53,143,49,0,0              ; movaps        0x318f(%rip),%xmm14        # 4ab0 <_sk_callback_sse41+0x3b6>
   DB  69,15,88,198                        ; addps         %xmm14,%xmm8
-  DB  68,15,40,61,141,49,0,0              ; movaps        0x318d(%rip),%xmm15        # 4a90 <_sk_callback_sse41+0x3c0>
+  DB  68,15,40,61,147,49,0,0              ; movaps        0x3193(%rip),%xmm15        # 4ac0 <_sk_callback_sse41+0x3c6>
   DB  65,15,194,199,1                     ; cmpltps       %xmm15,%xmm0
   DB  102,69,15,56,20,193                 ; blendvps      %xmm0,%xmm9,%xmm8
   DB  68,15,40,209                        ; movaps        %xmm1,%xmm10
@@ -13060,20 +13083,20 @@ _sk_to_srgb_sse41 LABEL PROC
   DB  68,15,82,192                        ; rsqrtps       %xmm0,%xmm8
   DB  69,15,83,200                        ; rcpps         %xmm8,%xmm9
   DB  69,15,82,208                        ; rsqrtps       %xmm8,%xmm10
-  DB  68,15,40,29,250,48,0,0              ; movaps        0x30fa(%rip),%xmm11        # 4aa0 <_sk_callback_sse41+0x3d0>
+  DB  68,15,40,29,0,49,0,0                ; movaps        0x3100(%rip),%xmm11        # 4ad0 <_sk_callback_sse41+0x3d6>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
-  DB  68,15,40,37,251,48,0,0              ; movaps        0x30fb(%rip),%xmm12        # 4ab0 <_sk_callback_sse41+0x3e0>
+  DB  68,15,40,37,1,49,0,0                ; movaps        0x3101(%rip),%xmm12        # 4ae0 <_sk_callback_sse41+0x3e6>
   DB  69,15,89,204                        ; mulps         %xmm12,%xmm9
-  DB  68,15,40,45,255,48,0,0              ; movaps        0x30ff(%rip),%xmm13        # 4ac0 <_sk_callback_sse41+0x3f0>
+  DB  68,15,40,45,5,49,0,0                ; movaps        0x3105(%rip),%xmm13        # 4af0 <_sk_callback_sse41+0x3f6>
   DB  69,15,88,205                        ; addps         %xmm13,%xmm9
-  DB  68,15,40,53,3,49,0,0                ; movaps        0x3103(%rip),%xmm14        # 4ad0 <_sk_callback_sse41+0x400>
+  DB  68,15,40,53,9,49,0,0                ; movaps        0x3109(%rip),%xmm14        # 4b00 <_sk_callback_sse41+0x406>
   DB  69,15,89,214                        ; mulps         %xmm14,%xmm10
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
-  DB  68,15,40,5,3,49,0,0                 ; movaps        0x3103(%rip),%xmm8        # 4ae0 <_sk_callback_sse41+0x410>
+  DB  68,15,40,5,9,49,0,0                 ; movaps        0x3109(%rip),%xmm8        # 4b10 <_sk_callback_sse41+0x416>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,93,202                        ; minps         %xmm10,%xmm9
-  DB  68,15,40,61,3,49,0,0                ; movaps        0x3103(%rip),%xmm15        # 4af0 <_sk_callback_sse41+0x420>
+  DB  68,15,40,61,9,49,0,0                ; movaps        0x3109(%rip),%xmm15        # 4b20 <_sk_callback_sse41+0x426>
   DB  65,15,194,199,1                     ; cmpltps       %xmm15,%xmm0
   DB  102,68,15,56,20,201                 ; blendvps      %xmm0,%xmm1,%xmm9
   DB  15,82,194                           ; rsqrtps       %xmm2,%xmm0
@@ -13126,7 +13149,7 @@ _sk_rgb_to_hsl_sse41 LABEL PROC
   DB  68,15,93,226                        ; minps         %xmm2,%xmm12
   DB  65,15,40,203                        ; movaps        %xmm11,%xmm1
   DB  65,15,92,204                        ; subps         %xmm12,%xmm1
-  DB  68,15,40,53,81,48,0,0               ; movaps        0x3051(%rip),%xmm14        # 4b00 <_sk_callback_sse41+0x430>
+  DB  68,15,40,53,87,48,0,0               ; movaps        0x3057(%rip),%xmm14        # 4b30 <_sk_callback_sse41+0x436>
   DB  68,15,94,241                        ; divps         %xmm1,%xmm14
   DB  69,15,40,211                        ; movaps        %xmm11,%xmm10
   DB  69,15,194,208,0                     ; cmpeqps       %xmm8,%xmm10
@@ -13135,27 +13158,27 @@ _sk_rgb_to_hsl_sse41 LABEL PROC
   DB  65,15,89,198                        ; mulps         %xmm14,%xmm0
   DB  69,15,40,249                        ; movaps        %xmm9,%xmm15
   DB  68,15,194,250,1                     ; cmpltps       %xmm2,%xmm15
-  DB  68,15,84,61,56,48,0,0               ; andps         0x3038(%rip),%xmm15        # 4b10 <_sk_callback_sse41+0x440>
+  DB  68,15,84,61,62,48,0,0               ; andps         0x303e(%rip),%xmm15        # 4b40 <_sk_callback_sse41+0x446>
   DB  68,15,88,248                        ; addps         %xmm0,%xmm15
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  65,15,194,193,0                     ; cmpeqps       %xmm9,%xmm0
   DB  65,15,92,208                        ; subps         %xmm8,%xmm2
   DB  65,15,89,214                        ; mulps         %xmm14,%xmm2
-  DB  68,15,40,45,43,48,0,0               ; movaps        0x302b(%rip),%xmm13        # 4b20 <_sk_callback_sse41+0x450>
+  DB  68,15,40,45,49,48,0,0               ; movaps        0x3031(%rip),%xmm13        # 4b50 <_sk_callback_sse41+0x456>
   DB  65,15,88,213                        ; addps         %xmm13,%xmm2
   DB  69,15,92,193                        ; subps         %xmm9,%xmm8
   DB  69,15,89,198                        ; mulps         %xmm14,%xmm8
-  DB  68,15,88,5,39,48,0,0                ; addps         0x3027(%rip),%xmm8        # 4b30 <_sk_callback_sse41+0x460>
+  DB  68,15,88,5,45,48,0,0                ; addps         0x302d(%rip),%xmm8        # 4b60 <_sk_callback_sse41+0x466>
   DB  102,68,15,56,20,194                 ; blendvps      %xmm0,%xmm2,%xmm8
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  102,69,15,56,20,199                 ; blendvps      %xmm0,%xmm15,%xmm8
-  DB  68,15,89,5,31,48,0,0                ; mulps         0x301f(%rip),%xmm8        # 4b40 <_sk_callback_sse41+0x470>
+  DB  68,15,89,5,37,48,0,0                ; mulps         0x3025(%rip),%xmm8        # 4b70 <_sk_callback_sse41+0x476>
   DB  69,15,40,203                        ; movaps        %xmm11,%xmm9
   DB  69,15,194,204,4                     ; cmpneqps      %xmm12,%xmm9
   DB  69,15,84,193                        ; andps         %xmm9,%xmm8
   DB  69,15,92,235                        ; subps         %xmm11,%xmm13
   DB  69,15,88,220                        ; addps         %xmm12,%xmm11
-  DB  15,40,5,19,48,0,0                   ; movaps        0x3013(%rip),%xmm0        # 4b50 <_sk_callback_sse41+0x480>
+  DB  15,40,5,25,48,0,0                   ; movaps        0x3019(%rip),%xmm0        # 4b80 <_sk_callback_sse41+0x486>
   DB  65,15,40,211                        ; movaps        %xmm11,%xmm2
   DB  15,89,208                           ; mulps         %xmm0,%xmm2
   DB  15,194,194,1                        ; cmpltps       %xmm2,%xmm0
@@ -13176,7 +13199,7 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  15,41,100,36,32                     ; movaps        %xmm4,0x20(%rsp)
   DB  15,41,92,36,16                      ; movaps        %xmm3,0x10(%rsp)
   DB  68,15,40,208                        ; movaps        %xmm0,%xmm10
-  DB  68,15,40,13,213,47,0,0              ; movaps        0x2fd5(%rip),%xmm9        # 4b60 <_sk_callback_sse41+0x490>
+  DB  68,15,40,13,219,47,0,0              ; movaps        0x2fdb(%rip),%xmm9        # 4b90 <_sk_callback_sse41+0x496>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  15,194,194,2                        ; cmpleps       %xmm2,%xmm0
   DB  15,40,217                           ; movaps        %xmm1,%xmm3
@@ -13189,19 +13212,19 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  15,41,20,36                         ; movaps        %xmm2,(%rsp)
   DB  69,15,88,192                        ; addps         %xmm8,%xmm8
   DB  68,15,92,197                        ; subps         %xmm5,%xmm8
-  DB  68,15,40,53,177,47,0,0              ; movaps        0x2fb1(%rip),%xmm14        # 4b70 <_sk_callback_sse41+0x4a0>
+  DB  68,15,40,53,183,47,0,0              ; movaps        0x2fb7(%rip),%xmm14        # 4ba0 <_sk_callback_sse41+0x4a6>
   DB  69,15,88,242                        ; addps         %xmm10,%xmm14
   DB  102,65,15,58,8,198,1                ; roundps       $0x1,%xmm14,%xmm0
   DB  68,15,92,240                        ; subps         %xmm0,%xmm14
-  DB  68,15,40,29,170,47,0,0              ; movaps        0x2faa(%rip),%xmm11        # 4b80 <_sk_callback_sse41+0x4b0>
+  DB  68,15,40,29,176,47,0,0              ; movaps        0x2fb0(%rip),%xmm11        # 4bb0 <_sk_callback_sse41+0x4b6>
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  65,15,194,198,2                     ; cmpleps       %xmm14,%xmm0
   DB  15,40,245                           ; movaps        %xmm5,%xmm6
   DB  65,15,92,240                        ; subps         %xmm8,%xmm6
-  DB  15,40,61,163,47,0,0                 ; movaps        0x2fa3(%rip),%xmm7        # 4b90 <_sk_callback_sse41+0x4c0>
+  DB  15,40,61,169,47,0,0                 ; movaps        0x2fa9(%rip),%xmm7        # 4bc0 <_sk_callback_sse41+0x4c6>
   DB  69,15,40,238                        ; movaps        %xmm14,%xmm13
   DB  68,15,89,239                        ; mulps         %xmm7,%xmm13
-  DB  15,40,29,164,47,0,0                 ; movaps        0x2fa4(%rip),%xmm3        # 4ba0 <_sk_callback_sse41+0x4d0>
+  DB  15,40,29,170,47,0,0                 ; movaps        0x2faa(%rip),%xmm3        # 4bd0 <_sk_callback_sse41+0x4d6>
   DB  68,15,40,227                        ; movaps        %xmm3,%xmm12
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  68,15,89,230                        ; mulps         %xmm6,%xmm12
@@ -13211,7 +13234,7 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  65,15,194,198,2                     ; cmpleps       %xmm14,%xmm0
   DB  68,15,40,253                        ; movaps        %xmm5,%xmm15
   DB  102,69,15,56,20,252                 ; blendvps      %xmm0,%xmm12,%xmm15
-  DB  68,15,40,37,131,47,0,0              ; movaps        0x2f83(%rip),%xmm12        # 4bb0 <_sk_callback_sse41+0x4e0>
+  DB  68,15,40,37,137,47,0,0              ; movaps        0x2f89(%rip),%xmm12        # 4be0 <_sk_callback_sse41+0x4e6>
   DB  65,15,40,196                        ; movaps        %xmm12,%xmm0
   DB  65,15,194,198,2                     ; cmpleps       %xmm14,%xmm0
   DB  68,15,89,238                        ; mulps         %xmm6,%xmm13
@@ -13245,7 +13268,7 @@ _sk_hsl_to_rgb_sse41 LABEL PROC
   DB  65,15,40,198                        ; movaps        %xmm14,%xmm0
   DB  15,40,20,36                         ; movaps        (%rsp),%xmm2
   DB  102,15,56,20,202                    ; blendvps      %xmm0,%xmm2,%xmm1
-  DB  68,15,88,21,252,46,0,0              ; addps         0x2efc(%rip),%xmm10        # 4bc0 <_sk_callback_sse41+0x4f0>
+  DB  68,15,88,21,2,47,0,0                ; addps         0x2f02(%rip),%xmm10        # 4bf0 <_sk_callback_sse41+0x4f6>
   DB  102,65,15,58,8,194,1                ; roundps       $0x1,%xmm10,%xmm0
   DB  68,15,92,208                        ; subps         %xmm0,%xmm10
   DB  69,15,194,218,2                     ; cmpleps       %xmm10,%xmm11
@@ -13294,7 +13317,7 @@ _sk_scale_u8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,68,15,56,49,4,56                ; pmovzxbd      (%rax,%rdi,1),%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,85,46,0,0                ; mulps         0x2e55(%rip),%xmm8        # 4bd0 <_sk_callback_sse41+0x500>
+  DB  68,15,89,5,91,46,0,0                ; mulps         0x2e5b(%rip),%xmm8        # 4c00 <_sk_callback_sse41+0x506>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
@@ -13328,7 +13351,7 @@ _sk_lerp_u8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,68,15,56,49,4,56                ; pmovzxbd      (%rax,%rdi,1),%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,1,46,0,0                 ; mulps         0x2e01(%rip),%xmm8        # 4be0 <_sk_callback_sse41+0x510>
+  DB  68,15,89,5,7,46,0,0                 ; mulps         0x2e07(%rip),%xmm8        # 4c10 <_sk_callback_sse41+0x516>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -13349,17 +13372,17 @@ _sk_lerp_565_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,68,15,56,51,20,120              ; pmovzxwd      (%rax,%rdi,2),%xmm10
-  DB  102,68,15,111,5,208,45,0,0          ; movdqa        0x2dd0(%rip),%xmm8        # 4bf0 <_sk_callback_sse41+0x520>
+  DB  102,68,15,111,5,214,45,0,0          ; movdqa        0x2dd6(%rip),%xmm8        # 4c20 <_sk_callback_sse41+0x526>
   DB  102,69,15,219,194                   ; pand          %xmm10,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,207,45,0,0               ; mulps         0x2dcf(%rip),%xmm8        # 4c00 <_sk_callback_sse41+0x530>
-  DB  102,68,15,111,13,214,45,0,0         ; movdqa        0x2dd6(%rip),%xmm9        # 4c10 <_sk_callback_sse41+0x540>
+  DB  68,15,89,5,213,45,0,0               ; mulps         0x2dd5(%rip),%xmm8        # 4c30 <_sk_callback_sse41+0x536>
+  DB  102,68,15,111,13,220,45,0,0         ; movdqa        0x2ddc(%rip),%xmm9        # 4c40 <_sk_callback_sse41+0x546>
   DB  102,69,15,219,202                   ; pand          %xmm10,%xmm9
   DB  69,15,91,201                        ; cvtdq2ps      %xmm9,%xmm9
-  DB  68,15,89,13,213,45,0,0              ; mulps         0x2dd5(%rip),%xmm9        # 4c20 <_sk_callback_sse41+0x550>
-  DB  102,68,15,219,21,220,45,0,0         ; pand          0x2ddc(%rip),%xmm10        # 4c30 <_sk_callback_sse41+0x560>
+  DB  68,15,89,13,219,45,0,0              ; mulps         0x2ddb(%rip),%xmm9        # 4c50 <_sk_callback_sse41+0x556>
+  DB  102,68,15,219,21,226,45,0,0         ; pand          0x2de2(%rip),%xmm10        # 4c60 <_sk_callback_sse41+0x566>
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
-  DB  68,15,89,21,224,45,0,0              ; mulps         0x2de0(%rip),%xmm10        # 4c40 <_sk_callback_sse41+0x570>
+  DB  68,15,89,21,230,45,0,0              ; mulps         0x2de6(%rip),%xmm10        # 4c70 <_sk_callback_sse41+0x576>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -13388,7 +13411,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,139,72,8                         ; mov           0x8(%rax),%r9
   DB  243,69,15,111,4,184                 ; movdqu        (%r8,%rdi,4),%xmm8
-  DB  102,15,111,5,145,45,0,0             ; movdqa        0x2d91(%rip),%xmm0        # 4c50 <_sk_callback_sse41+0x580>
+  DB  102,15,111,5,151,45,0,0             ; movdqa        0x2d97(%rip),%xmm0        # 4c80 <_sk_callback_sse41+0x586>
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,73,15,58,22,192,1               ; pextrq        $0x1,%xmm0,%r8
   DB  102,72,15,126,193                   ; movq          %xmm0,%rcx
@@ -13403,7 +13426,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  102,15,58,33,193,48                 ; insertps      $0x30,%xmm1,%xmm0
   DB  76,139,64,16                        ; mov           0x10(%rax),%r8
   DB  102,65,15,111,200                   ; movdqa        %xmm8,%xmm1
-  DB  102,15,56,0,13,76,45,0,0            ; pshufb        0x2d4c(%rip),%xmm1        # 4c60 <_sk_callback_sse41+0x590>
+  DB  102,15,56,0,13,82,45,0,0            ; pshufb        0x2d52(%rip),%xmm1        # 4c90 <_sk_callback_sse41+0x596>
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
   DB  102,72,15,126,201                   ; movq          %xmm1,%rcx
   DB  68,15,182,209                       ; movzbl        %cl,%r10d
@@ -13418,7 +13441,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  102,15,58,33,202,48                 ; insertps      $0x30,%xmm2,%xmm1
   DB  76,139,64,24                        ; mov           0x18(%rax),%r8
   DB  102,65,15,111,208                   ; movdqa        %xmm8,%xmm2
-  DB  102,15,56,0,21,8,45,0,0             ; pshufb        0x2d08(%rip),%xmm2        # 4c70 <_sk_callback_sse41+0x5a0>
+  DB  102,15,56,0,21,14,45,0,0            ; pshufb        0x2d0e(%rip),%xmm2        # 4ca0 <_sk_callback_sse41+0x5a6>
   DB  102,72,15,58,22,209,1               ; pextrq        $0x1,%xmm2,%rcx
   DB  102,72,15,126,208                   ; movq          %xmm2,%rax
   DB  68,15,182,200                       ; movzbl        %al,%r9d
@@ -13433,7 +13456,7 @@ _sk_load_tables_sse41 LABEL PROC
   DB  102,15,58,33,211,48                 ; insertps      $0x30,%xmm3,%xmm2
   DB  102,65,15,114,208,24                ; psrld         $0x18,%xmm8
   DB  65,15,91,216                        ; cvtdq2ps      %xmm8,%xmm3
-  DB  15,89,29,197,44,0,0                 ; mulps         0x2cc5(%rip),%xmm3        # 4c80 <_sk_callback_sse41+0x5b0>
+  DB  15,89,29,203,44,0,0                 ; mulps         0x2ccb(%rip),%xmm3        # 4cb0 <_sk_callback_sse41+0x5b6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -13450,7 +13473,7 @@ _sk_load_tables_u16_be_sse41 LABEL PROC
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,97,200                       ; punpcklwd     %xmm0,%xmm1
   DB  102,68,15,105,200                   ; punpckhwd     %xmm0,%xmm9
-  DB  102,68,15,111,5,152,44,0,0          ; movdqa        0x2c98(%rip),%xmm8        # 4c90 <_sk_callback_sse41+0x5c0>
+  DB  102,68,15,111,5,158,44,0,0          ; movdqa        0x2c9e(%rip),%xmm8        # 4cc0 <_sk_callback_sse41+0x5c6>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,15,56,51,192                    ; pmovzxwd      %xmm0,%xmm0
@@ -13467,7 +13490,7 @@ _sk_load_tables_u16_be_sse41 LABEL PROC
   DB  243,67,15,16,20,8                   ; movss         (%r8,%r9,1),%xmm2
   DB  102,15,58,33,194,48                 ; insertps      $0x30,%xmm2,%xmm0
   DB  76,139,64,16                        ; mov           0x10(%rax),%r8
-  DB  102,15,56,0,13,75,44,0,0            ; pshufb        0x2c4b(%rip),%xmm1        # 4ca0 <_sk_callback_sse41+0x5d0>
+  DB  102,15,56,0,13,81,44,0,0            ; pshufb        0x2c51(%rip),%xmm1        # 4cd0 <_sk_callback_sse41+0x5d6>
   DB  102,15,56,51,201                    ; pmovzxwd      %xmm1,%xmm1
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
   DB  102,72,15,126,201                   ; movq          %xmm1,%rcx
@@ -13503,7 +13526,7 @@ _sk_load_tables_u16_be_sse41 LABEL PROC
   DB  102,65,15,235,216                   ; por           %xmm8,%xmm3
   DB  102,15,56,51,219                    ; pmovzxwd      %xmm3,%xmm3
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,153,43,0,0                 ; mulps         0x2b99(%rip),%xmm3        # 4cb0 <_sk_callback_sse41+0x5e0>
+  DB  15,89,29,159,43,0,0                 ; mulps         0x2b9f(%rip),%xmm3        # 4ce0 <_sk_callback_sse41+0x5e6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -13523,7 +13546,7 @@ _sk_load_tables_rgb_u16_be_sse41 LABEL PROC
   DB  102,68,15,97,200                    ; punpcklwd     %xmm0,%xmm9
   DB  102,15,111,202                      ; movdqa        %xmm2,%xmm1
   DB  102,65,15,97,201                    ; punpcklwd     %xmm9,%xmm1
-  DB  102,68,15,111,5,91,43,0,0           ; movdqa        0x2b5b(%rip),%xmm8        # 4cc0 <_sk_callback_sse41+0x5f0>
+  DB  102,68,15,111,5,97,43,0,0           ; movdqa        0x2b61(%rip),%xmm8        # 4cf0 <_sk_callback_sse41+0x5f6>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,15,56,51,192                    ; pmovzxwd      %xmm0,%xmm0
@@ -13540,7 +13563,7 @@ _sk_load_tables_rgb_u16_be_sse41 LABEL PROC
   DB  243,67,15,16,28,8                   ; movss         (%r8,%r9,1),%xmm3
   DB  102,15,58,33,195,48                 ; insertps      $0x30,%xmm3,%xmm0
   DB  76,139,64,16                        ; mov           0x10(%rax),%r8
-  DB  102,15,56,0,13,14,43,0,0            ; pshufb        0x2b0e(%rip),%xmm1        # 4cd0 <_sk_callback_sse41+0x600>
+  DB  102,15,56,0,13,20,43,0,0            ; pshufb        0x2b14(%rip),%xmm1        # 4d00 <_sk_callback_sse41+0x606>
   DB  102,15,56,51,201                    ; pmovzxwd      %xmm1,%xmm1
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
   DB  102,72,15,126,201                   ; movq          %xmm1,%rcx
@@ -13571,7 +13594,7 @@ _sk_load_tables_rgb_u16_be_sse41 LABEL PROC
   DB  243,65,15,16,28,8                   ; movss         (%r8,%rcx,1),%xmm3
   DB  102,15,58,33,211,48                 ; insertps      $0x30,%xmm3,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,121,42,0,0                 ; movaps        0x2a79(%rip),%xmm3        # 4ce0 <_sk_callback_sse41+0x610>
+  DB  15,40,29,127,42,0,0                 ; movaps        0x2a7f(%rip),%xmm3        # 4d10 <_sk_callback_sse41+0x616>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_byte_tables_sse41
@@ -13579,7 +13602,7 @@ _sk_byte_tables_sse41 LABEL PROC
   DB  65,86                               ; push          %r14
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,122,42,0,0               ; movaps        0x2a7a(%rip),%xmm8        # 4cf0 <_sk_callback_sse41+0x620>
+  DB  68,15,40,5,128,42,0,0               ; movaps        0x2a80(%rip),%xmm8        # 4d20 <_sk_callback_sse41+0x626>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,91,192                       ; cvtps2dq      %xmm0,%xmm0
   DB  102,72,15,58,22,193,1               ; pextrq        $0x1,%xmm0,%rcx
@@ -13598,7 +13621,7 @@ _sk_byte_tables_sse41 LABEL PROC
   DB  102,15,58,32,193,3                  ; pinsrb        $0x3,%ecx,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,13,43,42,0,0               ; movaps        0x2a2b(%rip),%xmm9        # 4d00 <_sk_callback_sse41+0x630>
+  DB  68,15,40,13,49,42,0,0               ; movaps        0x2a31(%rip),%xmm9        # 4d30 <_sk_callback_sse41+0x636>
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -13687,7 +13710,7 @@ _sk_byte_tables_rgb_sse41 LABEL PROC
   DB  102,15,58,32,193,3                  ; pinsrb        $0x3,%ecx,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,13,179,40,0,0              ; movaps        0x28b3(%rip),%xmm9        # 4d10 <_sk_callback_sse41+0x640>
+  DB  68,15,40,13,185,40,0,0              ; movaps        0x28b9(%rip),%xmm9        # 4d40 <_sk_callback_sse41+0x646>
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -13854,31 +13877,31 @@ _sk_parametric_r_sse41 LABEL PROC
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,194                        ; cvtdq2ps      %xmm10,%xmm8
-  DB  68,15,89,5,10,38,0,0                ; mulps         0x260a(%rip),%xmm8        # 4d20 <_sk_callback_sse41+0x650>
-  DB  68,15,84,21,18,38,0,0               ; andps         0x2612(%rip),%xmm10        # 4d30 <_sk_callback_sse41+0x660>
-  DB  68,15,86,21,26,38,0,0               ; orps          0x261a(%rip),%xmm10        # 4d40 <_sk_callback_sse41+0x670>
-  DB  68,15,88,5,34,38,0,0                ; addps         0x2622(%rip),%xmm8        # 4d50 <_sk_callback_sse41+0x680>
-  DB  68,15,40,37,42,38,0,0               ; movaps        0x262a(%rip),%xmm12        # 4d60 <_sk_callback_sse41+0x690>
+  DB  68,15,89,5,16,38,0,0                ; mulps         0x2610(%rip),%xmm8        # 4d50 <_sk_callback_sse41+0x656>
+  DB  68,15,84,21,24,38,0,0               ; andps         0x2618(%rip),%xmm10        # 4d60 <_sk_callback_sse41+0x666>
+  DB  68,15,86,21,32,38,0,0               ; orps          0x2620(%rip),%xmm10        # 4d70 <_sk_callback_sse41+0x676>
+  DB  68,15,88,5,40,38,0,0                ; addps         0x2628(%rip),%xmm8        # 4d80 <_sk_callback_sse41+0x686>
+  DB  68,15,40,37,48,38,0,0               ; movaps        0x2630(%rip),%xmm12        # 4d90 <_sk_callback_sse41+0x696>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,196                        ; subps         %xmm12,%xmm8
-  DB  68,15,88,21,42,38,0,0               ; addps         0x262a(%rip),%xmm10        # 4d70 <_sk_callback_sse41+0x6a0>
-  DB  68,15,40,37,50,38,0,0               ; movaps        0x2632(%rip),%xmm12        # 4d80 <_sk_callback_sse41+0x6b0>
+  DB  68,15,88,21,48,38,0,0               ; addps         0x2630(%rip),%xmm10        # 4da0 <_sk_callback_sse41+0x6a6>
+  DB  68,15,40,37,56,38,0,0               ; movaps        0x2638(%rip),%xmm12        # 4db0 <_sk_callback_sse41+0x6b6>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,196                        ; subps         %xmm12,%xmm8
   DB  69,15,89,195                        ; mulps         %xmm11,%xmm8
   DB  102,69,15,58,8,208,1                ; roundps       $0x1,%xmm8,%xmm10
   DB  69,15,40,216                        ; movaps        %xmm8,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,5,31,38,0,0                ; addps         0x261f(%rip),%xmm8        # 4d90 <_sk_callback_sse41+0x6c0>
-  DB  68,15,40,21,39,38,0,0               ; movaps        0x2627(%rip),%xmm10        # 4da0 <_sk_callback_sse41+0x6d0>
+  DB  68,15,88,5,37,38,0,0                ; addps         0x2625(%rip),%xmm8        # 4dc0 <_sk_callback_sse41+0x6c6>
+  DB  68,15,40,21,45,38,0,0               ; movaps        0x262d(%rip),%xmm10        # 4dd0 <_sk_callback_sse41+0x6d6>
   DB  69,15,89,211                        ; mulps         %xmm11,%xmm10
   DB  69,15,92,194                        ; subps         %xmm10,%xmm8
-  DB  68,15,40,21,39,38,0,0               ; movaps        0x2627(%rip),%xmm10        # 4db0 <_sk_callback_sse41+0x6e0>
+  DB  68,15,40,21,45,38,0,0               ; movaps        0x262d(%rip),%xmm10        # 4de0 <_sk_callback_sse41+0x6e6>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  68,15,40,29,43,38,0,0               ; movaps        0x262b(%rip),%xmm11        # 4dc0 <_sk_callback_sse41+0x6f0>
+  DB  68,15,40,29,49,38,0,0               ; movaps        0x2631(%rip),%xmm11        # 4df0 <_sk_callback_sse41+0x6f6>
   DB  69,15,94,218                        ; divps         %xmm10,%xmm11
   DB  69,15,88,216                        ; addps         %xmm8,%xmm11
-  DB  68,15,89,29,43,38,0,0               ; mulps         0x262b(%rip),%xmm11        # 4dd0 <_sk_callback_sse41+0x700>
+  DB  68,15,89,29,49,38,0,0               ; mulps         0x2631(%rip),%xmm11        # 4e00 <_sk_callback_sse41+0x706>
   DB  102,69,15,91,211                    ; cvtps2dq      %xmm11,%xmm10
   DB  243,68,15,16,64,20                  ; movss         0x14(%rax),%xmm8
   DB  69,15,198,192,0                     ; shufps        $0x0,%xmm8,%xmm8
@@ -13886,7 +13909,7 @@ _sk_parametric_r_sse41 LABEL PROC
   DB  102,69,15,56,20,193                 ; blendvps      %xmm0,%xmm9,%xmm8
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  68,15,95,192                        ; maxps         %xmm0,%xmm8
-  DB  68,15,93,5,18,38,0,0                ; minps         0x2612(%rip),%xmm8        # 4de0 <_sk_callback_sse41+0x710>
+  DB  68,15,93,5,24,38,0,0                ; minps         0x2618(%rip),%xmm8        # 4e10 <_sk_callback_sse41+0x716>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -13914,31 +13937,31 @@ _sk_parametric_g_sse41 LABEL PROC
   DB  68,15,88,217                        ; addps         %xmm1,%xmm11
   DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
   DB  69,15,91,227                        ; cvtdq2ps      %xmm11,%xmm12
-  DB  68,15,89,37,179,37,0,0              ; mulps         0x25b3(%rip),%xmm12        # 4df0 <_sk_callback_sse41+0x720>
-  DB  68,15,84,29,187,37,0,0              ; andps         0x25bb(%rip),%xmm11        # 4e00 <_sk_callback_sse41+0x730>
-  DB  68,15,86,29,195,37,0,0              ; orps          0x25c3(%rip),%xmm11        # 4e10 <_sk_callback_sse41+0x740>
-  DB  68,15,88,37,203,37,0,0              ; addps         0x25cb(%rip),%xmm12        # 4e20 <_sk_callback_sse41+0x750>
-  DB  15,40,13,212,37,0,0                 ; movaps        0x25d4(%rip),%xmm1        # 4e30 <_sk_callback_sse41+0x760>
+  DB  68,15,89,37,185,37,0,0              ; mulps         0x25b9(%rip),%xmm12        # 4e20 <_sk_callback_sse41+0x726>
+  DB  68,15,84,29,193,37,0,0              ; andps         0x25c1(%rip),%xmm11        # 4e30 <_sk_callback_sse41+0x736>
+  DB  68,15,86,29,201,37,0,0              ; orps          0x25c9(%rip),%xmm11        # 4e40 <_sk_callback_sse41+0x746>
+  DB  68,15,88,37,209,37,0,0              ; addps         0x25d1(%rip),%xmm12        # 4e50 <_sk_callback_sse41+0x756>
+  DB  15,40,13,218,37,0,0                 ; movaps        0x25da(%rip),%xmm1        # 4e60 <_sk_callback_sse41+0x766>
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
   DB  68,15,92,225                        ; subps         %xmm1,%xmm12
-  DB  68,15,88,29,212,37,0,0              ; addps         0x25d4(%rip),%xmm11        # 4e40 <_sk_callback_sse41+0x770>
-  DB  15,40,13,221,37,0,0                 ; movaps        0x25dd(%rip),%xmm1        # 4e50 <_sk_callback_sse41+0x780>
+  DB  68,15,88,29,218,37,0,0              ; addps         0x25da(%rip),%xmm11        # 4e70 <_sk_callback_sse41+0x776>
+  DB  15,40,13,227,37,0,0                 ; movaps        0x25e3(%rip),%xmm1        # 4e80 <_sk_callback_sse41+0x786>
   DB  65,15,94,203                        ; divps         %xmm11,%xmm1
   DB  68,15,92,225                        ; subps         %xmm1,%xmm12
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  102,69,15,58,8,212,1                ; roundps       $0x1,%xmm12,%xmm10
   DB  69,15,40,220                        ; movaps        %xmm12,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,37,202,37,0,0              ; addps         0x25ca(%rip),%xmm12        # 4e60 <_sk_callback_sse41+0x790>
-  DB  15,40,13,211,37,0,0                 ; movaps        0x25d3(%rip),%xmm1        # 4e70 <_sk_callback_sse41+0x7a0>
+  DB  68,15,88,37,208,37,0,0              ; addps         0x25d0(%rip),%xmm12        # 4e90 <_sk_callback_sse41+0x796>
+  DB  15,40,13,217,37,0,0                 ; movaps        0x25d9(%rip),%xmm1        # 4ea0 <_sk_callback_sse41+0x7a6>
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
   DB  68,15,92,225                        ; subps         %xmm1,%xmm12
-  DB  68,15,40,21,211,37,0,0              ; movaps        0x25d3(%rip),%xmm10        # 4e80 <_sk_callback_sse41+0x7b0>
+  DB  68,15,40,21,217,37,0,0              ; movaps        0x25d9(%rip),%xmm10        # 4eb0 <_sk_callback_sse41+0x7b6>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  15,40,13,216,37,0,0                 ; movaps        0x25d8(%rip),%xmm1        # 4e90 <_sk_callback_sse41+0x7c0>
+  DB  15,40,13,222,37,0,0                 ; movaps        0x25de(%rip),%xmm1        # 4ec0 <_sk_callback_sse41+0x7c6>
   DB  65,15,94,202                        ; divps         %xmm10,%xmm1
   DB  65,15,88,204                        ; addps         %xmm12,%xmm1
-  DB  15,89,13,217,37,0,0                 ; mulps         0x25d9(%rip),%xmm1        # 4ea0 <_sk_callback_sse41+0x7d0>
+  DB  15,89,13,223,37,0,0                 ; mulps         0x25df(%rip),%xmm1        # 4ed0 <_sk_callback_sse41+0x7d6>
   DB  102,68,15,91,209                    ; cvtps2dq      %xmm1,%xmm10
   DB  243,15,16,72,20                     ; movss         0x14(%rax),%xmm1
   DB  15,198,201,0                        ; shufps        $0x0,%xmm1,%xmm1
@@ -13946,7 +13969,7 @@ _sk_parametric_g_sse41 LABEL PROC
   DB  102,65,15,56,20,201                 ; blendvps      %xmm0,%xmm9,%xmm1
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,95,200                           ; maxps         %xmm0,%xmm1
-  DB  15,93,13,196,37,0,0                 ; minps         0x25c4(%rip),%xmm1        # 4eb0 <_sk_callback_sse41+0x7e0>
+  DB  15,93,13,202,37,0,0                 ; minps         0x25ca(%rip),%xmm1        # 4ee0 <_sk_callback_sse41+0x7e6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -13974,31 +13997,31 @@ _sk_parametric_b_sse41 LABEL PROC
   DB  68,15,88,218                        ; addps         %xmm2,%xmm11
   DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
   DB  69,15,91,227                        ; cvtdq2ps      %xmm11,%xmm12
-  DB  68,15,89,37,101,37,0,0              ; mulps         0x2565(%rip),%xmm12        # 4ec0 <_sk_callback_sse41+0x7f0>
-  DB  68,15,84,29,109,37,0,0              ; andps         0x256d(%rip),%xmm11        # 4ed0 <_sk_callback_sse41+0x800>
-  DB  68,15,86,29,117,37,0,0              ; orps          0x2575(%rip),%xmm11        # 4ee0 <_sk_callback_sse41+0x810>
-  DB  68,15,88,37,125,37,0,0              ; addps         0x257d(%rip),%xmm12        # 4ef0 <_sk_callback_sse41+0x820>
-  DB  15,40,21,134,37,0,0                 ; movaps        0x2586(%rip),%xmm2        # 4f00 <_sk_callback_sse41+0x830>
+  DB  68,15,89,37,107,37,0,0              ; mulps         0x256b(%rip),%xmm12        # 4ef0 <_sk_callback_sse41+0x7f6>
+  DB  68,15,84,29,115,37,0,0              ; andps         0x2573(%rip),%xmm11        # 4f00 <_sk_callback_sse41+0x806>
+  DB  68,15,86,29,123,37,0,0              ; orps          0x257b(%rip),%xmm11        # 4f10 <_sk_callback_sse41+0x816>
+  DB  68,15,88,37,131,37,0,0              ; addps         0x2583(%rip),%xmm12        # 4f20 <_sk_callback_sse41+0x826>
+  DB  15,40,21,140,37,0,0                 ; movaps        0x258c(%rip),%xmm2        # 4f30 <_sk_callback_sse41+0x836>
   DB  65,15,89,211                        ; mulps         %xmm11,%xmm2
   DB  68,15,92,226                        ; subps         %xmm2,%xmm12
-  DB  68,15,88,29,134,37,0,0              ; addps         0x2586(%rip),%xmm11        # 4f10 <_sk_callback_sse41+0x840>
-  DB  15,40,21,143,37,0,0                 ; movaps        0x258f(%rip),%xmm2        # 4f20 <_sk_callback_sse41+0x850>
+  DB  68,15,88,29,140,37,0,0              ; addps         0x258c(%rip),%xmm11        # 4f40 <_sk_callback_sse41+0x846>
+  DB  15,40,21,149,37,0,0                 ; movaps        0x2595(%rip),%xmm2        # 4f50 <_sk_callback_sse41+0x856>
   DB  65,15,94,211                        ; divps         %xmm11,%xmm2
   DB  68,15,92,226                        ; subps         %xmm2,%xmm12
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  102,69,15,58,8,212,1                ; roundps       $0x1,%xmm12,%xmm10
   DB  69,15,40,220                        ; movaps        %xmm12,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,37,124,37,0,0              ; addps         0x257c(%rip),%xmm12        # 4f30 <_sk_callback_sse41+0x860>
-  DB  15,40,21,133,37,0,0                 ; movaps        0x2585(%rip),%xmm2        # 4f40 <_sk_callback_sse41+0x870>
+  DB  68,15,88,37,130,37,0,0              ; addps         0x2582(%rip),%xmm12        # 4f60 <_sk_callback_sse41+0x866>
+  DB  15,40,21,139,37,0,0                 ; movaps        0x258b(%rip),%xmm2        # 4f70 <_sk_callback_sse41+0x876>
   DB  65,15,89,211                        ; mulps         %xmm11,%xmm2
   DB  68,15,92,226                        ; subps         %xmm2,%xmm12
-  DB  68,15,40,21,133,37,0,0              ; movaps        0x2585(%rip),%xmm10        # 4f50 <_sk_callback_sse41+0x880>
+  DB  68,15,40,21,139,37,0,0              ; movaps        0x258b(%rip),%xmm10        # 4f80 <_sk_callback_sse41+0x886>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  15,40,21,138,37,0,0                 ; movaps        0x258a(%rip),%xmm2        # 4f60 <_sk_callback_sse41+0x890>
+  DB  15,40,21,144,37,0,0                 ; movaps        0x2590(%rip),%xmm2        # 4f90 <_sk_callback_sse41+0x896>
   DB  65,15,94,210                        ; divps         %xmm10,%xmm2
   DB  65,15,88,212                        ; addps         %xmm12,%xmm2
-  DB  15,89,21,139,37,0,0                 ; mulps         0x258b(%rip),%xmm2        # 4f70 <_sk_callback_sse41+0x8a0>
+  DB  15,89,21,145,37,0,0                 ; mulps         0x2591(%rip),%xmm2        # 4fa0 <_sk_callback_sse41+0x8a6>
   DB  102,68,15,91,210                    ; cvtps2dq      %xmm2,%xmm10
   DB  243,15,16,80,20                     ; movss         0x14(%rax),%xmm2
   DB  15,198,210,0                        ; shufps        $0x0,%xmm2,%xmm2
@@ -14006,7 +14029,7 @@ _sk_parametric_b_sse41 LABEL PROC
   DB  102,65,15,56,20,209                 ; blendvps      %xmm0,%xmm9,%xmm2
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,95,208                           ; maxps         %xmm0,%xmm2
-  DB  15,93,21,118,37,0,0                 ; minps         0x2576(%rip),%xmm2        # 4f80 <_sk_callback_sse41+0x8b0>
+  DB  15,93,21,124,37,0,0                 ; minps         0x257c(%rip),%xmm2        # 4fb0 <_sk_callback_sse41+0x8b6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -14034,31 +14057,31 @@ _sk_parametric_a_sse41 LABEL PROC
   DB  68,15,88,219                        ; addps         %xmm3,%xmm11
   DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
   DB  69,15,91,227                        ; cvtdq2ps      %xmm11,%xmm12
-  DB  68,15,89,37,23,37,0,0               ; mulps         0x2517(%rip),%xmm12        # 4f90 <_sk_callback_sse41+0x8c0>
-  DB  68,15,84,29,31,37,0,0               ; andps         0x251f(%rip),%xmm11        # 4fa0 <_sk_callback_sse41+0x8d0>
-  DB  68,15,86,29,39,37,0,0               ; orps          0x2527(%rip),%xmm11        # 4fb0 <_sk_callback_sse41+0x8e0>
-  DB  68,15,88,37,47,37,0,0               ; addps         0x252f(%rip),%xmm12        # 4fc0 <_sk_callback_sse41+0x8f0>
-  DB  15,40,29,56,37,0,0                  ; movaps        0x2538(%rip),%xmm3        # 4fd0 <_sk_callback_sse41+0x900>
+  DB  68,15,89,37,29,37,0,0               ; mulps         0x251d(%rip),%xmm12        # 4fc0 <_sk_callback_sse41+0x8c6>
+  DB  68,15,84,29,37,37,0,0               ; andps         0x2525(%rip),%xmm11        # 4fd0 <_sk_callback_sse41+0x8d6>
+  DB  68,15,86,29,45,37,0,0               ; orps          0x252d(%rip),%xmm11        # 4fe0 <_sk_callback_sse41+0x8e6>
+  DB  68,15,88,37,53,37,0,0               ; addps         0x2535(%rip),%xmm12        # 4ff0 <_sk_callback_sse41+0x8f6>
+  DB  15,40,29,62,37,0,0                  ; movaps        0x253e(%rip),%xmm3        # 5000 <_sk_callback_sse41+0x906>
   DB  65,15,89,219                        ; mulps         %xmm11,%xmm3
   DB  68,15,92,227                        ; subps         %xmm3,%xmm12
-  DB  68,15,88,29,56,37,0,0               ; addps         0x2538(%rip),%xmm11        # 4fe0 <_sk_callback_sse41+0x910>
-  DB  15,40,29,65,37,0,0                  ; movaps        0x2541(%rip),%xmm3        # 4ff0 <_sk_callback_sse41+0x920>
+  DB  68,15,88,29,62,37,0,0               ; addps         0x253e(%rip),%xmm11        # 5010 <_sk_callback_sse41+0x916>
+  DB  15,40,29,71,37,0,0                  ; movaps        0x2547(%rip),%xmm3        # 5020 <_sk_callback_sse41+0x926>
   DB  65,15,94,219                        ; divps         %xmm11,%xmm3
   DB  68,15,92,227                        ; subps         %xmm3,%xmm12
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  102,69,15,58,8,212,1                ; roundps       $0x1,%xmm12,%xmm10
   DB  69,15,40,220                        ; movaps        %xmm12,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  68,15,88,37,46,37,0,0               ; addps         0x252e(%rip),%xmm12        # 5000 <_sk_callback_sse41+0x930>
-  DB  15,40,29,55,37,0,0                  ; movaps        0x2537(%rip),%xmm3        # 5010 <_sk_callback_sse41+0x940>
+  DB  68,15,88,37,52,37,0,0               ; addps         0x2534(%rip),%xmm12        # 5030 <_sk_callback_sse41+0x936>
+  DB  15,40,29,61,37,0,0                  ; movaps        0x253d(%rip),%xmm3        # 5040 <_sk_callback_sse41+0x946>
   DB  65,15,89,219                        ; mulps         %xmm11,%xmm3
   DB  68,15,92,227                        ; subps         %xmm3,%xmm12
-  DB  68,15,40,21,55,37,0,0               ; movaps        0x2537(%rip),%xmm10        # 5020 <_sk_callback_sse41+0x950>
+  DB  68,15,40,21,61,37,0,0               ; movaps        0x253d(%rip),%xmm10        # 5050 <_sk_callback_sse41+0x956>
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
-  DB  15,40,29,60,37,0,0                  ; movaps        0x253c(%rip),%xmm3        # 5030 <_sk_callback_sse41+0x960>
+  DB  15,40,29,66,37,0,0                  ; movaps        0x2542(%rip),%xmm3        # 5060 <_sk_callback_sse41+0x966>
   DB  65,15,94,218                        ; divps         %xmm10,%xmm3
   DB  65,15,88,220                        ; addps         %xmm12,%xmm3
-  DB  15,89,29,61,37,0,0                  ; mulps         0x253d(%rip),%xmm3        # 5040 <_sk_callback_sse41+0x970>
+  DB  15,89,29,67,37,0,0                  ; mulps         0x2543(%rip),%xmm3        # 5070 <_sk_callback_sse41+0x976>
   DB  102,68,15,91,211                    ; cvtps2dq      %xmm3,%xmm10
   DB  243,15,16,88,20                     ; movss         0x14(%rax),%xmm3
   DB  15,198,219,0                        ; shufps        $0x0,%xmm3,%xmm3
@@ -14066,7 +14089,7 @@ _sk_parametric_a_sse41 LABEL PROC
   DB  102,65,15,56,20,217                 ; blendvps      %xmm0,%xmm9,%xmm3
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,95,216                           ; maxps         %xmm0,%xmm3
-  DB  15,93,29,40,37,0,0                  ; minps         0x2528(%rip),%xmm3        # 5050 <_sk_callback_sse41+0x980>
+  DB  15,93,29,46,37,0,0                  ; minps         0x252e(%rip),%xmm3        # 5080 <_sk_callback_sse41+0x986>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -14074,29 +14097,29 @@ _sk_parametric_a_sse41 LABEL PROC
 PUBLIC _sk_lab_to_xyz_sse41
 _sk_lab_to_xyz_sse41 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,89,5,36,37,0,0                ; mulps         0x2524(%rip),%xmm8        # 5060 <_sk_callback_sse41+0x990>
-  DB  68,15,40,13,44,37,0,0               ; movaps        0x252c(%rip),%xmm9        # 5070 <_sk_callback_sse41+0x9a0>
+  DB  68,15,89,5,42,37,0,0                ; mulps         0x252a(%rip),%xmm8        # 5090 <_sk_callback_sse41+0x996>
+  DB  68,15,40,13,50,37,0,0               ; movaps        0x2532(%rip),%xmm9        # 50a0 <_sk_callback_sse41+0x9a6>
   DB  65,15,89,201                        ; mulps         %xmm9,%xmm1
-  DB  15,40,5,49,37,0,0                   ; movaps        0x2531(%rip),%xmm0        # 5080 <_sk_callback_sse41+0x9b0>
+  DB  15,40,5,55,37,0,0                   ; movaps        0x2537(%rip),%xmm0        # 50b0 <_sk_callback_sse41+0x9b6>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
   DB  15,88,208                           ; addps         %xmm0,%xmm2
-  DB  68,15,88,5,47,37,0,0                ; addps         0x252f(%rip),%xmm8        # 5090 <_sk_callback_sse41+0x9c0>
-  DB  68,15,89,5,55,37,0,0                ; mulps         0x2537(%rip),%xmm8        # 50a0 <_sk_callback_sse41+0x9d0>
-  DB  15,89,13,64,37,0,0                  ; mulps         0x2540(%rip),%xmm1        # 50b0 <_sk_callback_sse41+0x9e0>
+  DB  68,15,88,5,53,37,0,0                ; addps         0x2535(%rip),%xmm8        # 50c0 <_sk_callback_sse41+0x9c6>
+  DB  68,15,89,5,61,37,0,0                ; mulps         0x253d(%rip),%xmm8        # 50d0 <_sk_callback_sse41+0x9d6>
+  DB  15,89,13,70,37,0,0                  ; mulps         0x2546(%rip),%xmm1        # 50e0 <_sk_callback_sse41+0x9e6>
   DB  65,15,88,200                        ; addps         %xmm8,%xmm1
-  DB  15,89,21,69,37,0,0                  ; mulps         0x2545(%rip),%xmm2        # 50c0 <_sk_callback_sse41+0x9f0>
+  DB  15,89,21,75,37,0,0                  ; mulps         0x254b(%rip),%xmm2        # 50f0 <_sk_callback_sse41+0x9f6>
   DB  69,15,40,208                        ; movaps        %xmm8,%xmm10
   DB  68,15,92,210                        ; subps         %xmm2,%xmm10
   DB  68,15,40,217                        ; movaps        %xmm1,%xmm11
   DB  69,15,89,219                        ; mulps         %xmm11,%xmm11
   DB  68,15,89,217                        ; mulps         %xmm1,%xmm11
-  DB  68,15,40,13,57,37,0,0               ; movaps        0x2539(%rip),%xmm9        # 50d0 <_sk_callback_sse41+0xa00>
+  DB  68,15,40,13,63,37,0,0               ; movaps        0x253f(%rip),%xmm9        # 5100 <_sk_callback_sse41+0xa06>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  65,15,194,195,1                     ; cmpltps       %xmm11,%xmm0
-  DB  15,40,21,57,37,0,0                  ; movaps        0x2539(%rip),%xmm2        # 50e0 <_sk_callback_sse41+0xa10>
+  DB  15,40,21,63,37,0,0                  ; movaps        0x253f(%rip),%xmm2        # 5110 <_sk_callback_sse41+0xa16>
   DB  15,88,202                           ; addps         %xmm2,%xmm1
-  DB  68,15,40,37,62,37,0,0               ; movaps        0x253e(%rip),%xmm12        # 50f0 <_sk_callback_sse41+0xa20>
+  DB  68,15,40,37,68,37,0,0               ; movaps        0x2544(%rip),%xmm12        # 5120 <_sk_callback_sse41+0xa26>
   DB  65,15,89,204                        ; mulps         %xmm12,%xmm1
   DB  102,65,15,56,20,203                 ; blendvps      %xmm0,%xmm11,%xmm1
   DB  69,15,40,216                        ; movaps        %xmm8,%xmm11
@@ -14115,8 +14138,8 @@ _sk_lab_to_xyz_sse41 LABEL PROC
   DB  65,15,89,212                        ; mulps         %xmm12,%xmm2
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  102,65,15,56,20,211                 ; blendvps      %xmm0,%xmm11,%xmm2
-  DB  15,89,13,247,36,0,0                 ; mulps         0x24f7(%rip),%xmm1        # 5100 <_sk_callback_sse41+0xa30>
-  DB  15,89,21,0,37,0,0                   ; mulps         0x2500(%rip),%xmm2        # 5110 <_sk_callback_sse41+0xa40>
+  DB  15,89,13,253,36,0,0                 ; mulps         0x24fd(%rip),%xmm1        # 5130 <_sk_callback_sse41+0xa36>
+  DB  15,89,21,6,37,0,0                   ; mulps         0x2506(%rip),%xmm2        # 5140 <_sk_callback_sse41+0xa46>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,40,193                           ; movaps        %xmm1,%xmm0
   DB  65,15,40,200                        ; movaps        %xmm8,%xmm1
@@ -14128,7 +14151,7 @@ _sk_load_a8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,49,4,56                   ; pmovzxbd      (%rax,%rdi,1),%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,240,36,0,0                 ; mulps         0x24f0(%rip),%xmm3        # 5120 <_sk_callback_sse41+0xa50>
+  DB  15,89,29,246,36,0,0                 ; mulps         0x24f6(%rip),%xmm3        # 5150 <_sk_callback_sse41+0xa56>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  15,87,201                           ; xorps         %xmm1,%xmm1
@@ -14159,7 +14182,7 @@ _sk_gather_a8_sse41 LABEL PROC
   DB  102,15,58,32,192,3                  ; pinsrb        $0x3,%eax,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,132,36,0,0                 ; mulps         0x2484(%rip),%xmm3        # 5130 <_sk_callback_sse41+0xa60>
+  DB  15,89,29,138,36,0,0                 ; mulps         0x248a(%rip),%xmm3        # 5160 <_sk_callback_sse41+0xa66>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
@@ -14170,7 +14193,7 @@ PUBLIC _sk_store_a8_sse41
 _sk_store_a8_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,120,36,0,0               ; movaps        0x2478(%rip),%xmm8        # 5140 <_sk_callback_sse41+0xa70>
+  DB  68,15,40,5,126,36,0,0               ; movaps        0x247e(%rip),%xmm8        # 5170 <_sk_callback_sse41+0xa76>
   DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
   DB  102,69,15,56,43,192                 ; packusdw      %xmm8,%xmm8
@@ -14185,9 +14208,9 @@ _sk_load_g8_sse41 LABEL PROC
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,49,4,56                   ; pmovzxbd      (%rax,%rdi,1),%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,85,36,0,0                   ; mulps         0x2455(%rip),%xmm0        # 5150 <_sk_callback_sse41+0xa80>
+  DB  15,89,5,91,36,0,0                   ; mulps         0x245b(%rip),%xmm0        # 5180 <_sk_callback_sse41+0xa86>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,92,36,0,0                  ; movaps        0x245c(%rip),%xmm3        # 5160 <_sk_callback_sse41+0xa90>
+  DB  15,40,29,98,36,0,0                  ; movaps        0x2462(%rip),%xmm3        # 5190 <_sk_callback_sse41+0xa96>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -14216,9 +14239,9 @@ _sk_gather_g8_sse41 LABEL PROC
   DB  102,15,58,32,192,3                  ; pinsrb        $0x3,%eax,%xmm0
   DB  102,15,56,49,192                    ; pmovzxbd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,245,35,0,0                  ; mulps         0x23f5(%rip),%xmm0        # 5170 <_sk_callback_sse41+0xaa0>
+  DB  15,89,5,251,35,0,0                  ; mulps         0x23fb(%rip),%xmm0        # 51a0 <_sk_callback_sse41+0xaa6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,252,35,0,0                 ; movaps        0x23fc(%rip),%xmm3        # 5180 <_sk_callback_sse41+0xab0>
+  DB  15,40,29,2,36,0,0                   ; movaps        0x2402(%rip),%xmm3        # 51b0 <_sk_callback_sse41+0xab6>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -14228,9 +14251,9 @@ _sk_gather_i8_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            2d9b <_sk_gather_i8_sse41+0xf>
+  DB  116,5                               ; je            2dc5 <_sk_gather_i8_sse41+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           2d9d <_sk_gather_i8_sse41+0x11>
+  DB  235,2                               ; jmp           2dc7 <_sk_gather_i8_sse41+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  243,15,91,201                       ; cvttps2dq     %xmm1,%xmm1
@@ -14261,17 +14284,17 @@ _sk_gather_i8_sse41 LABEL PROC
   DB  102,15,58,34,28,8,1                 ; pinsrd        $0x1,(%rax,%rcx,1),%xmm3
   DB  102,66,15,58,34,28,144,2            ; pinsrd        $0x2,(%rax,%r10,4),%xmm3
   DB  102,66,15,58,34,28,8,3              ; pinsrd        $0x3,(%rax,%r9,1),%xmm3
-  DB  102,15,111,5,83,35,0,0              ; movdqa        0x2353(%rip),%xmm0        # 5190 <_sk_callback_sse41+0xac0>
+  DB  102,15,111,5,89,35,0,0              ; movdqa        0x2359(%rip),%xmm0        # 51c0 <_sk_callback_sse41+0xac6>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,84,35,0,0                ; movaps        0x2354(%rip),%xmm8        # 51a0 <_sk_callback_sse41+0xad0>
+  DB  68,15,40,5,90,35,0,0                ; movaps        0x235a(%rip),%xmm8        # 51d0 <_sk_callback_sse41+0xad6>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
-  DB  102,15,56,0,13,83,35,0,0            ; pshufb        0x2353(%rip),%xmm1        # 51b0 <_sk_callback_sse41+0xae0>
+  DB  102,15,56,0,13,89,35,0,0            ; pshufb        0x2359(%rip),%xmm1        # 51e0 <_sk_callback_sse41+0xae6>
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,111,211                      ; movdqa        %xmm3,%xmm2
-  DB  102,15,56,0,21,79,35,0,0            ; pshufb        0x234f(%rip),%xmm2        # 51c0 <_sk_callback_sse41+0xaf0>
+  DB  102,15,56,0,21,85,35,0,0            ; pshufb        0x2355(%rip),%xmm2        # 51f0 <_sk_callback_sse41+0xaf6>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  102,15,114,211,24                   ; psrld         $0x18,%xmm3
@@ -14285,19 +14308,19 @@ _sk_load_565_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,51,20,120                 ; pmovzxwd      (%rax,%rdi,2),%xmm2
-  DB  102,15,111,5,53,35,0,0              ; movdqa        0x2335(%rip),%xmm0        # 51d0 <_sk_callback_sse41+0xb00>
+  DB  102,15,111,5,59,35,0,0              ; movdqa        0x233b(%rip),%xmm0        # 5200 <_sk_callback_sse41+0xb06>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,55,35,0,0                   ; mulps         0x2337(%rip),%xmm0        # 51e0 <_sk_callback_sse41+0xb10>
-  DB  102,15,111,13,63,35,0,0             ; movdqa        0x233f(%rip),%xmm1        # 51f0 <_sk_callback_sse41+0xb20>
+  DB  15,89,5,61,35,0,0                   ; mulps         0x233d(%rip),%xmm0        # 5210 <_sk_callback_sse41+0xb16>
+  DB  102,15,111,13,69,35,0,0             ; movdqa        0x2345(%rip),%xmm1        # 5220 <_sk_callback_sse41+0xb26>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,65,35,0,0                  ; mulps         0x2341(%rip),%xmm1        # 5200 <_sk_callback_sse41+0xb30>
-  DB  102,15,219,21,73,35,0,0             ; pand          0x2349(%rip),%xmm2        # 5210 <_sk_callback_sse41+0xb40>
+  DB  15,89,13,71,35,0,0                  ; mulps         0x2347(%rip),%xmm1        # 5230 <_sk_callback_sse41+0xb36>
+  DB  102,15,219,21,79,35,0,0             ; pand          0x234f(%rip),%xmm2        # 5240 <_sk_callback_sse41+0xb46>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,79,35,0,0                  ; mulps         0x234f(%rip),%xmm2        # 5220 <_sk_callback_sse41+0xb50>
+  DB  15,89,21,85,35,0,0                  ; mulps         0x2355(%rip),%xmm2        # 5250 <_sk_callback_sse41+0xb56>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,86,35,0,0                  ; movaps        0x2356(%rip),%xmm3        # 5230 <_sk_callback_sse41+0xb60>
+  DB  15,40,29,92,35,0,0                  ; movaps        0x235c(%rip),%xmm3        # 5260 <_sk_callback_sse41+0xb66>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_gather_565_sse41
@@ -14323,31 +14346,31 @@ _sk_gather_565_sse41 LABEL PROC
   DB  65,15,183,4,65                      ; movzwl        (%r9,%rax,2),%eax
   DB  102,15,196,192,3                    ; pinsrw        $0x3,%eax,%xmm0
   DB  102,15,56,51,208                    ; pmovzxwd      %xmm0,%xmm2
-  DB  102,15,111,5,251,34,0,0             ; movdqa        0x22fb(%rip),%xmm0        # 5240 <_sk_callback_sse41+0xb70>
+  DB  102,15,111,5,1,35,0,0               ; movdqa        0x2301(%rip),%xmm0        # 5270 <_sk_callback_sse41+0xb76>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,253,34,0,0                  ; mulps         0x22fd(%rip),%xmm0        # 5250 <_sk_callback_sse41+0xb80>
-  DB  102,15,111,13,5,35,0,0              ; movdqa        0x2305(%rip),%xmm1        # 5260 <_sk_callback_sse41+0xb90>
+  DB  15,89,5,3,35,0,0                    ; mulps         0x2303(%rip),%xmm0        # 5280 <_sk_callback_sse41+0xb86>
+  DB  102,15,111,13,11,35,0,0             ; movdqa        0x230b(%rip),%xmm1        # 5290 <_sk_callback_sse41+0xb96>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,7,35,0,0                   ; mulps         0x2307(%rip),%xmm1        # 5270 <_sk_callback_sse41+0xba0>
-  DB  102,15,219,21,15,35,0,0             ; pand          0x230f(%rip),%xmm2        # 5280 <_sk_callback_sse41+0xbb0>
+  DB  15,89,13,13,35,0,0                  ; mulps         0x230d(%rip),%xmm1        # 52a0 <_sk_callback_sse41+0xba6>
+  DB  102,15,219,21,21,35,0,0             ; pand          0x2315(%rip),%xmm2        # 52b0 <_sk_callback_sse41+0xbb6>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,21,35,0,0                  ; mulps         0x2315(%rip),%xmm2        # 5290 <_sk_callback_sse41+0xbc0>
+  DB  15,89,21,27,35,0,0                  ; mulps         0x231b(%rip),%xmm2        # 52c0 <_sk_callback_sse41+0xbc6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,28,35,0,0                  ; movaps        0x231c(%rip),%xmm3        # 52a0 <_sk_callback_sse41+0xbd0>
+  DB  15,40,29,34,35,0,0                  ; movaps        0x2322(%rip),%xmm3        # 52d0 <_sk_callback_sse41+0xbd6>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_565_sse41
 _sk_store_565_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,29,35,0,0                ; movaps        0x231d(%rip),%xmm8        # 52b0 <_sk_callback_sse41+0xbe0>
+  DB  68,15,40,5,35,35,0,0                ; movaps        0x2323(%rip),%xmm8        # 52e0 <_sk_callback_sse41+0xbe6>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
   DB  102,65,15,114,241,11                ; pslld         $0xb,%xmm9
-  DB  68,15,40,21,18,35,0,0               ; movaps        0x2312(%rip),%xmm10        # 52c0 <_sk_callback_sse41+0xbf0>
+  DB  68,15,40,21,24,35,0,0               ; movaps        0x2318(%rip),%xmm10        # 52f0 <_sk_callback_sse41+0xbf6>
   DB  68,15,89,209                        ; mulps         %xmm1,%xmm10
   DB  102,69,15,91,210                    ; cvtps2dq      %xmm10,%xmm10
   DB  102,65,15,114,242,5                 ; pslld         $0x5,%xmm10
@@ -14365,21 +14388,21 @@ _sk_load_4444_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  102,15,56,51,28,120                 ; pmovzxwd      (%rax,%rdi,2),%xmm3
-  DB  102,15,111,5,221,34,0,0             ; movdqa        0x22dd(%rip),%xmm0        # 52d0 <_sk_callback_sse41+0xc00>
+  DB  102,15,111,5,227,34,0,0             ; movdqa        0x22e3(%rip),%xmm0        # 5300 <_sk_callback_sse41+0xc06>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,223,34,0,0                  ; mulps         0x22df(%rip),%xmm0        # 52e0 <_sk_callback_sse41+0xc10>
-  DB  102,15,111,13,231,34,0,0            ; movdqa        0x22e7(%rip),%xmm1        # 52f0 <_sk_callback_sse41+0xc20>
+  DB  15,89,5,229,34,0,0                  ; mulps         0x22e5(%rip),%xmm0        # 5310 <_sk_callback_sse41+0xc16>
+  DB  102,15,111,13,237,34,0,0            ; movdqa        0x22ed(%rip),%xmm1        # 5320 <_sk_callback_sse41+0xc26>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,233,34,0,0                 ; mulps         0x22e9(%rip),%xmm1        # 5300 <_sk_callback_sse41+0xc30>
-  DB  102,15,111,21,241,34,0,0            ; movdqa        0x22f1(%rip),%xmm2        # 5310 <_sk_callback_sse41+0xc40>
+  DB  15,89,13,239,34,0,0                 ; mulps         0x22ef(%rip),%xmm1        # 5330 <_sk_callback_sse41+0xc36>
+  DB  102,15,111,21,247,34,0,0            ; movdqa        0x22f7(%rip),%xmm2        # 5340 <_sk_callback_sse41+0xc46>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,243,34,0,0                 ; mulps         0x22f3(%rip),%xmm2        # 5320 <_sk_callback_sse41+0xc50>
-  DB  102,15,219,29,251,34,0,0            ; pand          0x22fb(%rip),%xmm3        # 5330 <_sk_callback_sse41+0xc60>
+  DB  15,89,21,249,34,0,0                 ; mulps         0x22f9(%rip),%xmm2        # 5350 <_sk_callback_sse41+0xc56>
+  DB  102,15,219,29,1,35,0,0              ; pand          0x2301(%rip),%xmm3        # 5360 <_sk_callback_sse41+0xc66>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,1,35,0,0                   ; mulps         0x2301(%rip),%xmm3        # 5340 <_sk_callback_sse41+0xc70>
+  DB  15,89,29,7,35,0,0                   ; mulps         0x2307(%rip),%xmm3        # 5370 <_sk_callback_sse41+0xc76>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -14406,21 +14429,21 @@ _sk_gather_4444_sse41 LABEL PROC
   DB  65,15,183,4,65                      ; movzwl        (%r9,%rax,2),%eax
   DB  102,15,196,192,3                    ; pinsrw        $0x3,%eax,%xmm0
   DB  102,15,56,51,216                    ; pmovzxwd      %xmm0,%xmm3
-  DB  102,15,111,5,164,34,0,0             ; movdqa        0x22a4(%rip),%xmm0        # 5350 <_sk_callback_sse41+0xc80>
+  DB  102,15,111,5,170,34,0,0             ; movdqa        0x22aa(%rip),%xmm0        # 5380 <_sk_callback_sse41+0xc86>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,166,34,0,0                  ; mulps         0x22a6(%rip),%xmm0        # 5360 <_sk_callback_sse41+0xc90>
-  DB  102,15,111,13,174,34,0,0            ; movdqa        0x22ae(%rip),%xmm1        # 5370 <_sk_callback_sse41+0xca0>
+  DB  15,89,5,172,34,0,0                  ; mulps         0x22ac(%rip),%xmm0        # 5390 <_sk_callback_sse41+0xc96>
+  DB  102,15,111,13,180,34,0,0            ; movdqa        0x22b4(%rip),%xmm1        # 53a0 <_sk_callback_sse41+0xca6>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,176,34,0,0                 ; mulps         0x22b0(%rip),%xmm1        # 5380 <_sk_callback_sse41+0xcb0>
-  DB  102,15,111,21,184,34,0,0            ; movdqa        0x22b8(%rip),%xmm2        # 5390 <_sk_callback_sse41+0xcc0>
+  DB  15,89,13,182,34,0,0                 ; mulps         0x22b6(%rip),%xmm1        # 53b0 <_sk_callback_sse41+0xcb6>
+  DB  102,15,111,21,190,34,0,0            ; movdqa        0x22be(%rip),%xmm2        # 53c0 <_sk_callback_sse41+0xcc6>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,186,34,0,0                 ; mulps         0x22ba(%rip),%xmm2        # 53a0 <_sk_callback_sse41+0xcd0>
-  DB  102,15,219,29,194,34,0,0            ; pand          0x22c2(%rip),%xmm3        # 53b0 <_sk_callback_sse41+0xce0>
+  DB  15,89,21,192,34,0,0                 ; mulps         0x22c0(%rip),%xmm2        # 53d0 <_sk_callback_sse41+0xcd6>
+  DB  102,15,219,29,200,34,0,0            ; pand          0x22c8(%rip),%xmm3        # 53e0 <_sk_callback_sse41+0xce6>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,200,34,0,0                 ; mulps         0x22c8(%rip),%xmm3        # 53c0 <_sk_callback_sse41+0xcf0>
+  DB  15,89,29,206,34,0,0                 ; mulps         0x22ce(%rip),%xmm3        # 53f0 <_sk_callback_sse41+0xcf6>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -14428,7 +14451,7 @@ PUBLIC _sk_store_4444_sse41
 _sk_store_4444_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,199,34,0,0               ; movaps        0x22c7(%rip),%xmm8        # 53d0 <_sk_callback_sse41+0xd00>
+  DB  68,15,40,5,205,34,0,0               ; movaps        0x22cd(%rip),%xmm8        # 5400 <_sk_callback_sse41+0xd06>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -14456,17 +14479,17 @@ _sk_load_8888_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  15,16,28,184                        ; movups        (%rax,%rdi,4),%xmm3
-  DB  15,40,5,102,34,0,0                  ; movaps        0x2266(%rip),%xmm0        # 53e0 <_sk_callback_sse41+0xd10>
+  DB  15,40,5,108,34,0,0                  ; movaps        0x226c(%rip),%xmm0        # 5410 <_sk_callback_sse41+0xd16>
   DB  15,84,195                           ; andps         %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,104,34,0,0               ; movaps        0x2268(%rip),%xmm8        # 53f0 <_sk_callback_sse41+0xd20>
+  DB  68,15,40,5,110,34,0,0               ; movaps        0x226e(%rip),%xmm8        # 5420 <_sk_callback_sse41+0xd26>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,40,203                           ; movaps        %xmm3,%xmm1
-  DB  102,15,56,0,13,104,34,0,0           ; pshufb        0x2268(%rip),%xmm1        # 5400 <_sk_callback_sse41+0xd30>
+  DB  102,15,56,0,13,110,34,0,0           ; pshufb        0x226e(%rip),%xmm1        # 5430 <_sk_callback_sse41+0xd36>
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  15,40,211                           ; movaps        %xmm3,%xmm2
-  DB  102,15,56,0,21,101,34,0,0           ; pshufb        0x2265(%rip),%xmm2        # 5410 <_sk_callback_sse41+0xd40>
+  DB  102,15,56,0,21,107,34,0,0           ; pshufb        0x226b(%rip),%xmm2        # 5440 <_sk_callback_sse41+0xd46>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  102,15,114,211,24                   ; psrld         $0x18,%xmm3
@@ -14495,17 +14518,17 @@ _sk_gather_8888_sse41 LABEL PROC
   DB  102,65,15,58,34,28,129,1            ; pinsrd        $0x1,(%r9,%rax,4),%xmm3
   DB  102,67,15,58,34,28,145,2            ; pinsrd        $0x2,(%r9,%r10,4),%xmm3
   DB  102,65,15,58,34,28,137,3            ; pinsrd        $0x3,(%r9,%rcx,4),%xmm3
-  DB  102,15,111,5,254,33,0,0             ; movdqa        0x21fe(%rip),%xmm0        # 5420 <_sk_callback_sse41+0xd50>
+  DB  102,15,111,5,4,34,0,0               ; movdqa        0x2204(%rip),%xmm0        # 5450 <_sk_callback_sse41+0xd56>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,255,33,0,0               ; movaps        0x21ff(%rip),%xmm8        # 5430 <_sk_callback_sse41+0xd60>
+  DB  68,15,40,5,5,34,0,0                 ; movaps        0x2205(%rip),%xmm8        # 5460 <_sk_callback_sse41+0xd66>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
-  DB  102,15,56,0,13,254,33,0,0           ; pshufb        0x21fe(%rip),%xmm1        # 5440 <_sk_callback_sse41+0xd70>
+  DB  102,15,56,0,13,4,34,0,0             ; pshufb        0x2204(%rip),%xmm1        # 5470 <_sk_callback_sse41+0xd76>
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,111,211                      ; movdqa        %xmm3,%xmm2
-  DB  102,15,56,0,21,250,33,0,0           ; pshufb        0x21fa(%rip),%xmm2        # 5450 <_sk_callback_sse41+0xd80>
+  DB  102,15,56,0,21,0,34,0,0             ; pshufb        0x2200(%rip),%xmm2        # 5480 <_sk_callback_sse41+0xd86>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  102,15,114,211,24                   ; psrld         $0x18,%xmm3
@@ -14518,7 +14541,7 @@ PUBLIC _sk_store_8888_sse41
 _sk_store_8888_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,230,33,0,0               ; movaps        0x21e6(%rip),%xmm8        # 5460 <_sk_callback_sse41+0xd90>
+  DB  68,15,40,5,236,33,0,0               ; movaps        0x21ec(%rip),%xmm8        # 5490 <_sk_callback_sse41+0xd96>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -14553,18 +14576,18 @@ _sk_load_f16_sse41 LABEL PROC
   DB  102,68,15,97,216                    ; punpcklwd     %xmm0,%xmm11
   DB  102,68,15,105,200                   ; punpckhwd     %xmm0,%xmm9
   DB  102,65,15,56,51,203                 ; pmovzxwd      %xmm11,%xmm1
-  DB  102,68,15,111,5,95,33,0,0           ; movdqa        0x215f(%rip),%xmm8        # 5470 <_sk_callback_sse41+0xda0>
+  DB  102,68,15,111,5,101,33,0,0          ; movdqa        0x2165(%rip),%xmm8        # 54a0 <_sk_callback_sse41+0xda6>
   DB  102,15,111,209                      ; movdqa        %xmm1,%xmm2
   DB  102,65,15,219,208                   ; pand          %xmm8,%xmm2
   DB  102,15,239,202                      ; pxor          %xmm2,%xmm1
-  DB  102,15,111,29,90,33,0,0             ; movdqa        0x215a(%rip),%xmm3        # 5480 <_sk_callback_sse41+0xdb0>
+  DB  102,15,111,29,96,33,0,0             ; movdqa        0x2160(%rip),%xmm3        # 54b0 <_sk_callback_sse41+0xdb6>
   DB  102,15,114,242,16                   ; pslld         $0x10,%xmm2
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,15,56,63,195                    ; pmaxud        %xmm3,%xmm0
   DB  102,15,118,193                      ; pcmpeqd       %xmm1,%xmm0
   DB  102,15,114,241,13                   ; pslld         $0xd,%xmm1
   DB  102,15,235,202                      ; por           %xmm2,%xmm1
-  DB  102,68,15,111,21,70,33,0,0          ; movdqa        0x2146(%rip),%xmm10        # 5490 <_sk_callback_sse41+0xdc0>
+  DB  102,68,15,111,21,76,33,0,0          ; movdqa        0x214c(%rip),%xmm10        # 54c0 <_sk_callback_sse41+0xdc6>
   DB  102,65,15,254,202                   ; paddd         %xmm10,%xmm1
   DB  102,15,219,193                      ; pand          %xmm1,%xmm0
   DB  102,65,15,115,219,8                 ; psrldq        $0x8,%xmm11
@@ -14635,18 +14658,18 @@ _sk_gather_f16_sse41 LABEL PROC
   DB  102,68,15,97,218                    ; punpcklwd     %xmm2,%xmm11
   DB  102,68,15,105,202                   ; punpckhwd     %xmm2,%xmm9
   DB  102,65,15,56,51,203                 ; pmovzxwd      %xmm11,%xmm1
-  DB  102,68,15,111,5,4,32,0,0            ; movdqa        0x2004(%rip),%xmm8        # 54a0 <_sk_callback_sse41+0xdd0>
+  DB  102,68,15,111,5,10,32,0,0           ; movdqa        0x200a(%rip),%xmm8        # 54d0 <_sk_callback_sse41+0xdd6>
   DB  102,15,111,209                      ; movdqa        %xmm1,%xmm2
   DB  102,65,15,219,208                   ; pand          %xmm8,%xmm2
   DB  102,15,239,202                      ; pxor          %xmm2,%xmm1
-  DB  102,15,111,29,255,31,0,0            ; movdqa        0x1fff(%rip),%xmm3        # 54b0 <_sk_callback_sse41+0xde0>
+  DB  102,15,111,29,5,32,0,0              ; movdqa        0x2005(%rip),%xmm3        # 54e0 <_sk_callback_sse41+0xde6>
   DB  102,15,114,242,16                   ; pslld         $0x10,%xmm2
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,15,56,63,195                    ; pmaxud        %xmm3,%xmm0
   DB  102,15,118,193                      ; pcmpeqd       %xmm1,%xmm0
   DB  102,15,114,241,13                   ; pslld         $0xd,%xmm1
   DB  102,15,235,202                      ; por           %xmm2,%xmm1
-  DB  102,68,15,111,21,235,31,0,0         ; movdqa        0x1feb(%rip),%xmm10        # 54c0 <_sk_callback_sse41+0xdf0>
+  DB  102,68,15,111,21,241,31,0,0         ; movdqa        0x1ff1(%rip),%xmm10        # 54f0 <_sk_callback_sse41+0xdf6>
   DB  102,65,15,254,202                   ; paddd         %xmm10,%xmm1
   DB  102,15,219,193                      ; pand          %xmm1,%xmm0
   DB  102,65,15,115,219,8                 ; psrldq        $0x8,%xmm11
@@ -14692,17 +14715,17 @@ PUBLIC _sk_store_f16_sse41
 _sk_store_f16_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  102,68,15,111,21,33,31,0,0          ; movdqa        0x1f21(%rip),%xmm10        # 54d0 <_sk_callback_sse41+0xe00>
+  DB  102,68,15,111,21,39,31,0,0          ; movdqa        0x1f27(%rip),%xmm10        # 5500 <_sk_callback_sse41+0xe06>
   DB  102,68,15,111,224                   ; movdqa        %xmm0,%xmm12
   DB  102,68,15,111,232                   ; movdqa        %xmm0,%xmm13
   DB  102,69,15,219,234                   ; pand          %xmm10,%xmm13
   DB  102,69,15,239,229                   ; pxor          %xmm13,%xmm12
-  DB  102,68,15,111,13,20,31,0,0          ; movdqa        0x1f14(%rip),%xmm9        # 54e0 <_sk_callback_sse41+0xe10>
+  DB  102,68,15,111,13,26,31,0,0          ; movdqa        0x1f1a(%rip),%xmm9        # 5510 <_sk_callback_sse41+0xe16>
   DB  102,65,15,114,213,16                ; psrld         $0x10,%xmm13
   DB  102,69,15,111,193                   ; movdqa        %xmm9,%xmm8
   DB  102,69,15,102,196                   ; pcmpgtd       %xmm12,%xmm8
   DB  102,65,15,114,212,13                ; psrld         $0xd,%xmm12
-  DB  102,68,15,111,29,5,31,0,0           ; movdqa        0x1f05(%rip),%xmm11        # 54f0 <_sk_callback_sse41+0xe20>
+  DB  102,68,15,111,29,11,31,0,0          ; movdqa        0x1f0b(%rip),%xmm11        # 5520 <_sk_callback_sse41+0xe26>
   DB  102,69,15,235,235                   ; por           %xmm11,%xmm13
   DB  102,69,15,254,236                   ; paddd         %xmm12,%xmm13
   DB  102,69,15,223,197                   ; pandn         %xmm13,%xmm8
@@ -14770,7 +14793,7 @@ _sk_load_u16_be_sse41 LABEL PROC
   DB  102,15,235,200                      ; por           %xmm0,%xmm1
   DB  102,15,56,51,193                    ; pmovzxwd      %xmm1,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,212,29,0,0               ; movaps        0x1dd4(%rip),%xmm8        # 5500 <_sk_callback_sse41+0xe30>
+  DB  68,15,40,5,218,29,0,0               ; movaps        0x1dda(%rip),%xmm8        # 5530 <_sk_callback_sse41+0xe36>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -14820,7 +14843,7 @@ _sk_load_rgb_u16_be_sse41 LABEL PROC
   DB  102,15,235,193                      ; por           %xmm1,%xmm0
   DB  102,15,56,51,192                    ; pmovzxwd      %xmm0,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,21,29,0,0                ; movaps        0x1d15(%rip),%xmm8        # 5510 <_sk_callback_sse41+0xe40>
+  DB  68,15,40,5,27,29,0,0                ; movaps        0x1d1b(%rip),%xmm8        # 5540 <_sk_callback_sse41+0xe46>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -14837,14 +14860,14 @@ _sk_load_rgb_u16_be_sse41 LABEL PROC
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,220,28,0,0                 ; movaps        0x1cdc(%rip),%xmm3        # 5520 <_sk_callback_sse41+0xe50>
+  DB  15,40,29,226,28,0,0                 ; movaps        0x1ce2(%rip),%xmm3        # 5550 <_sk_callback_sse41+0xe56>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_u16_be_sse41
 _sk_store_u16_be_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,13,221,28,0,0              ; movaps        0x1cdd(%rip),%xmm9        # 5530 <_sk_callback_sse41+0xe60>
+  DB  68,15,40,13,227,28,0,0              ; movaps        0x1ce3(%rip),%xmm9        # 5560 <_sk_callback_sse41+0xe66>
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
@@ -15037,10 +15060,10 @@ _sk_mirror_y_sse41 LABEL PROC
 PUBLIC _sk_luminance_to_alpha_sse41
 _sk_luminance_to_alpha_sse41 LABEL PROC
   DB  15,40,218                           ; movaps        %xmm2,%xmm3
-  DB  15,89,5,57,26,0,0                   ; mulps         0x1a39(%rip),%xmm0        # 5540 <_sk_callback_sse41+0xe70>
-  DB  15,89,13,66,26,0,0                  ; mulps         0x1a42(%rip),%xmm1        # 5550 <_sk_callback_sse41+0xe80>
+  DB  15,89,5,63,26,0,0                   ; mulps         0x1a3f(%rip),%xmm0        # 5570 <_sk_callback_sse41+0xe76>
+  DB  15,89,13,72,26,0,0                  ; mulps         0x1a48(%rip),%xmm1        # 5580 <_sk_callback_sse41+0xe86>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  15,89,29,72,26,0,0                  ; mulps         0x1a48(%rip),%xmm3        # 5560 <_sk_callback_sse41+0xe90>
+  DB  15,89,29,78,26,0,0                  ; mulps         0x1a4e(%rip),%xmm3        # 5590 <_sk_callback_sse41+0xe96>
   DB  15,88,217                           ; addps         %xmm1,%xmm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
@@ -15256,9 +15279,9 @@ _sk_evenly_spaced_gradient_sse41 LABEL PROC
   DB  72,139,8                            ; mov           (%rax),%rcx
   DB  76,139,88,8                         ; mov           0x8(%rax),%r11
   DB  72,255,201                          ; dec           %rcx
-  DB  120,7                               ; js            3e97 <_sk_evenly_spaced_gradient_sse41+0x15>
+  DB  120,7                               ; js            3ec1 <_sk_evenly_spaced_gradient_sse41+0x15>
   DB  243,72,15,42,201                    ; cvtsi2ss      %rcx,%xmm1
-  DB  235,21                              ; jmp           3eac <_sk_evenly_spaced_gradient_sse41+0x2a>
+  DB  235,21                              ; jmp           3ed6 <_sk_evenly_spaced_gradient_sse41+0x2a>
   DB  73,137,200                          ; mov           %rcx,%r8
   DB  73,209,232                          ; shr           %r8
   DB  131,225,1                           ; and           $0x1,%ecx
@@ -15347,12 +15370,12 @@ _sk_gradient_sse41 LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
   DB  73,131,248,2                        ; cmp           $0x2,%r8
-  DB  114,50                              ; jb            408f <_sk_gradient_sse41+0x41>
+  DB  114,50                              ; jb            40b9 <_sk_gradient_sse41+0x41>
   DB  72,139,72,72                        ; mov           0x48(%rax),%rcx
   DB  73,255,200                          ; dec           %r8
   DB  72,131,193,4                        ; add           $0x4,%rcx
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
-  DB  15,40,21,253,20,0,0                 ; movaps        0x14fd(%rip),%xmm2        # 5570 <_sk_callback_sse41+0xea0>
+  DB  15,40,21,3,21,0,0                   ; movaps        0x1503(%rip),%xmm2        # 55a0 <_sk_callback_sse41+0xea6>
   DB  243,15,16,25                        ; movss         (%rcx),%xmm3
   DB  15,198,219,0                        ; shufps        $0x0,%xmm3,%xmm3
   DB  15,194,216,2                        ; cmpleps       %xmm0,%xmm3
@@ -15360,7 +15383,7 @@ _sk_gradient_sse41 LABEL PROC
   DB  102,15,254,203                      ; paddd         %xmm3,%xmm1
   DB  72,131,193,4                        ; add           $0x4,%rcx
   DB  73,255,200                          ; dec           %r8
-  DB  117,228                             ; jne           4073 <_sk_gradient_sse41+0x25>
+  DB  117,228                             ; jne           409d <_sk_gradient_sse41+0x25>
   DB  65,86                               ; push          %r14
   DB  83                                  ; push          %rbx
   DB  102,73,15,58,22,201,1               ; pextrq        $0x1,%xmm1,%r9
@@ -15487,26 +15510,26 @@ _sk_xy_to_unit_angle_sse41 LABEL PROC
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,40,236                        ; movaps        %xmm12,%xmm13
   DB  69,15,89,237                        ; mulps         %xmm13,%xmm13
-  DB  68,15,40,21,159,18,0,0              ; movaps        0x129f(%rip),%xmm10        # 5580 <_sk_callback_sse41+0xeb0>
+  DB  68,15,40,21,165,18,0,0              ; movaps        0x12a5(%rip),%xmm10        # 55b0 <_sk_callback_sse41+0xeb6>
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,88,21,163,18,0,0              ; addps         0x12a3(%rip),%xmm10        # 5590 <_sk_callback_sse41+0xec0>
+  DB  68,15,88,21,169,18,0,0              ; addps         0x12a9(%rip),%xmm10        # 55c0 <_sk_callback_sse41+0xec6>
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,88,21,167,18,0,0              ; addps         0x12a7(%rip),%xmm10        # 55a0 <_sk_callback_sse41+0xed0>
+  DB  68,15,88,21,173,18,0,0              ; addps         0x12ad(%rip),%xmm10        # 55d0 <_sk_callback_sse41+0xed6>
   DB  69,15,89,213                        ; mulps         %xmm13,%xmm10
-  DB  68,15,88,21,171,18,0,0              ; addps         0x12ab(%rip),%xmm10        # 55b0 <_sk_callback_sse41+0xee0>
+  DB  68,15,88,21,177,18,0,0              ; addps         0x12b1(%rip),%xmm10        # 55e0 <_sk_callback_sse41+0xee6>
   DB  69,15,89,212                        ; mulps         %xmm12,%xmm10
   DB  65,15,194,195,1                     ; cmpltps       %xmm11,%xmm0
-  DB  68,15,40,29,170,18,0,0              ; movaps        0x12aa(%rip),%xmm11        # 55c0 <_sk_callback_sse41+0xef0>
+  DB  68,15,40,29,176,18,0,0              ; movaps        0x12b0(%rip),%xmm11        # 55f0 <_sk_callback_sse41+0xef6>
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  102,69,15,56,20,211                 ; blendvps      %xmm0,%xmm11,%xmm10
   DB  69,15,194,200,1                     ; cmpltps       %xmm8,%xmm9
-  DB  68,15,40,29,163,18,0,0              ; movaps        0x12a3(%rip),%xmm11        # 55d0 <_sk_callback_sse41+0xf00>
+  DB  68,15,40,29,169,18,0,0              ; movaps        0x12a9(%rip),%xmm11        # 5600 <_sk_callback_sse41+0xf06>
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  102,69,15,56,20,211                 ; blendvps      %xmm0,%xmm11,%xmm10
   DB  15,40,193                           ; movaps        %xmm1,%xmm0
   DB  65,15,194,192,1                     ; cmpltps       %xmm8,%xmm0
-  DB  68,15,40,13,149,18,0,0              ; movaps        0x1295(%rip),%xmm9        # 55e0 <_sk_callback_sse41+0xf10>
+  DB  68,15,40,13,155,18,0,0              ; movaps        0x129b(%rip),%xmm9        # 5610 <_sk_callback_sse41+0xf16>
   DB  69,15,92,202                        ; subps         %xmm10,%xmm9
   DB  102,69,15,56,20,209                 ; blendvps      %xmm0,%xmm9,%xmm10
   DB  69,15,194,194,7                     ; cmpordps      %xmm10,%xmm8
@@ -15528,7 +15551,7 @@ _sk_xy_to_radius_sse41 LABEL PROC
 PUBLIC _sk_save_xy_sse41
 _sk_save_xy_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,105,18,0,0               ; movaps        0x1269(%rip),%xmm8        # 55f0 <_sk_callback_sse41+0xf20>
+  DB  68,15,40,5,111,18,0,0               ; movaps        0x126f(%rip),%xmm8        # 5620 <_sk_callback_sse41+0xf26>
   DB  15,17,0                             ; movups        %xmm0,(%rax)
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,88,200                        ; addps         %xmm8,%xmm9
@@ -15568,8 +15591,8 @@ _sk_bilinear_nx_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,235,17,0,0                  ; addps         0x11eb(%rip),%xmm0        # 5600 <_sk_callback_sse41+0xf30>
-  DB  68,15,40,13,243,17,0,0              ; movaps        0x11f3(%rip),%xmm9        # 5610 <_sk_callback_sse41+0xf40>
+  DB  15,88,5,241,17,0,0                  ; addps         0x11f1(%rip),%xmm0        # 5630 <_sk_callback_sse41+0xf36>
+  DB  68,15,40,13,249,17,0,0              ; movaps        0x11f9(%rip),%xmm9        # 5640 <_sk_callback_sse41+0xf46>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15580,7 +15603,7 @@ _sk_bilinear_px_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,226,17,0,0                  ; addps         0x11e2(%rip),%xmm0        # 5620 <_sk_callback_sse41+0xf50>
+  DB  15,88,5,232,17,0,0                  ; addps         0x11e8(%rip),%xmm0        # 5650 <_sk_callback_sse41+0xf56>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15590,8 +15613,8 @@ _sk_bilinear_ny_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,212,17,0,0                 ; addps         0x11d4(%rip),%xmm1        # 5630 <_sk_callback_sse41+0xf60>
-  DB  68,15,40,13,220,17,0,0              ; movaps        0x11dc(%rip),%xmm9        # 5640 <_sk_callback_sse41+0xf70>
+  DB  15,88,13,218,17,0,0                 ; addps         0x11da(%rip),%xmm1        # 5660 <_sk_callback_sse41+0xf66>
+  DB  68,15,40,13,226,17,0,0              ; movaps        0x11e2(%rip),%xmm9        # 5670 <_sk_callback_sse41+0xf76>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15602,7 +15625,7 @@ _sk_bilinear_py_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,202,17,0,0                 ; addps         0x11ca(%rip),%xmm1        # 5650 <_sk_callback_sse41+0xf80>
+  DB  15,88,13,208,17,0,0                 ; addps         0x11d0(%rip),%xmm1        # 5680 <_sk_callback_sse41+0xf86>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15612,13 +15635,13 @@ _sk_bicubic_n3x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,189,17,0,0                  ; addps         0x11bd(%rip),%xmm0        # 5660 <_sk_callback_sse41+0xf90>
-  DB  68,15,40,13,197,17,0,0              ; movaps        0x11c5(%rip),%xmm9        # 5670 <_sk_callback_sse41+0xfa0>
+  DB  15,88,5,195,17,0,0                  ; addps         0x11c3(%rip),%xmm0        # 5690 <_sk_callback_sse41+0xf96>
+  DB  68,15,40,13,203,17,0,0              ; movaps        0x11cb(%rip),%xmm9        # 56a0 <_sk_callback_sse41+0xfa6>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,193,17,0,0              ; mulps         0x11c1(%rip),%xmm9        # 5680 <_sk_callback_sse41+0xfb0>
-  DB  68,15,88,13,201,17,0,0              ; addps         0x11c9(%rip),%xmm9        # 5690 <_sk_callback_sse41+0xfc0>
+  DB  68,15,89,13,199,17,0,0              ; mulps         0x11c7(%rip),%xmm9        # 56b0 <_sk_callback_sse41+0xfb6>
+  DB  68,15,88,13,207,17,0,0              ; addps         0x11cf(%rip),%xmm9        # 56c0 <_sk_callback_sse41+0xfc6>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15629,16 +15652,16 @@ _sk_bicubic_n1x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,184,17,0,0                  ; addps         0x11b8(%rip),%xmm0        # 56a0 <_sk_callback_sse41+0xfd0>
-  DB  68,15,40,13,192,17,0,0              ; movaps        0x11c0(%rip),%xmm9        # 56b0 <_sk_callback_sse41+0xfe0>
+  DB  15,88,5,190,17,0,0                  ; addps         0x11be(%rip),%xmm0        # 56d0 <_sk_callback_sse41+0xfd6>
+  DB  68,15,40,13,198,17,0,0              ; movaps        0x11c6(%rip),%xmm9        # 56e0 <_sk_callback_sse41+0xfe6>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,196,17,0,0               ; movaps        0x11c4(%rip),%xmm8        # 56c0 <_sk_callback_sse41+0xff0>
+  DB  68,15,40,5,202,17,0,0               ; movaps        0x11ca(%rip),%xmm8        # 56f0 <_sk_callback_sse41+0xff6>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,200,17,0,0               ; addps         0x11c8(%rip),%xmm8        # 56d0 <_sk_callback_sse41+0x1000>
+  DB  68,15,88,5,206,17,0,0               ; addps         0x11ce(%rip),%xmm8        # 5700 <_sk_callback_sse41+0x1006>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,204,17,0,0               ; addps         0x11cc(%rip),%xmm8        # 56e0 <_sk_callback_sse41+0x1010>
+  DB  68,15,88,5,210,17,0,0               ; addps         0x11d2(%rip),%xmm8        # 5710 <_sk_callback_sse41+0x1016>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,208,17,0,0               ; addps         0x11d0(%rip),%xmm8        # 56f0 <_sk_callback_sse41+0x1020>
+  DB  68,15,88,5,214,17,0,0               ; addps         0x11d6(%rip),%xmm8        # 5720 <_sk_callback_sse41+0x1026>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15646,17 +15669,17 @@ _sk_bicubic_n1x_sse41 LABEL PROC
 PUBLIC _sk_bicubic_p1x_sse41
 _sk_bicubic_p1x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,202,17,0,0               ; movaps        0x11ca(%rip),%xmm8        # 5700 <_sk_callback_sse41+0x1030>
+  DB  68,15,40,5,208,17,0,0               ; movaps        0x11d0(%rip),%xmm8        # 5730 <_sk_callback_sse41+0x1036>
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,72,64                      ; movups        0x40(%rax),%xmm9
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
-  DB  68,15,40,21,198,17,0,0              ; movaps        0x11c6(%rip),%xmm10        # 5710 <_sk_callback_sse41+0x1040>
+  DB  68,15,40,21,204,17,0,0              ; movaps        0x11cc(%rip),%xmm10        # 5740 <_sk_callback_sse41+0x1046>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,202,17,0,0              ; addps         0x11ca(%rip),%xmm10        # 5720 <_sk_callback_sse41+0x1050>
+  DB  68,15,88,21,208,17,0,0              ; addps         0x11d0(%rip),%xmm10        # 5750 <_sk_callback_sse41+0x1056>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,198,17,0,0              ; addps         0x11c6(%rip),%xmm10        # 5730 <_sk_callback_sse41+0x1060>
+  DB  68,15,88,21,204,17,0,0              ; addps         0x11cc(%rip),%xmm10        # 5760 <_sk_callback_sse41+0x1066>
   DB  68,15,17,144,128,0,0,0              ; movups        %xmm10,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15666,11 +15689,11 @@ _sk_bicubic_p3x_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,185,17,0,0                  ; addps         0x11b9(%rip),%xmm0        # 5740 <_sk_callback_sse41+0x1070>
+  DB  15,88,5,191,17,0,0                  ; addps         0x11bf(%rip),%xmm0        # 5770 <_sk_callback_sse41+0x1076>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,185,17,0,0               ; mulps         0x11b9(%rip),%xmm8        # 5750 <_sk_callback_sse41+0x1080>
-  DB  68,15,88,5,193,17,0,0               ; addps         0x11c1(%rip),%xmm8        # 5760 <_sk_callback_sse41+0x1090>
+  DB  68,15,89,5,191,17,0,0               ; mulps         0x11bf(%rip),%xmm8        # 5780 <_sk_callback_sse41+0x1086>
+  DB  68,15,88,5,199,17,0,0               ; addps         0x11c7(%rip),%xmm8        # 5790 <_sk_callback_sse41+0x1096>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15681,13 +15704,13 @@ _sk_bicubic_n3y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,175,17,0,0                 ; addps         0x11af(%rip),%xmm1        # 5770 <_sk_callback_sse41+0x10a0>
-  DB  68,15,40,13,183,17,0,0              ; movaps        0x11b7(%rip),%xmm9        # 5780 <_sk_callback_sse41+0x10b0>
+  DB  15,88,13,181,17,0,0                 ; addps         0x11b5(%rip),%xmm1        # 57a0 <_sk_callback_sse41+0x10a6>
+  DB  68,15,40,13,189,17,0,0              ; movaps        0x11bd(%rip),%xmm9        # 57b0 <_sk_callback_sse41+0x10b6>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,179,17,0,0              ; mulps         0x11b3(%rip),%xmm9        # 5790 <_sk_callback_sse41+0x10c0>
-  DB  68,15,88,13,187,17,0,0              ; addps         0x11bb(%rip),%xmm9        # 57a0 <_sk_callback_sse41+0x10d0>
+  DB  68,15,89,13,185,17,0,0              ; mulps         0x11b9(%rip),%xmm9        # 57c0 <_sk_callback_sse41+0x10c6>
+  DB  68,15,88,13,193,17,0,0              ; addps         0x11c1(%rip),%xmm9        # 57d0 <_sk_callback_sse41+0x10d6>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15698,16 +15721,16 @@ _sk_bicubic_n1y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,169,17,0,0                 ; addps         0x11a9(%rip),%xmm1        # 57b0 <_sk_callback_sse41+0x10e0>
-  DB  68,15,40,13,177,17,0,0              ; movaps        0x11b1(%rip),%xmm9        # 57c0 <_sk_callback_sse41+0x10f0>
+  DB  15,88,13,175,17,0,0                 ; addps         0x11af(%rip),%xmm1        # 57e0 <_sk_callback_sse41+0x10e6>
+  DB  68,15,40,13,183,17,0,0              ; movaps        0x11b7(%rip),%xmm9        # 57f0 <_sk_callback_sse41+0x10f6>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,181,17,0,0               ; movaps        0x11b5(%rip),%xmm8        # 57d0 <_sk_callback_sse41+0x1100>
+  DB  68,15,40,5,187,17,0,0               ; movaps        0x11bb(%rip),%xmm8        # 5800 <_sk_callback_sse41+0x1106>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,185,17,0,0               ; addps         0x11b9(%rip),%xmm8        # 57e0 <_sk_callback_sse41+0x1110>
+  DB  68,15,88,5,191,17,0,0               ; addps         0x11bf(%rip),%xmm8        # 5810 <_sk_callback_sse41+0x1116>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,189,17,0,0               ; addps         0x11bd(%rip),%xmm8        # 57f0 <_sk_callback_sse41+0x1120>
+  DB  68,15,88,5,195,17,0,0               ; addps         0x11c3(%rip),%xmm8        # 5820 <_sk_callback_sse41+0x1126>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,193,17,0,0               ; addps         0x11c1(%rip),%xmm8        # 5800 <_sk_callback_sse41+0x1130>
+  DB  68,15,88,5,199,17,0,0               ; addps         0x11c7(%rip),%xmm8        # 5830 <_sk_callback_sse41+0x1136>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15715,17 +15738,17 @@ _sk_bicubic_n1y_sse41 LABEL PROC
 PUBLIC _sk_bicubic_p1y_sse41
 _sk_bicubic_p1y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,187,17,0,0               ; movaps        0x11bb(%rip),%xmm8        # 5810 <_sk_callback_sse41+0x1140>
+  DB  68,15,40,5,193,17,0,0               ; movaps        0x11c1(%rip),%xmm8        # 5840 <_sk_callback_sse41+0x1146>
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,72,96                      ; movups        0x60(%rax),%xmm9
   DB  65,15,88,200                        ; addps         %xmm8,%xmm1
-  DB  68,15,40,21,182,17,0,0              ; movaps        0x11b6(%rip),%xmm10        # 5820 <_sk_callback_sse41+0x1150>
+  DB  68,15,40,21,188,17,0,0              ; movaps        0x11bc(%rip),%xmm10        # 5850 <_sk_callback_sse41+0x1156>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,186,17,0,0              ; addps         0x11ba(%rip),%xmm10        # 5830 <_sk_callback_sse41+0x1160>
+  DB  68,15,88,21,192,17,0,0              ; addps         0x11c0(%rip),%xmm10        # 5860 <_sk_callback_sse41+0x1166>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,182,17,0,0              ; addps         0x11b6(%rip),%xmm10        # 5840 <_sk_callback_sse41+0x1170>
+  DB  68,15,88,21,188,17,0,0              ; addps         0x11bc(%rip),%xmm10        # 5870 <_sk_callback_sse41+0x1176>
   DB  68,15,17,144,160,0,0,0              ; movups        %xmm10,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -15735,11 +15758,11 @@ _sk_bicubic_p3y_sse41 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,168,17,0,0                 ; addps         0x11a8(%rip),%xmm1        # 5850 <_sk_callback_sse41+0x1180>
+  DB  15,88,13,174,17,0,0                 ; addps         0x11ae(%rip),%xmm1        # 5880 <_sk_callback_sse41+0x1186>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,168,17,0,0               ; mulps         0x11a8(%rip),%xmm8        # 5860 <_sk_callback_sse41+0x1190>
-  DB  68,15,88,5,176,17,0,0               ; addps         0x11b0(%rip),%xmm8        # 5870 <_sk_callback_sse41+0x11a0>
+  DB  68,15,89,5,174,17,0,0               ; mulps         0x11ae(%rip),%xmm8        # 5890 <_sk_callback_sse41+0x1196>
+  DB  68,15,88,5,182,17,0,0               ; addps         0x11b6(%rip),%xmm8        # 58a0 <_sk_callback_sse41+0x11a6>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -15944,11 +15967,11 @@ ALIGN 16
   DB  128,191,0,0,128,191,0               ; cmpb          $0x0,-0x40800000(%rdi)
   DB  0,224                               ; add           %ah,%al
   DB  64,0,0                              ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4958 <.literal16+0x1d8>
+  DB  224,64                              ; loopne        4988 <.literal16+0x1d8>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        495c <.literal16+0x1dc>
+  DB  224,64                              ; loopne        498c <.literal16+0x1dc>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4960 <.literal16+0x1e0>
+  DB  224,64                              ; loopne        4990 <.literal16+0x1e0>
   DB  154                                 ; (bad)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
@@ -15968,13 +15991,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4981 <.literal16+0x201>
+  DB  71,225,61                           ; rex.RXB       loope 49b1 <.literal16+0x201>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4985 <.literal16+0x205>
+  DB  71,225,61                           ; rex.RXB       loope 49b5 <.literal16+0x205>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4989 <.literal16+0x209>
+  DB  71,225,61                           ; rex.RXB       loope 49b9 <.literal16+0x209>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 498d <.literal16+0x20d>
+  DB  71,225,61                           ; rex.RXB       loope 49bd <.literal16+0x20d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -15999,13 +16022,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 49c1 <.literal16+0x241>
+  DB  71,225,61                           ; rex.RXB       loope 49f1 <.literal16+0x241>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 49c5 <.literal16+0x245>
+  DB  71,225,61                           ; rex.RXB       loope 49f5 <.literal16+0x245>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 49c9 <.literal16+0x249>
+  DB  71,225,61                           ; rex.RXB       loope 49f9 <.literal16+0x249>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 49cd <.literal16+0x24d>
+  DB  71,225,61                           ; rex.RXB       loope 49fd <.literal16+0x24d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -16030,13 +16053,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a01 <.literal16+0x281>
+  DB  71,225,61                           ; rex.RXB       loope 4a31 <.literal16+0x281>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a05 <.literal16+0x285>
+  DB  71,225,61                           ; rex.RXB       loope 4a35 <.literal16+0x285>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a09 <.literal16+0x289>
+  DB  71,225,61                           ; rex.RXB       loope 4a39 <.literal16+0x289>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a0d <.literal16+0x28d>
+  DB  71,225,61                           ; rex.RXB       loope 4a3d <.literal16+0x28d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -16061,13 +16084,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a41 <.literal16+0x2c1>
+  DB  71,225,61                           ; rex.RXB       loope 4a71 <.literal16+0x2c1>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a45 <.literal16+0x2c5>
+  DB  71,225,61                           ; rex.RXB       loope 4a75 <.literal16+0x2c5>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a49 <.literal16+0x2c9>
+  DB  71,225,61                           ; rex.RXB       loope 4a79 <.literal16+0x2c9>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4a4d <.literal16+0x2cd>
+  DB  71,225,61                           ; rex.RXB       loope 4a7d <.literal16+0x2cd>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -16291,13 +16314,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        4c19 <.literal16+0x499>
+  DB  224,7                               ; loopne        4c49 <.literal16+0x499>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4c1d <.literal16+0x49d>
+  DB  224,7                               ; loopne        4c4d <.literal16+0x49d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4c21 <.literal16+0x4a1>
+  DB  224,7                               ; loopne        4c51 <.literal16+0x4a1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        4c25 <.literal16+0x4a5>
+  DB  224,7                               ; loopne        4c55 <.literal16+0x4a5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -16331,10 +16354,10 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  1,255                               ; add           %edi,%edi
   DB  255                                 ; (bad)
-  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004c68 <_sk_callback_sse41+0xa000598>
+  DB  255,5,255,255,255,9                 ; incl          0x9ffffff(%rip)        # a004c98 <_sk_callback_sse41+0xa00059e>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3004c70 <_sk_callback_sse41+0x30005a0>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3004ca0 <_sk_callback_sse41+0x30005a6>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -16389,11 +16412,11 @@ ALIGN 16
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4d3b <.literal16+0x5bb>
+  DB  127,67                              ; jg            4d6b <.literal16+0x5bb>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4d3f <.literal16+0x5bf>
+  DB  127,67                              ; jg            4d6f <.literal16+0x5bf>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            4d43 <.literal16+0x5c3>
+  DB  127,67                              ; jg            4d73 <.literal16+0x5c3>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,129,128,128,59           ; addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -16408,16 +16431,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4d34 <.literal16+0x5b4>
+  DB  127,0                               ; jg            4d64 <.literal16+0x5b4>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4d38 <.literal16+0x5b8>
+  DB  127,0                               ; jg            4d68 <.literal16+0x5b8>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4d3c <.literal16+0x5bc>
+  DB  127,0                               ; jg            4d6c <.literal16+0x5bc>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4d40 <.literal16+0x5c0>
+  DB  127,0                               ; jg            4d70 <.literal16+0x5c0>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16426,7 +16449,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4dc5 <.literal16+0x645>
+  DB  119,115                             ; ja            4df5 <.literal16+0x645>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -16437,7 +16460,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4d29 <.literal16+0x5a9>
+  DB  117,191                             ; jne           4d59 <.literal16+0x5a9>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -16449,7 +16472,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38d6a <_sk_callback_sse41+0xffffffffe9a3469a>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38d9a <_sk_callback_sse41+0xffffffffe9a346a0>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -16504,16 +16527,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e04 <.literal16+0x684>
+  DB  127,0                               ; jg            4e34 <.literal16+0x684>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e08 <.literal16+0x688>
+  DB  127,0                               ; jg            4e38 <.literal16+0x688>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e0c <.literal16+0x68c>
+  DB  127,0                               ; jg            4e3c <.literal16+0x68c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4e10 <.literal16+0x690>
+  DB  127,0                               ; jg            4e40 <.literal16+0x690>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16522,7 +16545,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4e95 <.literal16+0x715>
+  DB  119,115                             ; ja            4ec5 <.literal16+0x715>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -16533,7 +16556,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4df9 <.literal16+0x679>
+  DB  117,191                             ; jne           4e29 <.literal16+0x679>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -16545,7 +16568,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38e3a <_sk_callback_sse41+0xffffffffe9a3476a>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38e6a <_sk_callback_sse41+0xffffffffe9a34770>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -16600,16 +16623,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4ed4 <.literal16+0x754>
+  DB  127,0                               ; jg            4f04 <.literal16+0x754>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4ed8 <.literal16+0x758>
+  DB  127,0                               ; jg            4f08 <.literal16+0x758>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4edc <.literal16+0x75c>
+  DB  127,0                               ; jg            4f0c <.literal16+0x75c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4ee0 <.literal16+0x760>
+  DB  127,0                               ; jg            4f10 <.literal16+0x760>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16618,7 +16641,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            4f65 <.literal16+0x7e5>
+  DB  119,115                             ; ja            4f95 <.literal16+0x7e5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -16629,7 +16652,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4ec9 <.literal16+0x749>
+  DB  117,191                             ; jne           4ef9 <.literal16+0x749>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -16641,7 +16664,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38f0a <_sk_callback_sse41+0xffffffffe9a3483a>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38f3a <_sk_callback_sse41+0xffffffffe9a34840>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -16696,16 +16719,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fa4 <.literal16+0x824>
+  DB  127,0                               ; jg            4fd4 <.literal16+0x824>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fa8 <.literal16+0x828>
+  DB  127,0                               ; jg            4fd8 <.literal16+0x828>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fac <.literal16+0x82c>
+  DB  127,0                               ; jg            4fdc <.literal16+0x82c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            4fb0 <.literal16+0x830>
+  DB  127,0                               ; jg            4fe0 <.literal16+0x830>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -16714,7 +16737,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5035 <.literal16+0x8b5>
+  DB  119,115                             ; ja            5065 <.literal16+0x8b5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -16725,7 +16748,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           4f99 <.literal16+0x819>
+  DB  117,191                             ; jne           4fc9 <.literal16+0x819>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -16737,7 +16760,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a38fda <_sk_callback_sse41+0xffffffffe9a3490a>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3900a <_sk_callback_sse41+0xffffffffe9a34910>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  81                                  ; push          %rcx
   DB  140,242                             ; mov           %?,%edx
@@ -16788,13 +16811,13 @@ ALIGN 16
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
-  DB  127,67                              ; jg            50b7 <.literal16+0x937>
+  DB  127,67                              ; jg            50e7 <.literal16+0x937>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            50bb <.literal16+0x93b>
+  DB  127,67                              ; jg            50eb <.literal16+0x93b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            50bf <.literal16+0x93f>
+  DB  127,67                              ; jg            50ef <.literal16+0x93f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            50c3 <.literal16+0x943>
+  DB  127,67                              ; jg            50f3 <.literal16+0x943>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -16841,16 +16864,16 @@ ALIGN 16
   DB  128,3,62                            ; addb          $0x3e,(%rbx)
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5143 <.literal16+0x9c3>
+  DB  118,63                              ; jbe           5173 <.literal16+0x9c3>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5147 <.literal16+0x9c7>
+  DB  118,63                              ; jbe           5177 <.literal16+0x9c7>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           514b <.literal16+0x9cb>
+  DB  118,63                              ; jbe           517b <.literal16+0x9cb>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           514f <.literal16+0x9cf>
+  DB  118,63                              ; jbe           517f <.literal16+0x9cf>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
@@ -16862,11 +16885,11 @@ ALIGN 16
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            518b <.literal16+0xa0b>
+  DB  127,67                              ; jg            51bb <.literal16+0xa0b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            518f <.literal16+0xa0f>
+  DB  127,67                              ; jg            51bf <.literal16+0xa0f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            5193 <.literal16+0xa13>
+  DB  127,67                              ; jg            51c3 <.literal16+0xa13>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,0,0,128,63               ; addb          $0x3f,-0x7fffffc5(%rax)
@@ -16895,7 +16918,7 @@ ALIGN 16
   DB  5,255,255,255,9                     ; add           $0x9ffffff,%eax
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 30051c0 <_sk_callback_sse41+0x3000af0>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 30051f0 <_sk_callback_sse41+0x3000af6>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -16924,13 +16947,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        51f9 <.literal16+0xa79>
+  DB  224,7                               ; loopne        5229 <.literal16+0xa79>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        51fd <.literal16+0xa7d>
+  DB  224,7                               ; loopne        522d <.literal16+0xa7d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5201 <.literal16+0xa81>
+  DB  224,7                               ; loopne        5231 <.literal16+0xa81>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5205 <.literal16+0xa85>
+  DB  224,7                               ; loopne        5235 <.literal16+0xa85>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -16976,13 +16999,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5269 <.literal16+0xae9>
+  DB  224,7                               ; loopne        5299 <.literal16+0xae9>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        526d <.literal16+0xaed>
+  DB  224,7                               ; loopne        529d <.literal16+0xaed>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5271 <.literal16+0xaf1>
+  DB  224,7                               ; loopne        52a1 <.literal16+0xaf1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5275 <.literal16+0xaf5>
+  DB  224,7                               ; loopne        52a5 <.literal16+0xaf5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -17020,13 +17043,13 @@ ALIGN 16
   DB  65,0,0                              ; add           %al,(%r8)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            5306 <.literal16+0xb86>
+  DB  124,66                              ; jl            5336 <.literal16+0xb86>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            530a <.literal16+0xb8a>
+  DB  124,66                              ; jl            533a <.literal16+0xb8a>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            530e <.literal16+0xb8e>
+  DB  124,66                              ; jl            533e <.literal16+0xb8e>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            5312 <.literal16+0xb92>
+  DB  124,66                              ; jl            5342 <.literal16+0xb92>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,240                               ; add           %dh,%al
@@ -17116,13 +17139,13 @@ ALIGN 16
   DB  136,136,61,137,136,136              ; mov           %cl,-0x777776c3(%rax)
   DB  61,137,136,136,61                   ; cmp           $0x3d888889,%eax
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5415 <.literal16+0xc95>
+  DB  112,65                              ; jo            5445 <.literal16+0xc95>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5419 <.literal16+0xc99>
+  DB  112,65                              ; jo            5449 <.literal16+0xc99>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            541d <.literal16+0xc9d>
+  DB  112,65                              ; jo            544d <.literal16+0xc9d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5421 <.literal16+0xca1>
+  DB  112,65                              ; jo            5451 <.literal16+0xca1>
   DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  255,0                               ; incl          (%rax)
@@ -17137,7 +17160,7 @@ ALIGN 16
   DB  5,255,255,255,9                     ; add           $0x9ffffff,%eax
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3005410 <_sk_callback_sse41+0x3000d40>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3005440 <_sk_callback_sse41+0x3000d46>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -17164,7 +17187,7 @@ ALIGN 16
   DB  5,255,255,255,9                     ; add           $0x9ffffff,%eax
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3005450 <_sk_callback_sse41+0x3000d80>
+  DB  255,13,255,255,255,2                ; decl          0x2ffffff(%rip)        # 3005480 <_sk_callback_sse41+0x3000d86>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
   DB  255,6                               ; incl          (%rsi)
@@ -17179,11 +17202,11 @@ ALIGN 16
   DB  255,0                               ; incl          (%rax)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            54ab <.literal16+0xd2b>
+  DB  127,67                              ; jg            54db <.literal16+0xd2b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            54af <.literal16+0xd2f>
+  DB  127,67                              ; jg            54df <.literal16+0xd2f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            54b3 <.literal16+0xd33>
+  DB  127,67                              ; jg            54e3 <.literal16+0xd33>
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
@@ -17259,13 +17282,13 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  255                                 ; (bad)
-  DB  127,71                              ; jg            557b <.literal16+0xdfb>
+  DB  127,71                              ; jg            55ab <.literal16+0xdfb>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            557f <.literal16+0xdff>
+  DB  127,71                              ; jg            55af <.literal16+0xdff>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            5583 <.literal16+0xe03>
+  DB  127,71                              ; jg            55b3 <.literal16+0xe03>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            5587 <.literal16+0xe07>
+  DB  127,71                              ; jg            55b7 <.literal16+0xe07>
   DB  208                                 ; (bad)
   DB  179,89                              ; mov           $0x59,%bl
   DB  62,208                              ; ds            (bad)
@@ -17399,11 +17422,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         56a2 <.literal16+0xf22>
+  DB  62,114,28                           ; jb,pt         56d2 <.literal16+0xf22>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         56a6 <.literal16+0xf26>
+  DB  62,114,28                           ; jb,pt         56d6 <.literal16+0xf26>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         56aa <.literal16+0xf2a>
+  DB  62,114,28                           ; jb,pt         56da <.literal16+0xf2a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -17447,7 +17470,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e535 <_sk_callback_sse41+0x3d639e65>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e565 <_sk_callback_sse41+0x3d639e6b>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -17473,7 +17496,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e575 <_sk_callback_sse41+0x3d639ea5>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e5a5 <_sk_callback_sse41+0x3d639eab>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -17482,13 +17505,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            576e <.literal16+0xfee>
+  DB  114,28                              ; jb            579e <.literal16+0xfee>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5772 <.literal16+0xff2>
+  DB  62,114,28                           ; jb,pt         57a2 <.literal16+0xff2>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5776 <.literal16+0xff6>
+  DB  62,114,28                           ; jb,pt         57a6 <.literal16+0xff6>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         577a <.literal16+0xffa>
+  DB  62,114,28                           ; jb,pt         57aa <.literal16+0xffa>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -17509,11 +17532,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         57b2 <.literal16+0x1032>
+  DB  62,114,28                           ; jb,pt         57e2 <.literal16+0x1032>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         57b6 <.literal16+0x1036>
+  DB  62,114,28                           ; jb,pt         57e6 <.literal16+0x1036>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         57ba <.literal16+0x103a>
+  DB  62,114,28                           ; jb,pt         57ea <.literal16+0x103a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -17557,7 +17580,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e645 <_sk_callback_sse41+0x3d639f75>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e675 <_sk_callback_sse41+0x3d639f7b>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -17583,7 +17606,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e685 <_sk_callback_sse41+0x3d639fb5>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e6b5 <_sk_callback_sse41+0x3d639fbb>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -17592,13 +17615,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            587e <.literal16+0x10fe>
+  DB  114,28                              ; jb            58ae <.literal16+0x10fe>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5882 <_sk_callback_sse41+0x11b2>
+  DB  62,114,28                           ; jb,pt         58b2 <_sk_callback_sse41+0x11b8>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5886 <_sk_callback_sse41+0x11b6>
+  DB  62,114,28                           ; jb,pt         58b6 <_sk_callback_sse41+0x11bc>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         588a <_sk_callback_sse41+0x11ba>
+  DB  62,114,28                           ; jb,pt         58ba <_sk_callback_sse41+0x11c0>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -17689,7 +17712,7 @@ _sk_seed_shader_sse2 LABEL PROC
   DB  102,15,110,199                      ; movd          %edi,%xmm0
   DB  102,15,112,192,0                    ; pshufd        $0x0,%xmm0,%xmm0
   DB  15,91,200                           ; cvtdq2ps      %xmm0,%xmm1
-  DB  15,40,21,225,74,0,0                 ; movaps        0x4ae1(%rip),%xmm2        # 4bf0 <_sk_callback_sse2+0xb4>
+  DB  15,40,21,17,75,0,0                  ; movaps        0x4b11(%rip),%xmm2        # 4c20 <_sk_callback_sse2+0xba>
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  15,16,2                             ; movups        (%rdx),%xmm0
   DB  15,88,193                           ; addps         %xmm1,%xmm0
@@ -17698,7 +17721,7 @@ _sk_seed_shader_sse2 LABEL PROC
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  15,88,202                           ; addps         %xmm2,%xmm1
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,21,208,74,0,0                 ; movaps        0x4ad0(%rip),%xmm2        # 4c00 <_sk_callback_sse2+0xc4>
+  DB  15,40,21,0,75,0,0                   ; movaps        0x4b00(%rip),%xmm2        # 4c30 <_sk_callback_sse2+0xca>
   DB  15,87,219                           ; xorps         %xmm3,%xmm3
   DB  15,87,228                           ; xorps         %xmm4,%xmm4
   DB  15,87,237                           ; xorps         %xmm5,%xmm5
@@ -17719,14 +17742,14 @@ _sk_dither_sse2 LABEL PROC
   DB  102,68,15,110,1                     ; movd          (%rcx),%xmm8
   DB  102,69,15,112,192,0                 ; pshufd        $0x0,%xmm8,%xmm8
   DB  102,69,15,239,193                   ; pxor          %xmm9,%xmm8
-  DB  102,68,15,111,21,149,74,0,0         ; movdqa        0x4a95(%rip),%xmm10        # 4c10 <_sk_callback_sse2+0xd4>
+  DB  102,68,15,111,21,197,74,0,0         ; movdqa        0x4ac5(%rip),%xmm10        # 4c40 <_sk_callback_sse2+0xda>
   DB  102,69,15,111,216                   ; movdqa        %xmm8,%xmm11
   DB  102,69,15,219,218                   ; pand          %xmm10,%xmm11
   DB  102,65,15,114,243,5                 ; pslld         $0x5,%xmm11
   DB  102,69,15,219,209                   ; pand          %xmm9,%xmm10
   DB  102,65,15,114,242,4                 ; pslld         $0x4,%xmm10
-  DB  102,68,15,111,37,129,74,0,0         ; movdqa        0x4a81(%rip),%xmm12        # 4c20 <_sk_callback_sse2+0xe4>
-  DB  102,68,15,111,45,136,74,0,0         ; movdqa        0x4a88(%rip),%xmm13        # 4c30 <_sk_callback_sse2+0xf4>
+  DB  102,68,15,111,37,177,74,0,0         ; movdqa        0x4ab1(%rip),%xmm12        # 4c50 <_sk_callback_sse2+0xea>
+  DB  102,68,15,111,45,184,74,0,0         ; movdqa        0x4ab8(%rip),%xmm13        # 4c60 <_sk_callback_sse2+0xfa>
   DB  102,69,15,111,240                   ; movdqa        %xmm8,%xmm14
   DB  102,69,15,219,245                   ; pand          %xmm13,%xmm14
   DB  102,65,15,114,246,2                 ; pslld         $0x2,%xmm14
@@ -17742,15 +17765,26 @@ _sk_dither_sse2 LABEL PROC
   DB  102,69,15,235,245                   ; por           %xmm13,%xmm14
   DB  102,69,15,235,240                   ; por           %xmm8,%xmm14
   DB  69,15,91,198                        ; cvtdq2ps      %xmm14,%xmm8
-  DB  68,15,89,5,67,74,0,0                ; mulps         0x4a43(%rip),%xmm8        # 4c40 <_sk_callback_sse2+0x104>
-  DB  68,15,88,5,75,74,0,0                ; addps         0x4a4b(%rip),%xmm8        # 4c50 <_sk_callback_sse2+0x114>
-  DB  243,68,15,16,72,8                   ; movss         0x8(%rax),%xmm9
-  DB  69,15,198,201,0                     ; shufps        $0x0,%xmm9,%xmm9
-  DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
-  DB  65,15,88,193                        ; addps         %xmm9,%xmm0
-  DB  65,15,88,201                        ; addps         %xmm9,%xmm1
-  DB  65,15,88,209                        ; addps         %xmm9,%xmm2
+  DB  68,15,89,5,115,74,0,0               ; mulps         0x4a73(%rip),%xmm8        # 4c70 <_sk_callback_sse2+0x10a>
+  DB  68,15,88,5,123,74,0,0               ; addps         0x4a7b(%rip),%xmm8        # 4c80 <_sk_callback_sse2+0x11a>
+  DB  243,68,15,16,80,8                   ; movss         0x8(%rax),%xmm10
+  DB  69,15,198,210,0                     ; shufps        $0x0,%xmm10,%xmm10
+  DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
+  DB  65,15,88,194                        ; addps         %xmm10,%xmm0
+  DB  65,15,88,202                        ; addps         %xmm10,%xmm1
+  DB  68,15,88,210                        ; addps         %xmm2,%xmm10
+  DB  15,93,195                           ; minps         %xmm3,%xmm0
+  DB  15,87,210                           ; xorps         %xmm2,%xmm2
+  DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
+  DB  68,15,95,192                        ; maxps         %xmm0,%xmm8
+  DB  15,93,203                           ; minps         %xmm3,%xmm1
+  DB  102,69,15,239,201                   ; pxor          %xmm9,%xmm9
+  DB  68,15,95,201                        ; maxps         %xmm1,%xmm9
+  DB  68,15,93,211                        ; minps         %xmm3,%xmm10
+  DB  65,15,95,210                        ; maxps         %xmm10,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
+  DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
+  DB  65,15,40,201                        ; movaps        %xmm9,%xmm1
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_constant_color_sse2
@@ -17799,7 +17833,7 @@ _sk_clear_sse2 LABEL PROC
 PUBLIC _sk_srcatop_sse2
 _sk_srcatop_sse2 LABEL PROC
   DB  15,89,199                           ; mulps         %xmm7,%xmm0
-  DB  68,15,40,5,206,73,0,0               ; movaps        0x49ce(%rip),%xmm8        # 4c60 <_sk_callback_sse2+0x124>
+  DB  68,15,40,5,212,73,0,0               ; movaps        0x49d4(%rip),%xmm8        # 4c90 <_sk_callback_sse2+0x12a>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -17822,7 +17856,7 @@ PUBLIC _sk_dstatop_sse2
 _sk_dstatop_sse2 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
   DB  68,15,89,196                        ; mulps         %xmm4,%xmm8
-  DB  68,15,40,13,145,73,0,0              ; movaps        0x4991(%rip),%xmm9        # 4c70 <_sk_callback_sse2+0x134>
+  DB  68,15,40,13,151,73,0,0              ; movaps        0x4997(%rip),%xmm9        # 4ca0 <_sk_callback_sse2+0x13a>
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
@@ -17863,7 +17897,7 @@ _sk_dstin_sse2 LABEL PROC
 
 PUBLIC _sk_srcout_sse2
 _sk_srcout_sse2 LABEL PROC
-  DB  68,15,40,5,53,73,0,0                ; movaps        0x4935(%rip),%xmm8        # 4c80 <_sk_callback_sse2+0x144>
+  DB  68,15,40,5,59,73,0,0                ; movaps        0x493b(%rip),%xmm8        # 4cb0 <_sk_callback_sse2+0x14a>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
@@ -17874,7 +17908,7 @@ _sk_srcout_sse2 LABEL PROC
 
 PUBLIC _sk_dstout_sse2
 _sk_dstout_sse2 LABEL PROC
-  DB  68,15,40,5,37,73,0,0                ; movaps        0x4925(%rip),%xmm8        # 4c90 <_sk_callback_sse2+0x154>
+  DB  68,15,40,5,43,73,0,0                ; movaps        0x492b(%rip),%xmm8        # 4cc0 <_sk_callback_sse2+0x15a>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  15,89,196                           ; mulps         %xmm4,%xmm0
@@ -17889,7 +17923,7 @@ _sk_dstout_sse2 LABEL PROC
 
 PUBLIC _sk_srcover_sse2
 _sk_srcover_sse2 LABEL PROC
-  DB  68,15,40,5,8,73,0,0                 ; movaps        0x4908(%rip),%xmm8        # 4ca0 <_sk_callback_sse2+0x164>
+  DB  68,15,40,5,14,73,0,0                ; movaps        0x490e(%rip),%xmm8        # 4cd0 <_sk_callback_sse2+0x16a>
   DB  68,15,92,195                        ; subps         %xmm3,%xmm8
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
@@ -17907,7 +17941,7 @@ _sk_srcover_sse2 LABEL PROC
 
 PUBLIC _sk_dstover_sse2
 _sk_dstover_sse2 LABEL PROC
-  DB  68,15,40,5,220,72,0,0               ; movaps        0x48dc(%rip),%xmm8        # 4cb0 <_sk_callback_sse2+0x174>
+  DB  68,15,40,5,226,72,0,0               ; movaps        0x48e2(%rip),%xmm8        # 4ce0 <_sk_callback_sse2+0x17a>
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -17931,7 +17965,7 @@ _sk_modulate_sse2 LABEL PROC
 
 PUBLIC _sk_multiply_sse2
 _sk_multiply_sse2 LABEL PROC
-  DB  68,15,40,5,176,72,0,0               ; movaps        0x48b0(%rip),%xmm8        # 4cc0 <_sk_callback_sse2+0x184>
+  DB  68,15,40,5,182,72,0,0               ; movaps        0x48b6(%rip),%xmm8        # 4cf0 <_sk_callback_sse2+0x18a>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
@@ -18001,7 +18035,7 @@ _sk_screen_sse2 LABEL PROC
 PUBLIC _sk_xor__sse2
 _sk_xor__sse2 LABEL PROC
   DB  68,15,40,195                        ; movaps        %xmm3,%xmm8
-  DB  15,40,29,225,71,0,0                 ; movaps        0x47e1(%rip),%xmm3        # 4cd0 <_sk_callback_sse2+0x194>
+  DB  15,40,29,231,71,0,0                 ; movaps        0x47e7(%rip),%xmm3        # 4d00 <_sk_callback_sse2+0x19a>
   DB  68,15,40,203                        ; movaps        %xmm3,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
@@ -18047,7 +18081,7 @@ _sk_darken_sse2 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,95,209                        ; maxps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,76,71,0,0                  ; movaps        0x474c(%rip),%xmm2        # 4ce0 <_sk_callback_sse2+0x1a4>
+  DB  15,40,21,82,71,0,0                  ; movaps        0x4752(%rip),%xmm2        # 4d10 <_sk_callback_sse2+0x1aa>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -18079,7 +18113,7 @@ _sk_lighten_sse2 LABEL PROC
   DB  68,15,89,206                        ; mulps         %xmm6,%xmm9
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,241,70,0,0                 ; movaps        0x46f1(%rip),%xmm2        # 4cf0 <_sk_callback_sse2+0x1b4>
+  DB  15,40,21,247,70,0,0                 ; movaps        0x46f7(%rip),%xmm2        # 4d20 <_sk_callback_sse2+0x1ba>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -18114,7 +18148,7 @@ _sk_difference_sse2 LABEL PROC
   DB  65,15,93,209                        ; minps         %xmm9,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,194                        ; subps         %xmm2,%xmm8
-  DB  15,40,21,139,70,0,0                 ; movaps        0x468b(%rip),%xmm2        # 4d00 <_sk_callback_sse2+0x1c4>
+  DB  15,40,21,145,70,0,0                 ; movaps        0x4691(%rip),%xmm2        # 4d30 <_sk_callback_sse2+0x1ca>
   DB  15,92,211                           ; subps         %xmm3,%xmm2
   DB  15,89,215                           ; mulps         %xmm7,%xmm2
   DB  15,88,218                           ; addps         %xmm2,%xmm3
@@ -18139,7 +18173,7 @@ _sk_exclusion_sse2 LABEL PROC
   DB  15,89,214                           ; mulps         %xmm6,%xmm2
   DB  15,88,210                           ; addps         %xmm2,%xmm2
   DB  68,15,92,202                        ; subps         %xmm2,%xmm9
-  DB  15,40,13,76,70,0,0                  ; movaps        0x464c(%rip),%xmm1        # 4d10 <_sk_callback_sse2+0x1d4>
+  DB  15,40,13,82,70,0,0                  ; movaps        0x4652(%rip),%xmm1        # 4d40 <_sk_callback_sse2+0x1da>
   DB  15,92,203                           ; subps         %xmm3,%xmm1
   DB  15,89,207                           ; mulps         %xmm7,%xmm1
   DB  15,88,217                           ; addps         %xmm1,%xmm3
@@ -18151,7 +18185,7 @@ _sk_exclusion_sse2 LABEL PROC
 PUBLIC _sk_colorburn_sse2
 _sk_colorburn_sse2 LABEL PROC
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
-  DB  68,15,40,21,59,70,0,0               ; movaps        0x463b(%rip),%xmm10        # 4d20 <_sk_callback_sse2+0x1e4>
+  DB  68,15,40,21,65,70,0,0               ; movaps        0x4641(%rip),%xmm10        # 4d50 <_sk_callback_sse2+0x1ea>
   DB  69,15,40,202                        ; movaps        %xmm10,%xmm9
   DB  68,15,92,207                        ; subps         %xmm7,%xmm9
   DB  69,15,40,217                        ; movaps        %xmm9,%xmm11
@@ -18243,7 +18277,7 @@ _sk_colorburn_sse2 LABEL PROC
 PUBLIC _sk_colordodge_sse2
 _sk_colordodge_sse2 LABEL PROC
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
-  DB  68,15,40,21,241,68,0,0              ; movaps        0x44f1(%rip),%xmm10        # 4d30 <_sk_callback_sse2+0x1f4>
+  DB  68,15,40,21,247,68,0,0              ; movaps        0x44f7(%rip),%xmm10        # 4d60 <_sk_callback_sse2+0x1fa>
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
   DB  68,15,92,223                        ; subps         %xmm7,%xmm11
   DB  69,15,40,227                        ; movaps        %xmm11,%xmm12
@@ -18336,7 +18370,7 @@ _sk_hardlight_sse2 LABEL PROC
   DB  15,41,52,36                         ; movaps        %xmm6,(%rsp)
   DB  15,40,245                           ; movaps        %xmm5,%xmm6
   DB  15,40,236                           ; movaps        %xmm4,%xmm5
-  DB  68,15,40,29,163,67,0,0              ; movaps        0x43a3(%rip),%xmm11        # 4d40 <_sk_callback_sse2+0x204>
+  DB  68,15,40,29,169,67,0,0              ; movaps        0x43a9(%rip),%xmm11        # 4d70 <_sk_callback_sse2+0x20a>
   DB  69,15,40,211                        ; movaps        %xmm11,%xmm10
   DB  68,15,92,215                        ; subps         %xmm7,%xmm10
   DB  69,15,40,194                        ; movaps        %xmm10,%xmm8
@@ -18423,7 +18457,7 @@ PUBLIC _sk_overlay_sse2
 _sk_overlay_sse2 LABEL PROC
   DB  68,15,40,193                        ; movaps        %xmm1,%xmm8
   DB  68,15,40,232                        ; movaps        %xmm0,%xmm13
-  DB  68,15,40,13,110,66,0,0              ; movaps        0x426e(%rip),%xmm9        # 4d50 <_sk_callback_sse2+0x214>
+  DB  68,15,40,13,116,66,0,0              ; movaps        0x4274(%rip),%xmm9        # 4d80 <_sk_callback_sse2+0x21a>
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
   DB  68,15,92,215                        ; subps         %xmm7,%xmm10
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
@@ -18513,7 +18547,7 @@ _sk_softlight_sse2 LABEL PROC
   DB  68,15,40,213                        ; movaps        %xmm5,%xmm10
   DB  68,15,94,215                        ; divps         %xmm7,%xmm10
   DB  69,15,84,212                        ; andps         %xmm12,%xmm10
-  DB  68,15,40,13,40,65,0,0               ; movaps        0x4128(%rip),%xmm9        # 4d60 <_sk_callback_sse2+0x224>
+  DB  68,15,40,13,46,65,0,0               ; movaps        0x412e(%rip),%xmm9        # 4d90 <_sk_callback_sse2+0x22a>
   DB  69,15,40,249                        ; movaps        %xmm9,%xmm15
   DB  69,15,92,250                        ; subps         %xmm10,%xmm15
   DB  69,15,40,218                        ; movaps        %xmm10,%xmm11
@@ -18526,10 +18560,10 @@ _sk_softlight_sse2 LABEL PROC
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  15,89,192                           ; mulps         %xmm0,%xmm0
   DB  65,15,88,194                        ; addps         %xmm10,%xmm0
-  DB  68,15,40,53,2,65,0,0                ; movaps        0x4102(%rip),%xmm14        # 4d70 <_sk_callback_sse2+0x234>
+  DB  68,15,40,53,8,65,0,0                ; movaps        0x4108(%rip),%xmm14        # 4da0 <_sk_callback_sse2+0x23a>
   DB  69,15,88,222                        ; addps         %xmm14,%xmm11
   DB  68,15,89,216                        ; mulps         %xmm0,%xmm11
-  DB  68,15,40,21,2,65,0,0                ; movaps        0x4102(%rip),%xmm10        # 4d80 <_sk_callback_sse2+0x244>
+  DB  68,15,40,21,8,65,0,0                ; movaps        0x4108(%rip),%xmm10        # 4db0 <_sk_callback_sse2+0x24a>
   DB  69,15,89,234                        ; mulps         %xmm10,%xmm13
   DB  69,15,88,235                        ; addps         %xmm11,%xmm13
   DB  15,88,228                           ; addps         %xmm4,%xmm4
@@ -18674,7 +18708,7 @@ _sk_hue_sse2 LABEL PROC
   DB  68,15,40,209                        ; movaps        %xmm1,%xmm10
   DB  68,15,40,225                        ; movaps        %xmm1,%xmm12
   DB  68,15,89,211                        ; mulps         %xmm3,%xmm10
-  DB  68,15,40,5,62,63,0,0                ; movaps        0x3f3e(%rip),%xmm8        # 4dc0 <_sk_callback_sse2+0x284>
+  DB  68,15,40,5,68,63,0,0                ; movaps        0x3f44(%rip),%xmm8        # 4df0 <_sk_callback_sse2+0x28a>
   DB  69,15,40,216                        ; movaps        %xmm8,%xmm11
   DB  15,40,207                           ; movaps        %xmm7,%xmm1
   DB  68,15,92,217                        ; subps         %xmm1,%xmm11
@@ -18720,12 +18754,12 @@ _sk_hue_sse2 LABEL PROC
   DB  69,15,84,206                        ; andps         %xmm14,%xmm9
   DB  69,15,84,214                        ; andps         %xmm14,%xmm10
   DB  65,15,84,214                        ; andps         %xmm14,%xmm2
-  DB  68,15,40,61,82,62,0,0               ; movaps        0x3e52(%rip),%xmm15        # 4d90 <_sk_callback_sse2+0x254>
+  DB  68,15,40,61,88,62,0,0               ; movaps        0x3e58(%rip),%xmm15        # 4dc0 <_sk_callback_sse2+0x25a>
   DB  65,15,89,231                        ; mulps         %xmm15,%xmm4
-  DB  15,40,5,87,62,0,0                   ; movaps        0x3e57(%rip),%xmm0        # 4da0 <_sk_callback_sse2+0x264>
+  DB  15,40,5,93,62,0,0                   ; movaps        0x3e5d(%rip),%xmm0        # 4dd0 <_sk_callback_sse2+0x26a>
   DB  15,89,240                           ; mulps         %xmm0,%xmm6
   DB  15,88,244                           ; addps         %xmm4,%xmm6
-  DB  68,15,40,53,89,62,0,0               ; movaps        0x3e59(%rip),%xmm14        # 4db0 <_sk_callback_sse2+0x274>
+  DB  68,15,40,53,95,62,0,0               ; movaps        0x3e5f(%rip),%xmm14        # 4de0 <_sk_callback_sse2+0x27a>
   DB  68,15,40,239                        ; movaps        %xmm7,%xmm13
   DB  69,15,89,238                        ; mulps         %xmm14,%xmm13
   DB  68,15,88,238                        ; addps         %xmm6,%xmm13
@@ -18902,14 +18936,14 @@ _sk_saturation_sse2 LABEL PROC
   DB  68,15,84,211                        ; andps         %xmm3,%xmm10
   DB  68,15,84,203                        ; andps         %xmm3,%xmm9
   DB  15,84,195                           ; andps         %xmm3,%xmm0
-  DB  68,15,40,5,233,59,0,0               ; movaps        0x3be9(%rip),%xmm8        # 4dd0 <_sk_callback_sse2+0x294>
+  DB  68,15,40,5,239,59,0,0               ; movaps        0x3bef(%rip),%xmm8        # 4e00 <_sk_callback_sse2+0x29a>
   DB  15,40,214                           ; movaps        %xmm6,%xmm2
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
-  DB  15,40,13,235,59,0,0                 ; movaps        0x3beb(%rip),%xmm1        # 4de0 <_sk_callback_sse2+0x2a4>
+  DB  15,40,13,241,59,0,0                 ; movaps        0x3bf1(%rip),%xmm1        # 4e10 <_sk_callback_sse2+0x2aa>
   DB  15,40,221                           ; movaps        %xmm5,%xmm3
   DB  15,89,217                           ; mulps         %xmm1,%xmm3
   DB  15,88,218                           ; addps         %xmm2,%xmm3
-  DB  68,15,40,37,234,59,0,0              ; movaps        0x3bea(%rip),%xmm12        # 4df0 <_sk_callback_sse2+0x2b4>
+  DB  68,15,40,37,240,59,0,0              ; movaps        0x3bf0(%rip),%xmm12        # 4e20 <_sk_callback_sse2+0x2ba>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
   DB  68,15,88,235                        ; addps         %xmm3,%xmm13
   DB  65,15,40,210                        ; movaps        %xmm10,%xmm2
@@ -18954,7 +18988,7 @@ _sk_saturation_sse2 LABEL PROC
   DB  15,40,223                           ; movaps        %xmm7,%xmm3
   DB  15,40,236                           ; movaps        %xmm4,%xmm5
   DB  15,89,221                           ; mulps         %xmm5,%xmm3
-  DB  68,15,40,5,79,59,0,0                ; movaps        0x3b4f(%rip),%xmm8        # 4e00 <_sk_callback_sse2+0x2c4>
+  DB  68,15,40,5,85,59,0,0                ; movaps        0x3b55(%rip),%xmm8        # 4e30 <_sk_callback_sse2+0x2ca>
   DB  65,15,40,224                        ; movaps        %xmm8,%xmm4
   DB  68,15,92,199                        ; subps         %xmm7,%xmm8
   DB  15,88,253                           ; addps         %xmm5,%xmm7
@@ -19055,14 +19089,14 @@ _sk_color_sse2 LABEL PROC
   DB  68,15,40,213                        ; movaps        %xmm5,%xmm10
   DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
   DB  65,15,40,208                        ; movaps        %xmm8,%xmm2
-  DB  68,15,40,45,231,57,0,0              ; movaps        0x39e7(%rip),%xmm13        # 4e10 <_sk_callback_sse2+0x2d4>
+  DB  68,15,40,45,237,57,0,0              ; movaps        0x39ed(%rip),%xmm13        # 4e40 <_sk_callback_sse2+0x2da>
   DB  68,15,40,198                        ; movaps        %xmm6,%xmm8
   DB  69,15,89,197                        ; mulps         %xmm13,%xmm8
-  DB  68,15,40,53,231,57,0,0              ; movaps        0x39e7(%rip),%xmm14        # 4e20 <_sk_callback_sse2+0x2e4>
+  DB  68,15,40,53,237,57,0,0              ; movaps        0x39ed(%rip),%xmm14        # 4e50 <_sk_callback_sse2+0x2ea>
   DB  65,15,40,195                        ; movaps        %xmm11,%xmm0
   DB  65,15,89,198                        ; mulps         %xmm14,%xmm0
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
-  DB  68,15,40,29,227,57,0,0              ; movaps        0x39e3(%rip),%xmm11        # 4e30 <_sk_callback_sse2+0x2f4>
+  DB  68,15,40,29,233,57,0,0              ; movaps        0x39e9(%rip),%xmm11        # 4e60 <_sk_callback_sse2+0x2fa>
   DB  69,15,89,227                        ; mulps         %xmm11,%xmm12
   DB  68,15,88,224                        ; addps         %xmm0,%xmm12
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
@@ -19070,7 +19104,7 @@ _sk_color_sse2 LABEL PROC
   DB  69,15,40,250                        ; movaps        %xmm10,%xmm15
   DB  69,15,89,254                        ; mulps         %xmm14,%xmm15
   DB  68,15,88,248                        ; addps         %xmm0,%xmm15
-  DB  68,15,40,5,207,57,0,0               ; movaps        0x39cf(%rip),%xmm8        # 4e40 <_sk_callback_sse2+0x304>
+  DB  68,15,40,5,213,57,0,0               ; movaps        0x39d5(%rip),%xmm8        # 4e70 <_sk_callback_sse2+0x30a>
   DB  65,15,40,224                        ; movaps        %xmm8,%xmm4
   DB  15,92,226                           ; subps         %xmm2,%xmm4
   DB  15,89,252                           ; mulps         %xmm4,%xmm7
@@ -19206,15 +19240,15 @@ _sk_luminosity_sse2 LABEL PROC
   DB  68,15,40,205                        ; movaps        %xmm5,%xmm9
   DB  68,15,89,204                        ; mulps         %xmm4,%xmm9
   DB  15,89,222                           ; mulps         %xmm6,%xmm3
-  DB  68,15,40,37,225,55,0,0              ; movaps        0x37e1(%rip),%xmm12        # 4e50 <_sk_callback_sse2+0x314>
+  DB  68,15,40,37,231,55,0,0              ; movaps        0x37e7(%rip),%xmm12        # 4e80 <_sk_callback_sse2+0x31a>
   DB  68,15,40,199                        ; movaps        %xmm7,%xmm8
   DB  69,15,89,196                        ; mulps         %xmm12,%xmm8
-  DB  68,15,40,45,225,55,0,0              ; movaps        0x37e1(%rip),%xmm13        # 4e60 <_sk_callback_sse2+0x324>
+  DB  68,15,40,45,231,55,0,0              ; movaps        0x37e7(%rip),%xmm13        # 4e90 <_sk_callback_sse2+0x32a>
   DB  68,15,40,241                        ; movaps        %xmm1,%xmm14
   DB  69,15,89,245                        ; mulps         %xmm13,%xmm14
   DB  69,15,88,240                        ; addps         %xmm8,%xmm14
-  DB  68,15,40,29,221,55,0,0              ; movaps        0x37dd(%rip),%xmm11        # 4e70 <_sk_callback_sse2+0x334>
-  DB  68,15,40,5,229,55,0,0               ; movaps        0x37e5(%rip),%xmm8        # 4e80 <_sk_callback_sse2+0x344>
+  DB  68,15,40,29,227,55,0,0              ; movaps        0x37e3(%rip),%xmm11        # 4ea0 <_sk_callback_sse2+0x33a>
+  DB  68,15,40,5,235,55,0,0               ; movaps        0x37eb(%rip),%xmm8        # 4eb0 <_sk_callback_sse2+0x34a>
   DB  69,15,40,248                        ; movaps        %xmm8,%xmm15
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  68,15,92,248                        ; subps         %xmm0,%xmm15
@@ -19356,7 +19390,7 @@ _sk_clamp_0_sse2 LABEL PROC
 
 PUBLIC _sk_clamp_1_sse2
 _sk_clamp_1_sse2 LABEL PROC
-  DB  68,15,40,5,236,53,0,0               ; movaps        0x35ec(%rip),%xmm8        # 4e90 <_sk_callback_sse2+0x354>
+  DB  68,15,40,5,242,53,0,0               ; movaps        0x35f2(%rip),%xmm8        # 4ec0 <_sk_callback_sse2+0x35a>
   DB  65,15,93,192                        ; minps         %xmm8,%xmm0
   DB  65,15,93,200                        ; minps         %xmm8,%xmm1
   DB  65,15,93,208                        ; minps         %xmm8,%xmm2
@@ -19366,7 +19400,7 @@ _sk_clamp_1_sse2 LABEL PROC
 
 PUBLIC _sk_clamp_a_sse2
 _sk_clamp_a_sse2 LABEL PROC
-  DB  15,93,29,225,53,0,0                 ; minps         0x35e1(%rip),%xmm3        # 4ea0 <_sk_callback_sse2+0x364>
+  DB  15,93,29,231,53,0,0                 ; minps         0x35e7(%rip),%xmm3        # 4ed0 <_sk_callback_sse2+0x36a>
   DB  15,93,195                           ; minps         %xmm3,%xmm0
   DB  15,93,203                           ; minps         %xmm3,%xmm1
   DB  15,93,211                           ; minps         %xmm3,%xmm2
@@ -19439,7 +19473,7 @@ _sk_premul_sse2 LABEL PROC
 PUBLIC _sk_unpremul_sse2
 _sk_unpremul_sse2 LABEL PROC
   DB  69,15,87,192                        ; xorps         %xmm8,%xmm8
-  DB  68,15,40,13,76,53,0,0               ; movaps        0x354c(%rip),%xmm9        # 4eb0 <_sk_callback_sse2+0x374>
+  DB  68,15,40,13,82,53,0,0               ; movaps        0x3552(%rip),%xmm9        # 4ee0 <_sk_callback_sse2+0x37a>
   DB  68,15,94,203                        ; divps         %xmm3,%xmm9
   DB  68,15,194,195,4                     ; cmpneqps      %xmm3,%xmm8
   DB  69,15,84,193                        ; andps         %xmm9,%xmm8
@@ -19451,20 +19485,20 @@ _sk_unpremul_sse2 LABEL PROC
 
 PUBLIC _sk_from_srgb_sse2
 _sk_from_srgb_sse2 LABEL PROC
-  DB  68,15,40,5,55,53,0,0                ; movaps        0x3537(%rip),%xmm8        # 4ec0 <_sk_callback_sse2+0x384>
+  DB  68,15,40,5,61,53,0,0                ; movaps        0x353d(%rip),%xmm8        # 4ef0 <_sk_callback_sse2+0x38a>
   DB  68,15,40,232                        ; movaps        %xmm0,%xmm13
   DB  69,15,89,232                        ; mulps         %xmm8,%xmm13
   DB  68,15,40,216                        ; movaps        %xmm0,%xmm11
   DB  69,15,89,219                        ; mulps         %xmm11,%xmm11
-  DB  68,15,40,13,47,53,0,0               ; movaps        0x352f(%rip),%xmm9        # 4ed0 <_sk_callback_sse2+0x394>
+  DB  68,15,40,13,53,53,0,0               ; movaps        0x3535(%rip),%xmm9        # 4f00 <_sk_callback_sse2+0x39a>
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
   DB  69,15,89,241                        ; mulps         %xmm9,%xmm14
-  DB  68,15,40,21,47,53,0,0               ; movaps        0x352f(%rip),%xmm10        # 4ee0 <_sk_callback_sse2+0x3a4>
+  DB  68,15,40,21,53,53,0,0               ; movaps        0x3535(%rip),%xmm10        # 4f10 <_sk_callback_sse2+0x3aa>
   DB  69,15,88,242                        ; addps         %xmm10,%xmm14
   DB  69,15,89,243                        ; mulps         %xmm11,%xmm14
-  DB  68,15,40,29,47,53,0,0               ; movaps        0x352f(%rip),%xmm11        # 4ef0 <_sk_callback_sse2+0x3b4>
+  DB  68,15,40,29,53,53,0,0               ; movaps        0x3535(%rip),%xmm11        # 4f20 <_sk_callback_sse2+0x3ba>
   DB  69,15,88,243                        ; addps         %xmm11,%xmm14
-  DB  68,15,40,37,51,53,0,0               ; movaps        0x3533(%rip),%xmm12        # 4f00 <_sk_callback_sse2+0x3c4>
+  DB  68,15,40,37,57,53,0,0               ; movaps        0x3539(%rip),%xmm12        # 4f30 <_sk_callback_sse2+0x3ca>
   DB  65,15,194,196,1                     ; cmpltps       %xmm12,%xmm0
   DB  68,15,84,232                        ; andps         %xmm0,%xmm13
   DB  65,15,85,198                        ; andnps        %xmm14,%xmm0
@@ -19501,20 +19535,20 @@ _sk_to_srgb_sse2 LABEL PROC
   DB  68,15,82,192                        ; rsqrtps       %xmm0,%xmm8
   DB  69,15,83,200                        ; rcpps         %xmm8,%xmm9
   DB  69,15,82,232                        ; rsqrtps       %xmm8,%xmm13
-  DB  68,15,40,5,184,52,0,0               ; movaps        0x34b8(%rip),%xmm8        # 4f10 <_sk_callback_sse2+0x3d4>
+  DB  68,15,40,5,190,52,0,0               ; movaps        0x34be(%rip),%xmm8        # 4f40 <_sk_callback_sse2+0x3da>
   DB  68,15,40,240                        ; movaps        %xmm0,%xmm14
   DB  69,15,89,240                        ; mulps         %xmm8,%xmm14
-  DB  68,15,40,21,184,52,0,0              ; movaps        0x34b8(%rip),%xmm10        # 4f20 <_sk_callback_sse2+0x3e4>
+  DB  68,15,40,21,190,52,0,0              ; movaps        0x34be(%rip),%xmm10        # 4f50 <_sk_callback_sse2+0x3ea>
   DB  69,15,89,202                        ; mulps         %xmm10,%xmm9
-  DB  68,15,40,29,188,52,0,0              ; movaps        0x34bc(%rip),%xmm11        # 4f30 <_sk_callback_sse2+0x3f4>
+  DB  68,15,40,29,194,52,0,0              ; movaps        0x34c2(%rip),%xmm11        # 4f60 <_sk_callback_sse2+0x3fa>
   DB  69,15,88,203                        ; addps         %xmm11,%xmm9
-  DB  68,15,40,37,192,52,0,0              ; movaps        0x34c0(%rip),%xmm12        # 4f40 <_sk_callback_sse2+0x404>
+  DB  68,15,40,37,198,52,0,0              ; movaps        0x34c6(%rip),%xmm12        # 4f70 <_sk_callback_sse2+0x40a>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,40,13,192,52,0,0              ; movaps        0x34c0(%rip),%xmm9        # 4f50 <_sk_callback_sse2+0x414>
+  DB  68,15,40,13,198,52,0,0              ; movaps        0x34c6(%rip),%xmm9        # 4f80 <_sk_callback_sse2+0x41a>
   DB  69,15,40,249                        ; movaps        %xmm9,%xmm15
   DB  69,15,93,253                        ; minps         %xmm13,%xmm15
-  DB  68,15,40,45,192,52,0,0              ; movaps        0x34c0(%rip),%xmm13        # 4f60 <_sk_callback_sse2+0x424>
+  DB  68,15,40,45,198,52,0,0              ; movaps        0x34c6(%rip),%xmm13        # 4f90 <_sk_callback_sse2+0x42a>
   DB  65,15,194,197,1                     ; cmpltps       %xmm13,%xmm0
   DB  68,15,84,240                        ; andps         %xmm0,%xmm14
   DB  65,15,85,199                        ; andnps        %xmm15,%xmm0
@@ -19562,7 +19596,7 @@ _sk_rgb_to_hsl_sse2 LABEL PROC
   DB  68,15,93,218                        ; minps         %xmm2,%xmm11
   DB  65,15,40,202                        ; movaps        %xmm10,%xmm1
   DB  65,15,92,203                        ; subps         %xmm11,%xmm1
-  DB  68,15,40,45,25,52,0,0               ; movaps        0x3419(%rip),%xmm13        # 4f70 <_sk_callback_sse2+0x434>
+  DB  68,15,40,45,31,52,0,0               ; movaps        0x341f(%rip),%xmm13        # 4fa0 <_sk_callback_sse2+0x43a>
   DB  68,15,94,233                        ; divps         %xmm1,%xmm13
   DB  65,15,40,194                        ; movaps        %xmm10,%xmm0
   DB  65,15,194,192,0                     ; cmpeqps       %xmm8,%xmm0
@@ -19571,30 +19605,30 @@ _sk_rgb_to_hsl_sse2 LABEL PROC
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,40,241                        ; movaps        %xmm9,%xmm14
   DB  68,15,194,242,1                     ; cmpltps       %xmm2,%xmm14
-  DB  68,15,84,53,255,51,0,0              ; andps         0x33ff(%rip),%xmm14        # 4f80 <_sk_callback_sse2+0x444>
+  DB  68,15,84,53,5,52,0,0                ; andps         0x3405(%rip),%xmm14        # 4fb0 <_sk_callback_sse2+0x44a>
   DB  69,15,88,244                        ; addps         %xmm12,%xmm14
   DB  69,15,40,250                        ; movaps        %xmm10,%xmm15
   DB  69,15,194,249,0                     ; cmpeqps       %xmm9,%xmm15
   DB  65,15,92,208                        ; subps         %xmm8,%xmm2
   DB  65,15,89,213                        ; mulps         %xmm13,%xmm2
-  DB  68,15,40,37,242,51,0,0              ; movaps        0x33f2(%rip),%xmm12        # 4f90 <_sk_callback_sse2+0x454>
+  DB  68,15,40,37,248,51,0,0              ; movaps        0x33f8(%rip),%xmm12        # 4fc0 <_sk_callback_sse2+0x45a>
   DB  65,15,88,212                        ; addps         %xmm12,%xmm2
   DB  69,15,92,193                        ; subps         %xmm9,%xmm8
   DB  69,15,89,197                        ; mulps         %xmm13,%xmm8
-  DB  68,15,88,5,238,51,0,0               ; addps         0x33ee(%rip),%xmm8        # 4fa0 <_sk_callback_sse2+0x464>
+  DB  68,15,88,5,244,51,0,0               ; addps         0x33f4(%rip),%xmm8        # 4fd0 <_sk_callback_sse2+0x46a>
   DB  65,15,84,215                        ; andps         %xmm15,%xmm2
   DB  69,15,85,248                        ; andnps        %xmm8,%xmm15
   DB  68,15,86,250                        ; orps          %xmm2,%xmm15
   DB  68,15,84,240                        ; andps         %xmm0,%xmm14
   DB  65,15,85,199                        ; andnps        %xmm15,%xmm0
   DB  65,15,86,198                        ; orps          %xmm14,%xmm0
-  DB  15,89,5,223,51,0,0                  ; mulps         0x33df(%rip),%xmm0        # 4fb0 <_sk_callback_sse2+0x474>
+  DB  15,89,5,229,51,0,0                  ; mulps         0x33e5(%rip),%xmm0        # 4fe0 <_sk_callback_sse2+0x47a>
   DB  69,15,40,194                        ; movaps        %xmm10,%xmm8
   DB  69,15,194,195,4                     ; cmpneqps      %xmm11,%xmm8
   DB  65,15,84,192                        ; andps         %xmm8,%xmm0
   DB  69,15,92,226                        ; subps         %xmm10,%xmm12
   DB  69,15,88,211                        ; addps         %xmm11,%xmm10
-  DB  68,15,40,13,210,51,0,0              ; movaps        0x33d2(%rip),%xmm9        # 4fc0 <_sk_callback_sse2+0x484>
+  DB  68,15,40,13,216,51,0,0              ; movaps        0x33d8(%rip),%xmm9        # 4ff0 <_sk_callback_sse2+0x48a>
   DB  65,15,40,210                        ; movaps        %xmm10,%xmm2
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
   DB  68,15,194,202,1                     ; cmpltps       %xmm2,%xmm9
@@ -19617,7 +19651,7 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  15,41,92,36,32                      ; movaps        %xmm3,0x20(%rsp)
   DB  68,15,40,218                        ; movaps        %xmm2,%xmm11
   DB  15,40,240                           ; movaps        %xmm0,%xmm6
-  DB  68,15,40,13,141,51,0,0              ; movaps        0x338d(%rip),%xmm9        # 4fd0 <_sk_callback_sse2+0x494>
+  DB  68,15,40,13,147,51,0,0              ; movaps        0x3393(%rip),%xmm9        # 5000 <_sk_callback_sse2+0x49a>
   DB  69,15,40,209                        ; movaps        %xmm9,%xmm10
   DB  69,15,194,211,2                     ; cmpleps       %xmm11,%xmm10
   DB  15,40,193                           ; movaps        %xmm1,%xmm0
@@ -19634,28 +19668,28 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  69,15,88,211                        ; addps         %xmm11,%xmm10
   DB  69,15,88,219                        ; addps         %xmm11,%xmm11
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
-  DB  15,40,5,87,51,0,0                   ; movaps        0x3357(%rip),%xmm0        # 4fe0 <_sk_callback_sse2+0x4a4>
+  DB  15,40,5,93,51,0,0                   ; movaps        0x335d(%rip),%xmm0        # 5010 <_sk_callback_sse2+0x4aa>
   DB  15,88,198                           ; addps         %xmm6,%xmm0
   DB  243,15,91,200                       ; cvttps2dq     %xmm0,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
   DB  15,40,216                           ; movaps        %xmm0,%xmm3
   DB  15,194,217,1                        ; cmpltps       %xmm1,%xmm3
-  DB  15,84,29,79,51,0,0                  ; andps         0x334f(%rip),%xmm3        # 4ff0 <_sk_callback_sse2+0x4b4>
+  DB  15,84,29,85,51,0,0                  ; andps         0x3355(%rip),%xmm3        # 5020 <_sk_callback_sse2+0x4ba>
   DB  15,92,203                           ; subps         %xmm3,%xmm1
   DB  15,92,193                           ; subps         %xmm1,%xmm0
-  DB  68,15,40,45,81,51,0,0               ; movaps        0x3351(%rip),%xmm13        # 5000 <_sk_callback_sse2+0x4c4>
+  DB  68,15,40,45,87,51,0,0               ; movaps        0x3357(%rip),%xmm13        # 5030 <_sk_callback_sse2+0x4ca>
   DB  69,15,40,197                        ; movaps        %xmm13,%xmm8
   DB  68,15,194,192,2                     ; cmpleps       %xmm0,%xmm8
   DB  69,15,40,242                        ; movaps        %xmm10,%xmm14
   DB  69,15,92,243                        ; subps         %xmm11,%xmm14
   DB  65,15,40,217                        ; movaps        %xmm9,%xmm3
   DB  15,194,216,2                        ; cmpleps       %xmm0,%xmm3
-  DB  15,40,21,97,51,0,0                  ; movaps        0x3361(%rip),%xmm2        # 5030 <_sk_callback_sse2+0x4f4>
+  DB  15,40,21,103,51,0,0                 ; movaps        0x3367(%rip),%xmm2        # 5060 <_sk_callback_sse2+0x4fa>
   DB  68,15,40,250                        ; movaps        %xmm2,%xmm15
   DB  68,15,194,248,2                     ; cmpleps       %xmm0,%xmm15
-  DB  15,40,13,49,51,0,0                  ; movaps        0x3331(%rip),%xmm1        # 5010 <_sk_callback_sse2+0x4d4>
+  DB  15,40,13,55,51,0,0                  ; movaps        0x3337(%rip),%xmm1        # 5040 <_sk_callback_sse2+0x4da>
   DB  15,89,193                           ; mulps         %xmm1,%xmm0
-  DB  15,40,45,55,51,0,0                  ; movaps        0x3337(%rip),%xmm5        # 5020 <_sk_callback_sse2+0x4e4>
+  DB  15,40,45,61,51,0,0                  ; movaps        0x333d(%rip),%xmm5        # 5050 <_sk_callback_sse2+0x4ea>
   DB  15,40,229                           ; movaps        %xmm5,%xmm4
   DB  15,92,224                           ; subps         %xmm0,%xmm4
   DB  65,15,89,230                        ; mulps         %xmm14,%xmm4
@@ -19678,7 +19712,7 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
   DB  15,40,222                           ; movaps        %xmm6,%xmm3
   DB  15,194,216,1                        ; cmpltps       %xmm0,%xmm3
-  DB  15,84,29,172,50,0,0                 ; andps         0x32ac(%rip),%xmm3        # 4ff0 <_sk_callback_sse2+0x4b4>
+  DB  15,84,29,178,50,0,0                 ; andps         0x32b2(%rip),%xmm3        # 5020 <_sk_callback_sse2+0x4ba>
   DB  15,92,195                           ; subps         %xmm3,%xmm0
   DB  68,15,40,230                        ; movaps        %xmm6,%xmm12
   DB  68,15,92,224                        ; subps         %xmm0,%xmm12
@@ -19708,12 +19742,12 @@ _sk_hsl_to_rgb_sse2 LABEL PROC
   DB  15,40,60,36                         ; movaps        (%rsp),%xmm7
   DB  15,40,231                           ; movaps        %xmm7,%xmm4
   DB  15,85,227                           ; andnps        %xmm3,%xmm4
-  DB  15,88,53,133,50,0,0                 ; addps         0x3285(%rip),%xmm6        # 5040 <_sk_callback_sse2+0x504>
+  DB  15,88,53,139,50,0,0                 ; addps         0x328b(%rip),%xmm6        # 5070 <_sk_callback_sse2+0x50a>
   DB  243,15,91,198                       ; cvttps2dq     %xmm6,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
   DB  15,40,222                           ; movaps        %xmm6,%xmm3
   DB  15,194,216,1                        ; cmpltps       %xmm0,%xmm3
-  DB  15,84,29,32,50,0,0                  ; andps         0x3220(%rip),%xmm3        # 4ff0 <_sk_callback_sse2+0x4b4>
+  DB  15,84,29,38,50,0,0                  ; andps         0x3226(%rip),%xmm3        # 5020 <_sk_callback_sse2+0x4ba>
   DB  15,92,195                           ; subps         %xmm3,%xmm0
   DB  15,92,240                           ; subps         %xmm0,%xmm6
   DB  15,89,206                           ; mulps         %xmm6,%xmm1
@@ -19774,7 +19808,7 @@ _sk_scale_u8_sse2 LABEL PROC
   DB  102,69,15,96,193                    ; punpcklbw     %xmm9,%xmm8
   DB  102,69,15,97,193                    ; punpcklwd     %xmm9,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,170,49,0,0               ; mulps         0x31aa(%rip),%xmm8        # 5050 <_sk_callback_sse2+0x514>
+  DB  68,15,89,5,176,49,0,0               ; mulps         0x31b0(%rip),%xmm8        # 5080 <_sk_callback_sse2+0x51a>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
@@ -19811,7 +19845,7 @@ _sk_lerp_u8_sse2 LABEL PROC
   DB  102,69,15,96,193                    ; punpcklbw     %xmm9,%xmm8
   DB  102,69,15,97,193                    ; punpcklwd     %xmm9,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,72,49,0,0                ; mulps         0x3148(%rip),%xmm8        # 5060 <_sk_callback_sse2+0x524>
+  DB  68,15,89,5,78,49,0,0                ; mulps         0x314e(%rip),%xmm8        # 5090 <_sk_callback_sse2+0x52a>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -19834,17 +19868,17 @@ _sk_lerp_565_sse2 LABEL PROC
   DB  243,68,15,126,20,120                ; movq          (%rax,%rdi,2),%xmm10
   DB  102,69,15,239,192                   ; pxor          %xmm8,%xmm8
   DB  102,69,15,97,208                    ; punpcklwd     %xmm8,%xmm10
-  DB  102,68,15,111,5,14,49,0,0           ; movdqa        0x310e(%rip),%xmm8        # 5070 <_sk_callback_sse2+0x534>
+  DB  102,68,15,111,5,20,49,0,0           ; movdqa        0x3114(%rip),%xmm8        # 50a0 <_sk_callback_sse2+0x53a>
   DB  102,69,15,219,194                   ; pand          %xmm10,%xmm8
   DB  69,15,91,192                        ; cvtdq2ps      %xmm8,%xmm8
-  DB  68,15,89,5,13,49,0,0                ; mulps         0x310d(%rip),%xmm8        # 5080 <_sk_callback_sse2+0x544>
-  DB  102,68,15,111,13,20,49,0,0          ; movdqa        0x3114(%rip),%xmm9        # 5090 <_sk_callback_sse2+0x554>
+  DB  68,15,89,5,19,49,0,0                ; mulps         0x3113(%rip),%xmm8        # 50b0 <_sk_callback_sse2+0x54a>
+  DB  102,68,15,111,13,26,49,0,0          ; movdqa        0x311a(%rip),%xmm9        # 50c0 <_sk_callback_sse2+0x55a>
   DB  102,69,15,219,202                   ; pand          %xmm10,%xmm9
   DB  69,15,91,201                        ; cvtdq2ps      %xmm9,%xmm9
-  DB  68,15,89,13,19,49,0,0               ; mulps         0x3113(%rip),%xmm9        # 50a0 <_sk_callback_sse2+0x564>
-  DB  102,68,15,219,21,26,49,0,0          ; pand          0x311a(%rip),%xmm10        # 50b0 <_sk_callback_sse2+0x574>
+  DB  68,15,89,13,25,49,0,0               ; mulps         0x3119(%rip),%xmm9        # 50d0 <_sk_callback_sse2+0x56a>
+  DB  102,68,15,219,21,32,49,0,0          ; pand          0x3120(%rip),%xmm10        # 50e0 <_sk_callback_sse2+0x57a>
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
-  DB  68,15,89,21,30,49,0,0               ; mulps         0x311e(%rip),%xmm10        # 50c0 <_sk_callback_sse2+0x584>
+  DB  68,15,89,21,36,49,0,0               ; mulps         0x3124(%rip),%xmm10        # 50f0 <_sk_callback_sse2+0x58a>
   DB  15,92,196                           ; subps         %xmm4,%xmm0
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  15,88,196                           ; addps         %xmm4,%xmm0
@@ -19873,7 +19907,7 @@ _sk_load_tables_sse2 LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  76,139,72,8                         ; mov           0x8(%rax),%r9
   DB  243,69,15,111,12,184                ; movdqu        (%r8,%rdi,4),%xmm9
-  DB  102,68,15,111,5,206,48,0,0          ; movdqa        0x30ce(%rip),%xmm8        # 50d0 <_sk_callback_sse2+0x594>
+  DB  102,68,15,111,5,212,48,0,0          ; movdqa        0x30d4(%rip),%xmm8        # 5100 <_sk_callback_sse2+0x59a>
   DB  102,65,15,111,193                   ; movdqa        %xmm9,%xmm0
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,15,112,200,78                   ; pshufd        $0x4e,%xmm0,%xmm1
@@ -19928,7 +19962,7 @@ _sk_load_tables_sse2 LABEL PROC
   DB  65,15,20,208                        ; unpcklps      %xmm8,%xmm2
   DB  102,65,15,114,209,24                ; psrld         $0x18,%xmm9
   DB  65,15,91,217                        ; cvtdq2ps      %xmm9,%xmm3
-  DB  15,89,29,219,47,0,0                 ; mulps         0x2fdb(%rip),%xmm3        # 50e0 <_sk_callback_sse2+0x5a4>
+  DB  15,89,29,225,47,0,0                 ; mulps         0x2fe1(%rip),%xmm3        # 5110 <_sk_callback_sse2+0x5aa>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -19945,7 +19979,7 @@ _sk_load_tables_u16_be_sse2 LABEL PROC
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,97,200                       ; punpcklwd     %xmm0,%xmm1
   DB  102,68,15,105,200                   ; punpckhwd     %xmm0,%xmm9
-  DB  102,68,15,111,21,174,47,0,0         ; movdqa        0x2fae(%rip),%xmm10        # 50f0 <_sk_callback_sse2+0x5b4>
+  DB  102,68,15,111,21,180,47,0,0         ; movdqa        0x2fb4(%rip),%xmm10        # 5120 <_sk_callback_sse2+0x5ba>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,194                   ; pand          %xmm10,%xmm0
   DB  102,69,15,239,192                   ; pxor          %xmm8,%xmm8
@@ -20006,7 +20040,7 @@ _sk_load_tables_u16_be_sse2 LABEL PROC
   DB  102,65,15,235,217                   ; por           %xmm9,%xmm3
   DB  102,65,15,97,216                    ; punpcklwd     %xmm8,%xmm3
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,157,46,0,0                 ; mulps         0x2e9d(%rip),%xmm3        # 5100 <_sk_callback_sse2+0x5c4>
+  DB  15,89,29,163,46,0,0                 ; mulps         0x2ea3(%rip),%xmm3        # 5130 <_sk_callback_sse2+0x5ca>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -20026,7 +20060,7 @@ _sk_load_tables_rgb_u16_be_sse2 LABEL PROC
   DB  102,68,15,97,208                    ; punpcklwd     %xmm0,%xmm10
   DB  102,65,15,111,195                   ; movdqa        %xmm11,%xmm0
   DB  102,65,15,97,194                    ; punpcklwd     %xmm10,%xmm0
-  DB  102,68,15,111,5,93,46,0,0           ; movdqa        0x2e5d(%rip),%xmm8        # 5110 <_sk_callback_sse2+0x5d4>
+  DB  102,68,15,111,5,99,46,0,0           ; movdqa        0x2e63(%rip),%xmm8        # 5140 <_sk_callback_sse2+0x5da>
   DB  102,15,112,200,78                   ; pshufd        $0x4e,%xmm0,%xmm1
   DB  102,65,15,219,192                   ; pand          %xmm8,%xmm0
   DB  102,69,15,239,201                   ; pxor          %xmm9,%xmm9
@@ -20081,7 +20115,7 @@ _sk_load_tables_rgb_u16_be_sse2 LABEL PROC
   DB  15,20,211                           ; unpcklps      %xmm3,%xmm2
   DB  65,15,20,208                        ; unpcklps      %xmm8,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,108,45,0,0                 ; movaps        0x2d6c(%rip),%xmm3        # 5120 <_sk_callback_sse2+0x5e4>
+  DB  15,40,29,114,45,0,0                 ; movaps        0x2d72(%rip),%xmm3        # 5150 <_sk_callback_sse2+0x5ea>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_byte_tables_sse2
@@ -20089,7 +20123,7 @@ _sk_byte_tables_sse2 LABEL PROC
   DB  65,86                               ; push          %r14
   DB  83                                  ; push          %rbx
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,109,45,0,0               ; movaps        0x2d6d(%rip),%xmm8        # 5130 <_sk_callback_sse2+0x5f4>
+  DB  68,15,40,5,115,45,0,0               ; movaps        0x2d73(%rip),%xmm8        # 5160 <_sk_callback_sse2+0x5fa>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,91,192                       ; cvtps2dq      %xmm0,%xmm0
   DB  102,72,15,126,193                   ; movq          %xmm0,%rcx
@@ -20116,7 +20150,7 @@ _sk_byte_tables_sse2 LABEL PROC
   DB  102,65,15,96,193                    ; punpcklbw     %xmm9,%xmm0
   DB  102,65,15,97,193                    ; punpcklwd     %xmm9,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,21,10,45,0,0               ; movaps        0x2d0a(%rip),%xmm10        # 5140 <_sk_callback_sse2+0x604>
+  DB  68,15,40,21,16,45,0,0               ; movaps        0x2d10(%rip),%xmm10        # 5170 <_sk_callback_sse2+0x60a>
   DB  65,15,89,194                        ; mulps         %xmm10,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -20230,7 +20264,7 @@ _sk_byte_tables_rgb_sse2 LABEL PROC
   DB  102,65,15,96,193                    ; punpcklbw     %xmm9,%xmm0
   DB  102,65,15,97,193                    ; punpcklwd     %xmm9,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,21,93,43,0,0               ; movaps        0x2b5d(%rip),%xmm10        # 5150 <_sk_callback_sse2+0x614>
+  DB  68,15,40,21,99,43,0,0               ; movaps        0x2b63(%rip),%xmm10        # 5180 <_sk_callback_sse2+0x61a>
   DB  65,15,89,194                        ; mulps         %xmm10,%xmm0
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
   DB  102,15,91,201                       ; cvtps2dq      %xmm1,%xmm1
@@ -20417,15 +20451,15 @@ _sk_parametric_r_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,156,40,0,0              ; mulps         0x289c(%rip),%xmm9        # 5160 <_sk_callback_sse2+0x624>
-  DB  68,15,84,21,164,40,0,0              ; andps         0x28a4(%rip),%xmm10        # 5170 <_sk_callback_sse2+0x634>
-  DB  68,15,86,21,172,40,0,0              ; orps          0x28ac(%rip),%xmm10        # 5180 <_sk_callback_sse2+0x644>
-  DB  68,15,88,13,180,40,0,0              ; addps         0x28b4(%rip),%xmm9        # 5190 <_sk_callback_sse2+0x654>
-  DB  68,15,40,37,188,40,0,0              ; movaps        0x28bc(%rip),%xmm12        # 51a0 <_sk_callback_sse2+0x664>
+  DB  68,15,89,13,162,40,0,0              ; mulps         0x28a2(%rip),%xmm9        # 5190 <_sk_callback_sse2+0x62a>
+  DB  68,15,84,21,170,40,0,0              ; andps         0x28aa(%rip),%xmm10        # 51a0 <_sk_callback_sse2+0x63a>
+  DB  68,15,86,21,178,40,0,0              ; orps          0x28b2(%rip),%xmm10        # 51b0 <_sk_callback_sse2+0x64a>
+  DB  68,15,88,13,186,40,0,0              ; addps         0x28ba(%rip),%xmm9        # 51c0 <_sk_callback_sse2+0x65a>
+  DB  68,15,40,37,194,40,0,0              ; movaps        0x28c2(%rip),%xmm12        # 51d0 <_sk_callback_sse2+0x66a>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,188,40,0,0              ; addps         0x28bc(%rip),%xmm10        # 51b0 <_sk_callback_sse2+0x674>
-  DB  68,15,40,37,196,40,0,0              ; movaps        0x28c4(%rip),%xmm12        # 51c0 <_sk_callback_sse2+0x684>
+  DB  68,15,88,21,194,40,0,0              ; addps         0x28c2(%rip),%xmm10        # 51e0 <_sk_callback_sse2+0x67a>
+  DB  68,15,40,37,202,40,0,0              ; movaps        0x28ca(%rip),%xmm12        # 51f0 <_sk_callback_sse2+0x68a>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -20433,22 +20467,22 @@ _sk_parametric_r_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,174,40,0,0              ; movaps        0x28ae(%rip),%xmm10        # 51d0 <_sk_callback_sse2+0x694>
+  DB  68,15,40,21,180,40,0,0              ; movaps        0x28b4(%rip),%xmm10        # 5200 <_sk_callback_sse2+0x69a>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,162,40,0,0              ; addps         0x28a2(%rip),%xmm9        # 51e0 <_sk_callback_sse2+0x6a4>
-  DB  68,15,40,37,170,40,0,0              ; movaps        0x28aa(%rip),%xmm12        # 51f0 <_sk_callback_sse2+0x6b4>
+  DB  68,15,88,13,168,40,0,0              ; addps         0x28a8(%rip),%xmm9        # 5210 <_sk_callback_sse2+0x6aa>
+  DB  68,15,40,37,176,40,0,0              ; movaps        0x28b0(%rip),%xmm12        # 5220 <_sk_callback_sse2+0x6ba>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,170,40,0,0              ; movaps        0x28aa(%rip),%xmm12        # 5200 <_sk_callback_sse2+0x6c4>
+  DB  68,15,40,37,176,40,0,0              ; movaps        0x28b0(%rip),%xmm12        # 5230 <_sk_callback_sse2+0x6ca>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,174,40,0,0              ; movaps        0x28ae(%rip),%xmm13        # 5210 <_sk_callback_sse2+0x6d4>
+  DB  68,15,40,45,180,40,0,0              ; movaps        0x28b4(%rip),%xmm13        # 5240 <_sk_callback_sse2+0x6da>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,174,40,0,0              ; mulps         0x28ae(%rip),%xmm13        # 5220 <_sk_callback_sse2+0x6e4>
+  DB  68,15,89,45,180,40,0,0              ; mulps         0x28b4(%rip),%xmm13        # 5250 <_sk_callback_sse2+0x6ea>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -20482,15 +20516,15 @@ _sk_parametric_g_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,46,40,0,0               ; mulps         0x282e(%rip),%xmm9        # 5230 <_sk_callback_sse2+0x6f4>
-  DB  68,15,84,21,54,40,0,0               ; andps         0x2836(%rip),%xmm10        # 5240 <_sk_callback_sse2+0x704>
-  DB  68,15,86,21,62,40,0,0               ; orps          0x283e(%rip),%xmm10        # 5250 <_sk_callback_sse2+0x714>
-  DB  68,15,88,13,70,40,0,0               ; addps         0x2846(%rip),%xmm9        # 5260 <_sk_callback_sse2+0x724>
-  DB  68,15,40,37,78,40,0,0               ; movaps        0x284e(%rip),%xmm12        # 5270 <_sk_callback_sse2+0x734>
+  DB  68,15,89,13,52,40,0,0               ; mulps         0x2834(%rip),%xmm9        # 5260 <_sk_callback_sse2+0x6fa>
+  DB  68,15,84,21,60,40,0,0               ; andps         0x283c(%rip),%xmm10        # 5270 <_sk_callback_sse2+0x70a>
+  DB  68,15,86,21,68,40,0,0               ; orps          0x2844(%rip),%xmm10        # 5280 <_sk_callback_sse2+0x71a>
+  DB  68,15,88,13,76,40,0,0               ; addps         0x284c(%rip),%xmm9        # 5290 <_sk_callback_sse2+0x72a>
+  DB  68,15,40,37,84,40,0,0               ; movaps        0x2854(%rip),%xmm12        # 52a0 <_sk_callback_sse2+0x73a>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,78,40,0,0               ; addps         0x284e(%rip),%xmm10        # 5280 <_sk_callback_sse2+0x744>
-  DB  68,15,40,37,86,40,0,0               ; movaps        0x2856(%rip),%xmm12        # 5290 <_sk_callback_sse2+0x754>
+  DB  68,15,88,21,84,40,0,0               ; addps         0x2854(%rip),%xmm10        # 52b0 <_sk_callback_sse2+0x74a>
+  DB  68,15,40,37,92,40,0,0               ; movaps        0x285c(%rip),%xmm12        # 52c0 <_sk_callback_sse2+0x75a>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -20498,22 +20532,22 @@ _sk_parametric_g_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,64,40,0,0               ; movaps        0x2840(%rip),%xmm10        # 52a0 <_sk_callback_sse2+0x764>
+  DB  68,15,40,21,70,40,0,0               ; movaps        0x2846(%rip),%xmm10        # 52d0 <_sk_callback_sse2+0x76a>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,52,40,0,0               ; addps         0x2834(%rip),%xmm9        # 52b0 <_sk_callback_sse2+0x774>
-  DB  68,15,40,37,60,40,0,0               ; movaps        0x283c(%rip),%xmm12        # 52c0 <_sk_callback_sse2+0x784>
+  DB  68,15,88,13,58,40,0,0               ; addps         0x283a(%rip),%xmm9        # 52e0 <_sk_callback_sse2+0x77a>
+  DB  68,15,40,37,66,40,0,0               ; movaps        0x2842(%rip),%xmm12        # 52f0 <_sk_callback_sse2+0x78a>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,60,40,0,0               ; movaps        0x283c(%rip),%xmm12        # 52d0 <_sk_callback_sse2+0x794>
+  DB  68,15,40,37,66,40,0,0               ; movaps        0x2842(%rip),%xmm12        # 5300 <_sk_callback_sse2+0x79a>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,64,40,0,0               ; movaps        0x2840(%rip),%xmm13        # 52e0 <_sk_callback_sse2+0x7a4>
+  DB  68,15,40,45,70,40,0,0               ; movaps        0x2846(%rip),%xmm13        # 5310 <_sk_callback_sse2+0x7aa>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,64,40,0,0               ; mulps         0x2840(%rip),%xmm13        # 52f0 <_sk_callback_sse2+0x7b4>
+  DB  68,15,89,45,70,40,0,0               ; mulps         0x2846(%rip),%xmm13        # 5320 <_sk_callback_sse2+0x7ba>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -20547,15 +20581,15 @@ _sk_parametric_b_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,192,39,0,0              ; mulps         0x27c0(%rip),%xmm9        # 5300 <_sk_callback_sse2+0x7c4>
-  DB  68,15,84,21,200,39,0,0              ; andps         0x27c8(%rip),%xmm10        # 5310 <_sk_callback_sse2+0x7d4>
-  DB  68,15,86,21,208,39,0,0              ; orps          0x27d0(%rip),%xmm10        # 5320 <_sk_callback_sse2+0x7e4>
-  DB  68,15,88,13,216,39,0,0              ; addps         0x27d8(%rip),%xmm9        # 5330 <_sk_callback_sse2+0x7f4>
-  DB  68,15,40,37,224,39,0,0              ; movaps        0x27e0(%rip),%xmm12        # 5340 <_sk_callback_sse2+0x804>
+  DB  68,15,89,13,198,39,0,0              ; mulps         0x27c6(%rip),%xmm9        # 5330 <_sk_callback_sse2+0x7ca>
+  DB  68,15,84,21,206,39,0,0              ; andps         0x27ce(%rip),%xmm10        # 5340 <_sk_callback_sse2+0x7da>
+  DB  68,15,86,21,214,39,0,0              ; orps          0x27d6(%rip),%xmm10        # 5350 <_sk_callback_sse2+0x7ea>
+  DB  68,15,88,13,222,39,0,0              ; addps         0x27de(%rip),%xmm9        # 5360 <_sk_callback_sse2+0x7fa>
+  DB  68,15,40,37,230,39,0,0              ; movaps        0x27e6(%rip),%xmm12        # 5370 <_sk_callback_sse2+0x80a>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,224,39,0,0              ; addps         0x27e0(%rip),%xmm10        # 5350 <_sk_callback_sse2+0x814>
-  DB  68,15,40,37,232,39,0,0              ; movaps        0x27e8(%rip),%xmm12        # 5360 <_sk_callback_sse2+0x824>
+  DB  68,15,88,21,230,39,0,0              ; addps         0x27e6(%rip),%xmm10        # 5380 <_sk_callback_sse2+0x81a>
+  DB  68,15,40,37,238,39,0,0              ; movaps        0x27ee(%rip),%xmm12        # 5390 <_sk_callback_sse2+0x82a>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -20563,22 +20597,22 @@ _sk_parametric_b_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,210,39,0,0              ; movaps        0x27d2(%rip),%xmm10        # 5370 <_sk_callback_sse2+0x834>
+  DB  68,15,40,21,216,39,0,0              ; movaps        0x27d8(%rip),%xmm10        # 53a0 <_sk_callback_sse2+0x83a>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,198,39,0,0              ; addps         0x27c6(%rip),%xmm9        # 5380 <_sk_callback_sse2+0x844>
-  DB  68,15,40,37,206,39,0,0              ; movaps        0x27ce(%rip),%xmm12        # 5390 <_sk_callback_sse2+0x854>
+  DB  68,15,88,13,204,39,0,0              ; addps         0x27cc(%rip),%xmm9        # 53b0 <_sk_callback_sse2+0x84a>
+  DB  68,15,40,37,212,39,0,0              ; movaps        0x27d4(%rip),%xmm12        # 53c0 <_sk_callback_sse2+0x85a>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,206,39,0,0              ; movaps        0x27ce(%rip),%xmm12        # 53a0 <_sk_callback_sse2+0x864>
+  DB  68,15,40,37,212,39,0,0              ; movaps        0x27d4(%rip),%xmm12        # 53d0 <_sk_callback_sse2+0x86a>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,210,39,0,0              ; movaps        0x27d2(%rip),%xmm13        # 53b0 <_sk_callback_sse2+0x874>
+  DB  68,15,40,45,216,39,0,0              ; movaps        0x27d8(%rip),%xmm13        # 53e0 <_sk_callback_sse2+0x87a>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,210,39,0,0              ; mulps         0x27d2(%rip),%xmm13        # 53c0 <_sk_callback_sse2+0x884>
+  DB  68,15,89,45,216,39,0,0              ; mulps         0x27d8(%rip),%xmm13        # 53f0 <_sk_callback_sse2+0x88a>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -20612,15 +20646,15 @@ _sk_parametric_a_sse2 LABEL PROC
   DB  69,15,88,209                        ; addps         %xmm9,%xmm10
   DB  69,15,198,219,0                     ; shufps        $0x0,%xmm11,%xmm11
   DB  69,15,91,202                        ; cvtdq2ps      %xmm10,%xmm9
-  DB  68,15,89,13,82,39,0,0               ; mulps         0x2752(%rip),%xmm9        # 53d0 <_sk_callback_sse2+0x894>
-  DB  68,15,84,21,90,39,0,0               ; andps         0x275a(%rip),%xmm10        # 53e0 <_sk_callback_sse2+0x8a4>
-  DB  68,15,86,21,98,39,0,0               ; orps          0x2762(%rip),%xmm10        # 53f0 <_sk_callback_sse2+0x8b4>
-  DB  68,15,88,13,106,39,0,0              ; addps         0x276a(%rip),%xmm9        # 5400 <_sk_callback_sse2+0x8c4>
-  DB  68,15,40,37,114,39,0,0              ; movaps        0x2772(%rip),%xmm12        # 5410 <_sk_callback_sse2+0x8d4>
+  DB  68,15,89,13,88,39,0,0               ; mulps         0x2758(%rip),%xmm9        # 5400 <_sk_callback_sse2+0x89a>
+  DB  68,15,84,21,96,39,0,0               ; andps         0x2760(%rip),%xmm10        # 5410 <_sk_callback_sse2+0x8aa>
+  DB  68,15,86,21,104,39,0,0              ; orps          0x2768(%rip),%xmm10        # 5420 <_sk_callback_sse2+0x8ba>
+  DB  68,15,88,13,112,39,0,0              ; addps         0x2770(%rip),%xmm9        # 5430 <_sk_callback_sse2+0x8ca>
+  DB  68,15,40,37,120,39,0,0              ; movaps        0x2778(%rip),%xmm12        # 5440 <_sk_callback_sse2+0x8da>
   DB  69,15,89,226                        ; mulps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,88,21,114,39,0,0              ; addps         0x2772(%rip),%xmm10        # 5420 <_sk_callback_sse2+0x8e4>
-  DB  68,15,40,37,122,39,0,0              ; movaps        0x277a(%rip),%xmm12        # 5430 <_sk_callback_sse2+0x8f4>
+  DB  68,15,88,21,120,39,0,0              ; addps         0x2778(%rip),%xmm10        # 5450 <_sk_callback_sse2+0x8ea>
+  DB  68,15,40,37,128,39,0,0              ; movaps        0x2780(%rip),%xmm12        # 5460 <_sk_callback_sse2+0x8fa>
   DB  69,15,94,226                        ; divps         %xmm10,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
   DB  69,15,89,203                        ; mulps         %xmm11,%xmm9
@@ -20628,22 +20662,22 @@ _sk_parametric_a_sse2 LABEL PROC
   DB  69,15,91,226                        ; cvtdq2ps      %xmm10,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,194,236,1                     ; cmpltps       %xmm12,%xmm13
-  DB  68,15,40,21,100,39,0,0              ; movaps        0x2764(%rip),%xmm10        # 5440 <_sk_callback_sse2+0x904>
+  DB  68,15,40,21,106,39,0,0              ; movaps        0x276a(%rip),%xmm10        # 5470 <_sk_callback_sse2+0x90a>
   DB  69,15,84,234                        ; andps         %xmm10,%xmm13
   DB  69,15,87,219                        ; xorps         %xmm11,%xmm11
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
   DB  69,15,40,233                        ; movaps        %xmm9,%xmm13
   DB  69,15,92,236                        ; subps         %xmm12,%xmm13
-  DB  68,15,88,13,88,39,0,0               ; addps         0x2758(%rip),%xmm9        # 5450 <_sk_callback_sse2+0x914>
-  DB  68,15,40,37,96,39,0,0               ; movaps        0x2760(%rip),%xmm12        # 5460 <_sk_callback_sse2+0x924>
+  DB  68,15,88,13,94,39,0,0               ; addps         0x275e(%rip),%xmm9        # 5480 <_sk_callback_sse2+0x91a>
+  DB  68,15,40,37,102,39,0,0              ; movaps        0x2766(%rip),%xmm12        # 5490 <_sk_callback_sse2+0x92a>
   DB  69,15,89,229                        ; mulps         %xmm13,%xmm12
   DB  69,15,92,204                        ; subps         %xmm12,%xmm9
-  DB  68,15,40,37,96,39,0,0               ; movaps        0x2760(%rip),%xmm12        # 5470 <_sk_callback_sse2+0x934>
+  DB  68,15,40,37,102,39,0,0              ; movaps        0x2766(%rip),%xmm12        # 54a0 <_sk_callback_sse2+0x93a>
   DB  69,15,92,229                        ; subps         %xmm13,%xmm12
-  DB  68,15,40,45,100,39,0,0              ; movaps        0x2764(%rip),%xmm13        # 5480 <_sk_callback_sse2+0x944>
+  DB  68,15,40,45,106,39,0,0              ; movaps        0x276a(%rip),%xmm13        # 54b0 <_sk_callback_sse2+0x94a>
   DB  69,15,94,236                        ; divps         %xmm12,%xmm13
   DB  69,15,88,233                        ; addps         %xmm9,%xmm13
-  DB  68,15,89,45,100,39,0,0              ; mulps         0x2764(%rip),%xmm13        # 5490 <_sk_callback_sse2+0x954>
+  DB  68,15,89,45,106,39,0,0              ; mulps         0x276a(%rip),%xmm13        # 54c0 <_sk_callback_sse2+0x95a>
   DB  102,69,15,91,205                    ; cvtps2dq      %xmm13,%xmm9
   DB  243,68,15,16,96,20                  ; movss         0x14(%rax),%xmm12
   DB  69,15,198,228,0                     ; shufps        $0x0,%xmm12,%xmm12
@@ -20658,29 +20692,29 @@ _sk_parametric_a_sse2 LABEL PROC
 
 PUBLIC _sk_lab_to_xyz_sse2
 _sk_lab_to_xyz_sse2 LABEL PROC
-  DB  15,89,5,65,39,0,0                   ; mulps         0x2741(%rip),%xmm0        # 54a0 <_sk_callback_sse2+0x964>
-  DB  68,15,40,5,73,39,0,0                ; movaps        0x2749(%rip),%xmm8        # 54b0 <_sk_callback_sse2+0x974>
+  DB  15,89,5,71,39,0,0                   ; mulps         0x2747(%rip),%xmm0        # 54d0 <_sk_callback_sse2+0x96a>
+  DB  68,15,40,5,79,39,0,0                ; movaps        0x274f(%rip),%xmm8        # 54e0 <_sk_callback_sse2+0x97a>
   DB  65,15,89,200                        ; mulps         %xmm8,%xmm1
-  DB  68,15,40,13,77,39,0,0               ; movaps        0x274d(%rip),%xmm9        # 54c0 <_sk_callback_sse2+0x984>
+  DB  68,15,40,13,83,39,0,0               ; movaps        0x2753(%rip),%xmm9        # 54f0 <_sk_callback_sse2+0x98a>
   DB  65,15,88,201                        ; addps         %xmm9,%xmm1
   DB  65,15,89,208                        ; mulps         %xmm8,%xmm2
   DB  65,15,88,209                        ; addps         %xmm9,%xmm2
-  DB  15,88,5,74,39,0,0                   ; addps         0x274a(%rip),%xmm0        # 54d0 <_sk_callback_sse2+0x994>
-  DB  15,89,5,83,39,0,0                   ; mulps         0x2753(%rip),%xmm0        # 54e0 <_sk_callback_sse2+0x9a4>
-  DB  15,89,13,92,39,0,0                  ; mulps         0x275c(%rip),%xmm1        # 54f0 <_sk_callback_sse2+0x9b4>
+  DB  15,88,5,80,39,0,0                   ; addps         0x2750(%rip),%xmm0        # 5500 <_sk_callback_sse2+0x99a>
+  DB  15,89,5,89,39,0,0                   ; mulps         0x2759(%rip),%xmm0        # 5510 <_sk_callback_sse2+0x9aa>
+  DB  15,89,13,98,39,0,0                  ; mulps         0x2762(%rip),%xmm1        # 5520 <_sk_callback_sse2+0x9ba>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  15,89,21,98,39,0,0                  ; mulps         0x2762(%rip),%xmm2        # 5500 <_sk_callback_sse2+0x9c4>
+  DB  15,89,21,104,39,0,0                 ; mulps         0x2768(%rip),%xmm2        # 5530 <_sk_callback_sse2+0x9ca>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  68,15,92,202                        ; subps         %xmm2,%xmm9
   DB  68,15,40,225                        ; movaps        %xmm1,%xmm12
   DB  69,15,89,228                        ; mulps         %xmm12,%xmm12
   DB  68,15,89,225                        ; mulps         %xmm1,%xmm12
-  DB  15,40,21,87,39,0,0                  ; movaps        0x2757(%rip),%xmm2        # 5510 <_sk_callback_sse2+0x9d4>
+  DB  15,40,21,93,39,0,0                  ; movaps        0x275d(%rip),%xmm2        # 5540 <_sk_callback_sse2+0x9da>
   DB  68,15,40,194                        ; movaps        %xmm2,%xmm8
   DB  69,15,194,196,1                     ; cmpltps       %xmm12,%xmm8
-  DB  68,15,40,21,86,39,0,0               ; movaps        0x2756(%rip),%xmm10        # 5520 <_sk_callback_sse2+0x9e4>
+  DB  68,15,40,21,92,39,0,0               ; movaps        0x275c(%rip),%xmm10        # 5550 <_sk_callback_sse2+0x9ea>
   DB  65,15,88,202                        ; addps         %xmm10,%xmm1
-  DB  68,15,40,29,90,39,0,0               ; movaps        0x275a(%rip),%xmm11        # 5530 <_sk_callback_sse2+0x9f4>
+  DB  68,15,40,29,96,39,0,0               ; movaps        0x2760(%rip),%xmm11        # 5560 <_sk_callback_sse2+0x9fa>
   DB  65,15,89,203                        ; mulps         %xmm11,%xmm1
   DB  69,15,84,224                        ; andps         %xmm8,%xmm12
   DB  68,15,85,193                        ; andnps        %xmm1,%xmm8
@@ -20704,8 +20738,8 @@ _sk_lab_to_xyz_sse2 LABEL PROC
   DB  15,84,194                           ; andps         %xmm2,%xmm0
   DB  65,15,85,209                        ; andnps        %xmm9,%xmm2
   DB  15,86,208                           ; orps          %xmm0,%xmm2
-  DB  68,15,89,5,10,39,0,0                ; mulps         0x270a(%rip),%xmm8        # 5540 <_sk_callback_sse2+0xa04>
-  DB  15,89,21,19,39,0,0                  ; mulps         0x2713(%rip),%xmm2        # 5550 <_sk_callback_sse2+0xa14>
+  DB  68,15,89,5,16,39,0,0                ; mulps         0x2710(%rip),%xmm8        # 5570 <_sk_callback_sse2+0xa0a>
+  DB  15,89,21,25,39,0,0                  ; mulps         0x2719(%rip),%xmm2        # 5580 <_sk_callback_sse2+0xa1a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  65,15,40,192                        ; movaps        %xmm8,%xmm0
   DB  255,224                             ; jmpq          *%rax
@@ -20719,7 +20753,7 @@ _sk_load_a8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,251,38,0,0                 ; mulps         0x26fb(%rip),%xmm3        # 5560 <_sk_callback_sse2+0xa24>
+  DB  15,89,29,1,39,0,0                   ; mulps         0x2701(%rip),%xmm3        # 5590 <_sk_callback_sse2+0xa2a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
@@ -20762,7 +20796,7 @@ _sk_gather_a8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,216                           ; cvtdq2ps      %xmm0,%xmm3
-  DB  15,89,29,106,38,0,0                 ; mulps         0x266a(%rip),%xmm3        # 5570 <_sk_callback_sse2+0xa34>
+  DB  15,89,29,112,38,0,0                 ; mulps         0x2670(%rip),%xmm3        # 55a0 <_sk_callback_sse2+0xa3a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
@@ -20773,7 +20807,7 @@ PUBLIC _sk_store_a8_sse2
 _sk_store_a8_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,94,38,0,0                ; movaps        0x265e(%rip),%xmm8        # 5580 <_sk_callback_sse2+0xa44>
+  DB  68,15,40,5,100,38,0,0               ; movaps        0x2664(%rip),%xmm8        # 55b0 <_sk_callback_sse2+0xa4a>
   DB  68,15,89,195                        ; mulps         %xmm3,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
   DB  102,65,15,114,240,16                ; pslld         $0x10,%xmm8
@@ -20793,9 +20827,9 @@ _sk_load_g8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,37,38,0,0                   ; mulps         0x2625(%rip),%xmm0        # 5590 <_sk_callback_sse2+0xa54>
+  DB  15,89,5,43,38,0,0                   ; mulps         0x262b(%rip),%xmm0        # 55c0 <_sk_callback_sse2+0xa5a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,44,38,0,0                  ; movaps        0x262c(%rip),%xmm3        # 55a0 <_sk_callback_sse2+0xa64>
+  DB  15,40,29,50,38,0,0                  ; movaps        0x2632(%rip),%xmm3        # 55d0 <_sk_callback_sse2+0xa6a>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -20836,9 +20870,9 @@ _sk_gather_g8_sse2 LABEL PROC
   DB  102,15,96,193                       ; punpcklbw     %xmm1,%xmm0
   DB  102,15,97,193                       ; punpcklwd     %xmm1,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,161,37,0,0                  ; mulps         0x25a1(%rip),%xmm0        # 55b0 <_sk_callback_sse2+0xa74>
+  DB  15,89,5,167,37,0,0                  ; mulps         0x25a7(%rip),%xmm0        # 55e0 <_sk_callback_sse2+0xa7a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,168,37,0,0                 ; movaps        0x25a8(%rip),%xmm3        # 55c0 <_sk_callback_sse2+0xa84>
+  DB  15,40,29,174,37,0,0                 ; movaps        0x25ae(%rip),%xmm3        # 55f0 <_sk_callback_sse2+0xa8a>
   DB  15,40,200                           ; movaps        %xmm0,%xmm1
   DB  15,40,208                           ; movaps        %xmm0,%xmm2
   DB  255,224                             ; jmpq          *%rax
@@ -20848,9 +20882,9 @@ _sk_gather_i8_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  73,137,192                          ; mov           %rax,%r8
   DB  77,133,192                          ; test          %r8,%r8
-  DB  116,5                               ; je            302f <_sk_gather_i8_sse2+0xf>
+  DB  116,5                               ; je            3059 <_sk_gather_i8_sse2+0xf>
   DB  76,137,192                          ; mov           %r8,%rax
-  DB  235,2                               ; jmp           3031 <_sk_gather_i8_sse2+0x11>
+  DB  235,2                               ; jmp           305b <_sk_gather_i8_sse2+0x11>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  76,139,16                           ; mov           (%rax),%r10
   DB  243,15,91,201                       ; cvttps2dq     %xmm1,%xmm1
@@ -20899,11 +20933,11 @@ _sk_gather_i8_sse2 LABEL PROC
   DB  102,67,15,110,12,136                ; movd          (%r8,%r9,4),%xmm1
   DB  102,68,15,98,201                    ; punpckldq     %xmm1,%xmm9
   DB  102,68,15,98,200                    ; punpckldq     %xmm0,%xmm9
-  DB  102,15,111,21,199,36,0,0            ; movdqa        0x24c7(%rip),%xmm2        # 55d0 <_sk_callback_sse2+0xa94>
+  DB  102,15,111,21,205,36,0,0            ; movdqa        0x24cd(%rip),%xmm2        # 5600 <_sk_callback_sse2+0xa9a>
   DB  102,65,15,111,193                   ; movdqa        %xmm9,%xmm0
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,195,36,0,0               ; movaps        0x24c3(%rip),%xmm8        # 55e0 <_sk_callback_sse2+0xaa4>
+  DB  68,15,40,5,201,36,0,0               ; movaps        0x24c9(%rip),%xmm8        # 5610 <_sk_callback_sse2+0xaaa>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,114,209,8                    ; psrld         $0x8,%xmm1
@@ -20928,19 +20962,19 @@ _sk_load_565_sse2 LABEL PROC
   DB  243,15,126,20,120                   ; movq          (%rax,%rdi,2),%xmm2
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,208                       ; punpcklwd     %xmm0,%xmm2
-  DB  102,15,111,5,121,36,0,0             ; movdqa        0x2479(%rip),%xmm0        # 55f0 <_sk_callback_sse2+0xab4>
+  DB  102,15,111,5,127,36,0,0             ; movdqa        0x247f(%rip),%xmm0        # 5620 <_sk_callback_sse2+0xaba>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,123,36,0,0                  ; mulps         0x247b(%rip),%xmm0        # 5600 <_sk_callback_sse2+0xac4>
-  DB  102,15,111,13,131,36,0,0            ; movdqa        0x2483(%rip),%xmm1        # 5610 <_sk_callback_sse2+0xad4>
+  DB  15,89,5,129,36,0,0                  ; mulps         0x2481(%rip),%xmm0        # 5630 <_sk_callback_sse2+0xaca>
+  DB  102,15,111,13,137,36,0,0            ; movdqa        0x2489(%rip),%xmm1        # 5640 <_sk_callback_sse2+0xada>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,133,36,0,0                 ; mulps         0x2485(%rip),%xmm1        # 5620 <_sk_callback_sse2+0xae4>
-  DB  102,15,219,21,141,36,0,0            ; pand          0x248d(%rip),%xmm2        # 5630 <_sk_callback_sse2+0xaf4>
+  DB  15,89,13,139,36,0,0                 ; mulps         0x248b(%rip),%xmm1        # 5650 <_sk_callback_sse2+0xaea>
+  DB  102,15,219,21,147,36,0,0            ; pand          0x2493(%rip),%xmm2        # 5660 <_sk_callback_sse2+0xafa>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,147,36,0,0                 ; mulps         0x2493(%rip),%xmm2        # 5640 <_sk_callback_sse2+0xb04>
+  DB  15,89,21,153,36,0,0                 ; mulps         0x2499(%rip),%xmm2        # 5670 <_sk_callback_sse2+0xb0a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,154,36,0,0                 ; movaps        0x249a(%rip),%xmm3        # 5650 <_sk_callback_sse2+0xb14>
+  DB  15,40,29,160,36,0,0                 ; movaps        0x24a0(%rip),%xmm3        # 5680 <_sk_callback_sse2+0xb1a>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_gather_565_sse2
@@ -20973,31 +21007,31 @@ _sk_gather_565_sse2 LABEL PROC
   DB  102,15,196,208,3                    ; pinsrw        $0x3,%eax,%xmm2
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,208                       ; punpcklwd     %xmm0,%xmm2
-  DB  102,15,111,5,35,36,0,0              ; movdqa        0x2423(%rip),%xmm0        # 5660 <_sk_callback_sse2+0xb24>
+  DB  102,15,111,5,41,36,0,0              ; movdqa        0x2429(%rip),%xmm0        # 5690 <_sk_callback_sse2+0xb2a>
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,37,36,0,0                   ; mulps         0x2425(%rip),%xmm0        # 5670 <_sk_callback_sse2+0xb34>
-  DB  102,15,111,13,45,36,0,0             ; movdqa        0x242d(%rip),%xmm1        # 5680 <_sk_callback_sse2+0xb44>
+  DB  15,89,5,43,36,0,0                   ; mulps         0x242b(%rip),%xmm0        # 56a0 <_sk_callback_sse2+0xb3a>
+  DB  102,15,111,13,51,36,0,0             ; movdqa        0x2433(%rip),%xmm1        # 56b0 <_sk_callback_sse2+0xb4a>
   DB  102,15,219,202                      ; pand          %xmm2,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,47,36,0,0                  ; mulps         0x242f(%rip),%xmm1        # 5690 <_sk_callback_sse2+0xb54>
-  DB  102,15,219,21,55,36,0,0             ; pand          0x2437(%rip),%xmm2        # 56a0 <_sk_callback_sse2+0xb64>
+  DB  15,89,13,53,36,0,0                  ; mulps         0x2435(%rip),%xmm1        # 56c0 <_sk_callback_sse2+0xb5a>
+  DB  102,15,219,21,61,36,0,0             ; pand          0x243d(%rip),%xmm2        # 56d0 <_sk_callback_sse2+0xb6a>
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,61,36,0,0                  ; mulps         0x243d(%rip),%xmm2        # 56b0 <_sk_callback_sse2+0xb74>
+  DB  15,89,21,67,36,0,0                  ; mulps         0x2443(%rip),%xmm2        # 56e0 <_sk_callback_sse2+0xb7a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,68,36,0,0                  ; movaps        0x2444(%rip),%xmm3        # 56c0 <_sk_callback_sse2+0xb84>
+  DB  15,40,29,74,36,0,0                  ; movaps        0x244a(%rip),%xmm3        # 56f0 <_sk_callback_sse2+0xb8a>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_565_sse2
 _sk_store_565_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,69,36,0,0                ; movaps        0x2445(%rip),%xmm8        # 56d0 <_sk_callback_sse2+0xb94>
+  DB  68,15,40,5,75,36,0,0                ; movaps        0x244b(%rip),%xmm8        # 5700 <_sk_callback_sse2+0xb9a>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
   DB  102,65,15,114,241,11                ; pslld         $0xb,%xmm9
-  DB  68,15,40,21,58,36,0,0               ; movaps        0x243a(%rip),%xmm10        # 56e0 <_sk_callback_sse2+0xba4>
+  DB  68,15,40,21,64,36,0,0               ; movaps        0x2440(%rip),%xmm10        # 5710 <_sk_callback_sse2+0xbaa>
   DB  68,15,89,209                        ; mulps         %xmm1,%xmm10
   DB  102,69,15,91,210                    ; cvtps2dq      %xmm10,%xmm10
   DB  102,65,15,114,242,5                 ; pslld         $0x5,%xmm10
@@ -21019,21 +21053,21 @@ _sk_load_4444_sse2 LABEL PROC
   DB  243,15,126,28,120                   ; movq          (%rax,%rdi,2),%xmm3
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,216                       ; punpcklwd     %xmm0,%xmm3
-  DB  102,15,111,5,243,35,0,0             ; movdqa        0x23f3(%rip),%xmm0        # 56f0 <_sk_callback_sse2+0xbb4>
+  DB  102,15,111,5,249,35,0,0             ; movdqa        0x23f9(%rip),%xmm0        # 5720 <_sk_callback_sse2+0xbba>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,245,35,0,0                  ; mulps         0x23f5(%rip),%xmm0        # 5700 <_sk_callback_sse2+0xbc4>
-  DB  102,15,111,13,253,35,0,0            ; movdqa        0x23fd(%rip),%xmm1        # 5710 <_sk_callback_sse2+0xbd4>
+  DB  15,89,5,251,35,0,0                  ; mulps         0x23fb(%rip),%xmm0        # 5730 <_sk_callback_sse2+0xbca>
+  DB  102,15,111,13,3,36,0,0              ; movdqa        0x2403(%rip),%xmm1        # 5740 <_sk_callback_sse2+0xbda>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,255,35,0,0                 ; mulps         0x23ff(%rip),%xmm1        # 5720 <_sk_callback_sse2+0xbe4>
-  DB  102,15,111,21,7,36,0,0              ; movdqa        0x2407(%rip),%xmm2        # 5730 <_sk_callback_sse2+0xbf4>
+  DB  15,89,13,5,36,0,0                   ; mulps         0x2405(%rip),%xmm1        # 5750 <_sk_callback_sse2+0xbea>
+  DB  102,15,111,21,13,36,0,0             ; movdqa        0x240d(%rip),%xmm2        # 5760 <_sk_callback_sse2+0xbfa>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,9,36,0,0                   ; mulps         0x2409(%rip),%xmm2        # 5740 <_sk_callback_sse2+0xc04>
-  DB  102,15,219,29,17,36,0,0             ; pand          0x2411(%rip),%xmm3        # 5750 <_sk_callback_sse2+0xc14>
+  DB  15,89,21,15,36,0,0                  ; mulps         0x240f(%rip),%xmm2        # 5770 <_sk_callback_sse2+0xc0a>
+  DB  102,15,219,29,23,36,0,0             ; pand          0x2417(%rip),%xmm3        # 5780 <_sk_callback_sse2+0xc1a>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,23,36,0,0                  ; mulps         0x2417(%rip),%xmm3        # 5760 <_sk_callback_sse2+0xc24>
+  DB  15,89,29,29,36,0,0                  ; mulps         0x241d(%rip),%xmm3        # 5790 <_sk_callback_sse2+0xc2a>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -21067,21 +21101,21 @@ _sk_gather_4444_sse2 LABEL PROC
   DB  102,15,196,216,3                    ; pinsrw        $0x3,%eax,%xmm3
   DB  102,15,239,192                      ; pxor          %xmm0,%xmm0
   DB  102,15,97,216                       ; punpcklwd     %xmm0,%xmm3
-  DB  102,15,111,5,158,35,0,0             ; movdqa        0x239e(%rip),%xmm0        # 5770 <_sk_callback_sse2+0xc34>
+  DB  102,15,111,5,164,35,0,0             ; movdqa        0x23a4(%rip),%xmm0        # 57a0 <_sk_callback_sse2+0xc3a>
   DB  102,15,219,195                      ; pand          %xmm3,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  15,89,5,160,35,0,0                  ; mulps         0x23a0(%rip),%xmm0        # 5780 <_sk_callback_sse2+0xc44>
-  DB  102,15,111,13,168,35,0,0            ; movdqa        0x23a8(%rip),%xmm1        # 5790 <_sk_callback_sse2+0xc54>
+  DB  15,89,5,166,35,0,0                  ; mulps         0x23a6(%rip),%xmm0        # 57b0 <_sk_callback_sse2+0xc4a>
+  DB  102,15,111,13,174,35,0,0            ; movdqa        0x23ae(%rip),%xmm1        # 57c0 <_sk_callback_sse2+0xc5a>
   DB  102,15,219,203                      ; pand          %xmm3,%xmm1
   DB  15,91,201                           ; cvtdq2ps      %xmm1,%xmm1
-  DB  15,89,13,170,35,0,0                 ; mulps         0x23aa(%rip),%xmm1        # 57a0 <_sk_callback_sse2+0xc64>
-  DB  102,15,111,21,178,35,0,0            ; movdqa        0x23b2(%rip),%xmm2        # 57b0 <_sk_callback_sse2+0xc74>
+  DB  15,89,13,176,35,0,0                 ; mulps         0x23b0(%rip),%xmm1        # 57d0 <_sk_callback_sse2+0xc6a>
+  DB  102,15,111,21,184,35,0,0            ; movdqa        0x23b8(%rip),%xmm2        # 57e0 <_sk_callback_sse2+0xc7a>
   DB  102,15,219,211                      ; pand          %xmm3,%xmm2
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
-  DB  15,89,21,180,35,0,0                 ; mulps         0x23b4(%rip),%xmm2        # 57c0 <_sk_callback_sse2+0xc84>
-  DB  102,15,219,29,188,35,0,0            ; pand          0x23bc(%rip),%xmm3        # 57d0 <_sk_callback_sse2+0xc94>
+  DB  15,89,21,186,35,0,0                 ; mulps         0x23ba(%rip),%xmm2        # 57f0 <_sk_callback_sse2+0xc8a>
+  DB  102,15,219,29,194,35,0,0            ; pand          0x23c2(%rip),%xmm3        # 5800 <_sk_callback_sse2+0xc9a>
   DB  15,91,219                           ; cvtdq2ps      %xmm3,%xmm3
-  DB  15,89,29,194,35,0,0                 ; mulps         0x23c2(%rip),%xmm3        # 57e0 <_sk_callback_sse2+0xca4>
+  DB  15,89,29,200,35,0,0                 ; mulps         0x23c8(%rip),%xmm3        # 5810 <_sk_callback_sse2+0xcaa>
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
 
@@ -21089,7 +21123,7 @@ PUBLIC _sk_store_4444_sse2
 _sk_store_4444_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,193,35,0,0               ; movaps        0x23c1(%rip),%xmm8        # 57f0 <_sk_callback_sse2+0xcb4>
+  DB  68,15,40,5,199,35,0,0               ; movaps        0x23c7(%rip),%xmm8        # 5820 <_sk_callback_sse2+0xcba>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -21119,11 +21153,11 @@ _sk_load_8888_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
   DB  68,15,16,12,184                     ; movups        (%rax,%rdi,4),%xmm9
-  DB  15,40,21,84,35,0,0                  ; movaps        0x2354(%rip),%xmm2        # 5800 <_sk_callback_sse2+0xcc4>
+  DB  15,40,21,90,35,0,0                  ; movaps        0x235a(%rip),%xmm2        # 5830 <_sk_callback_sse2+0xcca>
   DB  65,15,40,193                        ; movaps        %xmm9,%xmm0
   DB  15,84,194                           ; andps         %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,82,35,0,0                ; movaps        0x2352(%rip),%xmm8        # 5810 <_sk_callback_sse2+0xcd4>
+  DB  68,15,40,5,88,35,0,0                ; movaps        0x2358(%rip),%xmm8        # 5840 <_sk_callback_sse2+0xcda>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  65,15,40,201                        ; movaps        %xmm9,%xmm1
   DB  102,15,114,209,8                    ; psrld         $0x8,%xmm1
@@ -21170,11 +21204,11 @@ _sk_gather_8888_sse2 LABEL PROC
   DB  102,67,15,110,12,129                ; movd          (%r9,%r8,4),%xmm1
   DB  102,68,15,98,201                    ; punpckldq     %xmm1,%xmm9
   DB  102,68,15,98,200                    ; punpckldq     %xmm0,%xmm9
-  DB  102,15,111,21,163,34,0,0            ; movdqa        0x22a3(%rip),%xmm2        # 5820 <_sk_callback_sse2+0xce4>
+  DB  102,15,111,21,169,34,0,0            ; movdqa        0x22a9(%rip),%xmm2        # 5850 <_sk_callback_sse2+0xcea>
   DB  102,65,15,111,193                   ; movdqa        %xmm9,%xmm0
   DB  102,15,219,194                      ; pand          %xmm2,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,5,159,34,0,0               ; movaps        0x229f(%rip),%xmm8        # 5830 <_sk_callback_sse2+0xcf4>
+  DB  68,15,40,5,165,34,0,0               ; movaps        0x22a5(%rip),%xmm8        # 5860 <_sk_callback_sse2+0xcfa>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,65,15,111,201                   ; movdqa        %xmm9,%xmm1
   DB  102,15,114,209,8                    ; psrld         $0x8,%xmm1
@@ -21196,7 +21230,7 @@ PUBLIC _sk_store_8888_sse2
 _sk_store_8888_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,5,98,34,0,0                ; movaps        0x2262(%rip),%xmm8        # 5840 <_sk_callback_sse2+0xd04>
+  DB  68,15,40,5,104,34,0,0               ; movaps        0x2268(%rip),%xmm8        # 5870 <_sk_callback_sse2+0xd0a>
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  102,69,15,91,201                    ; cvtps2dq      %xmm9,%xmm9
@@ -21233,7 +21267,7 @@ _sk_load_f16_sse2 LABEL PROC
   DB  102,69,15,239,210                   ; pxor          %xmm10,%xmm10
   DB  102,65,15,111,206                   ; movdqa        %xmm14,%xmm1
   DB  102,65,15,97,202                    ; punpcklwd     %xmm10,%xmm1
-  DB  102,68,15,111,13,210,33,0,0         ; movdqa        0x21d2(%rip),%xmm9        # 5850 <_sk_callback_sse2+0xd14>
+  DB  102,68,15,111,13,216,33,0,0         ; movdqa        0x21d8(%rip),%xmm9        # 5880 <_sk_callback_sse2+0xd1a>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,193                   ; pand          %xmm9,%xmm0
   DB  102,15,239,200                      ; pxor          %xmm0,%xmm1
@@ -21241,11 +21275,11 @@ _sk_load_f16_sse2 LABEL PROC
   DB  102,68,15,111,233                   ; movdqa        %xmm1,%xmm13
   DB  102,65,15,114,245,13                ; pslld         $0xd,%xmm13
   DB  102,68,15,235,232                   ; por           %xmm0,%xmm13
-  DB  102,68,15,111,29,183,33,0,0         ; movdqa        0x21b7(%rip),%xmm11        # 5860 <_sk_callback_sse2+0xd24>
+  DB  102,68,15,111,29,189,33,0,0         ; movdqa        0x21bd(%rip),%xmm11        # 5890 <_sk_callback_sse2+0xd2a>
   DB  102,69,15,254,235                   ; paddd         %xmm11,%xmm13
-  DB  102,68,15,111,37,185,33,0,0         ; movdqa        0x21b9(%rip),%xmm12        # 5870 <_sk_callback_sse2+0xd34>
+  DB  102,68,15,111,37,191,33,0,0         ; movdqa        0x21bf(%rip),%xmm12        # 58a0 <_sk_callback_sse2+0xd3a>
   DB  102,65,15,239,204                   ; pxor          %xmm12,%xmm1
-  DB  102,15,111,29,188,33,0,0            ; movdqa        0x21bc(%rip),%xmm3        # 5880 <_sk_callback_sse2+0xd44>
+  DB  102,15,111,29,194,33,0,0            ; movdqa        0x21c2(%rip),%xmm3        # 58b0 <_sk_callback_sse2+0xd4a>
   DB  102,15,111,195                      ; movdqa        %xmm3,%xmm0
   DB  102,15,102,193                      ; pcmpgtd       %xmm1,%xmm0
   DB  102,65,15,223,197                   ; pandn         %xmm13,%xmm0
@@ -21329,7 +21363,7 @@ _sk_gather_f16_sse2 LABEL PROC
   DB  102,69,15,239,210                   ; pxor          %xmm10,%xmm10
   DB  102,65,15,111,206                   ; movdqa        %xmm14,%xmm1
   DB  102,65,15,97,202                    ; punpcklwd     %xmm10,%xmm1
-  DB  102,68,15,111,13,74,32,0,0          ; movdqa        0x204a(%rip),%xmm9        # 5890 <_sk_callback_sse2+0xd54>
+  DB  102,68,15,111,13,80,32,0,0          ; movdqa        0x2050(%rip),%xmm9        # 58c0 <_sk_callback_sse2+0xd5a>
   DB  102,15,111,193                      ; movdqa        %xmm1,%xmm0
   DB  102,65,15,219,193                   ; pand          %xmm9,%xmm0
   DB  102,15,239,200                      ; pxor          %xmm0,%xmm1
@@ -21337,11 +21371,11 @@ _sk_gather_f16_sse2 LABEL PROC
   DB  102,68,15,111,233                   ; movdqa        %xmm1,%xmm13
   DB  102,65,15,114,245,13                ; pslld         $0xd,%xmm13
   DB  102,68,15,235,232                   ; por           %xmm0,%xmm13
-  DB  102,68,15,111,29,47,32,0,0          ; movdqa        0x202f(%rip),%xmm11        # 58a0 <_sk_callback_sse2+0xd64>
+  DB  102,68,15,111,29,53,32,0,0          ; movdqa        0x2035(%rip),%xmm11        # 58d0 <_sk_callback_sse2+0xd6a>
   DB  102,69,15,254,235                   ; paddd         %xmm11,%xmm13
-  DB  102,68,15,111,37,49,32,0,0          ; movdqa        0x2031(%rip),%xmm12        # 58b0 <_sk_callback_sse2+0xd74>
+  DB  102,68,15,111,37,55,32,0,0          ; movdqa        0x2037(%rip),%xmm12        # 58e0 <_sk_callback_sse2+0xd7a>
   DB  102,65,15,239,204                   ; pxor          %xmm12,%xmm1
-  DB  102,15,111,29,52,32,0,0             ; movdqa        0x2034(%rip),%xmm3        # 58c0 <_sk_callback_sse2+0xd84>
+  DB  102,15,111,29,58,32,0,0             ; movdqa        0x203a(%rip),%xmm3        # 58f0 <_sk_callback_sse2+0xd8a>
   DB  102,15,111,195                      ; movdqa        %xmm3,%xmm0
   DB  102,15,102,193                      ; pcmpgtd       %xmm1,%xmm0
   DB  102,65,15,223,197                   ; pandn         %xmm13,%xmm0
@@ -21392,17 +21426,17 @@ PUBLIC _sk_store_f16_sse2
 _sk_store_f16_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  102,68,15,111,21,92,31,0,0          ; movdqa        0x1f5c(%rip),%xmm10        # 58d0 <_sk_callback_sse2+0xd94>
+  DB  102,68,15,111,21,98,31,0,0          ; movdqa        0x1f62(%rip),%xmm10        # 5900 <_sk_callback_sse2+0xd9a>
   DB  102,68,15,111,224                   ; movdqa        %xmm0,%xmm12
   DB  102,68,15,111,232                   ; movdqa        %xmm0,%xmm13
   DB  102,69,15,219,234                   ; pand          %xmm10,%xmm13
   DB  102,69,15,239,229                   ; pxor          %xmm13,%xmm12
-  DB  102,68,15,111,13,79,31,0,0          ; movdqa        0x1f4f(%rip),%xmm9        # 58e0 <_sk_callback_sse2+0xda4>
+  DB  102,68,15,111,13,85,31,0,0          ; movdqa        0x1f55(%rip),%xmm9        # 5910 <_sk_callback_sse2+0xdaa>
   DB  102,65,15,114,213,16                ; psrld         $0x10,%xmm13
   DB  102,69,15,111,193                   ; movdqa        %xmm9,%xmm8
   DB  102,69,15,102,196                   ; pcmpgtd       %xmm12,%xmm8
   DB  102,65,15,114,212,13                ; psrld         $0xd,%xmm12
-  DB  102,68,15,111,29,64,31,0,0          ; movdqa        0x1f40(%rip),%xmm11        # 58f0 <_sk_callback_sse2+0xdb4>
+  DB  102,68,15,111,29,70,31,0,0          ; movdqa        0x1f46(%rip),%xmm11        # 5920 <_sk_callback_sse2+0xdba>
   DB  102,69,15,235,235                   ; por           %xmm11,%xmm13
   DB  102,69,15,254,236                   ; paddd         %xmm12,%xmm13
   DB  102,65,15,114,245,16                ; pslld         $0x10,%xmm13
@@ -21479,7 +21513,7 @@ _sk_load_u16_be_sse2 LABEL PROC
   DB  102,69,15,239,201                   ; pxor          %xmm9,%xmm9
   DB  102,65,15,97,201                    ; punpcklwd     %xmm9,%xmm1
   DB  15,91,193                           ; cvtdq2ps      %xmm1,%xmm0
-  DB  68,15,40,5,222,29,0,0               ; movaps        0x1dde(%rip),%xmm8        # 5900 <_sk_callback_sse2+0xdc4>
+  DB  68,15,40,5,228,29,0,0               ; movaps        0x1de4(%rip),%xmm8        # 5930 <_sk_callback_sse2+0xdca>
   DB  65,15,89,192                        ; mulps         %xmm8,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -21530,7 +21564,7 @@ _sk_load_rgb_u16_be_sse2 LABEL PROC
   DB  102,69,15,239,192                   ; pxor          %xmm8,%xmm8
   DB  102,65,15,97,192                    ; punpcklwd     %xmm8,%xmm0
   DB  15,91,192                           ; cvtdq2ps      %xmm0,%xmm0
-  DB  68,15,40,13,26,29,0,0               ; movaps        0x1d1a(%rip),%xmm9        # 5910 <_sk_callback_sse2+0xdd4>
+  DB  68,15,40,13,32,29,0,0               ; movaps        0x1d20(%rip),%xmm9        # 5940 <_sk_callback_sse2+0xdda>
   DB  65,15,89,193                        ; mulps         %xmm9,%xmm0
   DB  102,15,111,203                      ; movdqa        %xmm3,%xmm1
   DB  102,15,113,241,8                    ; psllw         $0x8,%xmm1
@@ -21547,14 +21581,14 @@ _sk_load_rgb_u16_be_sse2 LABEL PROC
   DB  15,91,210                           ; cvtdq2ps      %xmm2,%xmm2
   DB  65,15,89,209                        ; mulps         %xmm9,%xmm2
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  15,40,29,225,28,0,0                 ; movaps        0x1ce1(%rip),%xmm3        # 5920 <_sk_callback_sse2+0xde4>
+  DB  15,40,29,231,28,0,0                 ; movaps        0x1ce7(%rip),%xmm3        # 5950 <_sk_callback_sse2+0xdea>
   DB  255,224                             ; jmpq          *%rax
 
 PUBLIC _sk_store_u16_be_sse2
 _sk_store_u16_be_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  72,139,0                            ; mov           (%rax),%rax
-  DB  68,15,40,13,226,28,0,0              ; movaps        0x1ce2(%rip),%xmm9        # 5930 <_sk_callback_sse2+0xdf4>
+  DB  68,15,40,13,232,28,0,0              ; movaps        0x1ce8(%rip),%xmm9        # 5960 <_sk_callback_sse2+0xdfa>
   DB  68,15,40,192                        ; movaps        %xmm0,%xmm8
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  102,69,15,91,192                    ; cvtps2dq      %xmm8,%xmm8
@@ -21690,7 +21724,7 @@ _sk_repeat_x_sse2 LABEL PROC
   DB  243,69,15,91,209                    ; cvttps2dq     %xmm9,%xmm10
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
   DB  69,15,194,202,1                     ; cmpltps       %xmm10,%xmm9
-  DB  68,15,84,13,226,26,0,0              ; andps         0x1ae2(%rip),%xmm9        # 5940 <_sk_callback_sse2+0xe04>
+  DB  68,15,84,13,232,26,0,0              ; andps         0x1ae8(%rip),%xmm9        # 5970 <_sk_callback_sse2+0xe0a>
   DB  69,15,92,209                        ; subps         %xmm9,%xmm10
   DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
   DB  65,15,92,194                        ; subps         %xmm10,%xmm0
@@ -21708,7 +21742,7 @@ _sk_repeat_y_sse2 LABEL PROC
   DB  243,69,15,91,209                    ; cvttps2dq     %xmm9,%xmm10
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
   DB  69,15,194,202,1                     ; cmpltps       %xmm10,%xmm9
-  DB  68,15,84,13,180,26,0,0              ; andps         0x1ab4(%rip),%xmm9        # 5950 <_sk_callback_sse2+0xe14>
+  DB  68,15,84,13,186,26,0,0              ; andps         0x1aba(%rip),%xmm9        # 5980 <_sk_callback_sse2+0xe1a>
   DB  69,15,92,209                        ; subps         %xmm9,%xmm10
   DB  69,15,89,208                        ; mulps         %xmm8,%xmm10
   DB  65,15,92,202                        ; subps         %xmm10,%xmm1
@@ -21730,7 +21764,7 @@ _sk_mirror_x_sse2 LABEL PROC
   DB  243,69,15,91,218                    ; cvttps2dq     %xmm10,%xmm11
   DB  69,15,91,219                        ; cvtdq2ps      %xmm11,%xmm11
   DB  69,15,194,211,1                     ; cmpltps       %xmm11,%xmm10
-  DB  68,15,84,21,116,26,0,0              ; andps         0x1a74(%rip),%xmm10        # 5960 <_sk_callback_sse2+0xe24>
+  DB  68,15,84,21,122,26,0,0              ; andps         0x1a7a(%rip),%xmm10        # 5990 <_sk_callback_sse2+0xe2a>
   DB  69,15,87,228                        ; xorps         %xmm12,%xmm12
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  69,15,89,216                        ; mulps         %xmm8,%xmm11
@@ -21756,7 +21790,7 @@ _sk_mirror_y_sse2 LABEL PROC
   DB  243,69,15,91,218                    ; cvttps2dq     %xmm10,%xmm11
   DB  69,15,91,219                        ; cvtdq2ps      %xmm11,%xmm11
   DB  69,15,194,211,1                     ; cmpltps       %xmm11,%xmm10
-  DB  68,15,84,21,36,26,0,0               ; andps         0x1a24(%rip),%xmm10        # 5970 <_sk_callback_sse2+0xe34>
+  DB  68,15,84,21,42,26,0,0               ; andps         0x1a2a(%rip),%xmm10        # 59a0 <_sk_callback_sse2+0xe3a>
   DB  69,15,87,228                        ; xorps         %xmm12,%xmm12
   DB  69,15,92,218                        ; subps         %xmm10,%xmm11
   DB  69,15,89,216                        ; mulps         %xmm8,%xmm11
@@ -21771,10 +21805,10 @@ _sk_mirror_y_sse2 LABEL PROC
 PUBLIC _sk_luminance_to_alpha_sse2
 _sk_luminance_to_alpha_sse2 LABEL PROC
   DB  15,40,218                           ; movaps        %xmm2,%xmm3
-  DB  15,89,5,6,26,0,0                    ; mulps         0x1a06(%rip),%xmm0        # 5980 <_sk_callback_sse2+0xe44>
-  DB  15,89,13,15,26,0,0                  ; mulps         0x1a0f(%rip),%xmm1        # 5990 <_sk_callback_sse2+0xe54>
+  DB  15,89,5,12,26,0,0                   ; mulps         0x1a0c(%rip),%xmm0        # 59b0 <_sk_callback_sse2+0xe4a>
+  DB  15,89,13,21,26,0,0                  ; mulps         0x1a15(%rip),%xmm1        # 59c0 <_sk_callback_sse2+0xe5a>
   DB  15,88,200                           ; addps         %xmm0,%xmm1
-  DB  15,89,29,21,26,0,0                  ; mulps         0x1a15(%rip),%xmm3        # 59a0 <_sk_callback_sse2+0xe64>
+  DB  15,89,29,27,26,0,0                  ; mulps         0x1a1b(%rip),%xmm3        # 59d0 <_sk_callback_sse2+0xe6a>
   DB  15,88,217                           ; addps         %xmm1,%xmm3
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,87,192                           ; xorps         %xmm0,%xmm0
@@ -21990,9 +22024,9 @@ _sk_evenly_spaced_gradient_sse2 LABEL PROC
   DB  72,139,8                            ; mov           (%rax),%rcx
   DB  76,139,88,8                         ; mov           0x8(%rax),%r11
   DB  72,255,201                          ; dec           %rcx
-  DB  120,7                               ; js            430a <_sk_evenly_spaced_gradient_sse2+0x15>
+  DB  120,7                               ; js            4334 <_sk_evenly_spaced_gradient_sse2+0x15>
   DB  243,72,15,42,201                    ; cvtsi2ss      %rcx,%xmm1
-  DB  235,21                              ; jmp           431f <_sk_evenly_spaced_gradient_sse2+0x2a>
+  DB  235,21                              ; jmp           4349 <_sk_evenly_spaced_gradient_sse2+0x2a>
   DB  73,137,200                          ; mov           %rcx,%r8
   DB  73,209,232                          ; shr           %r8
   DB  131,225,1                           ; and           $0x1,%ecx
@@ -22090,12 +22124,12 @@ _sk_gradient_sse2 LABEL PROC
   DB  76,139,0                            ; mov           (%rax),%r8
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
   DB  73,131,248,2                        ; cmp           $0x2,%r8
-  DB  114,50                              ; jb            44e2 <_sk_gradient_sse2+0x41>
+  DB  114,50                              ; jb            450c <_sk_gradient_sse2+0x41>
   DB  72,139,72,72                        ; mov           0x48(%rax),%rcx
   DB  73,255,200                          ; dec           %r8
   DB  72,131,193,4                        ; add           $0x4,%rcx
   DB  102,15,239,201                      ; pxor          %xmm1,%xmm1
-  DB  15,40,21,234,20,0,0                 ; movaps        0x14ea(%rip),%xmm2        # 59b0 <_sk_callback_sse2+0xe74>
+  DB  15,40,21,240,20,0,0                 ; movaps        0x14f0(%rip),%xmm2        # 59e0 <_sk_callback_sse2+0xe7a>
   DB  243,15,16,25                        ; movss         (%rcx),%xmm3
   DB  15,198,219,0                        ; shufps        $0x0,%xmm3,%xmm3
   DB  15,194,216,2                        ; cmpleps       %xmm0,%xmm3
@@ -22103,7 +22137,7 @@ _sk_gradient_sse2 LABEL PROC
   DB  102,15,254,203                      ; paddd         %xmm3,%xmm1
   DB  72,131,193,4                        ; add           $0x4,%rcx
   DB  73,255,200                          ; dec           %r8
-  DB  117,228                             ; jne           44c6 <_sk_gradient_sse2+0x25>
+  DB  117,228                             ; jne           44f0 <_sk_gradient_sse2+0x25>
   DB  65,86                               ; push          %r14
   DB  83                                  ; push          %rbx
   DB  102,15,112,209,78                   ; pshufd        $0x4e,%xmm1,%xmm2
@@ -22239,29 +22273,29 @@ _sk_xy_to_unit_angle_sse2 LABEL PROC
   DB  69,15,94,220                        ; divps         %xmm12,%xmm11
   DB  69,15,40,227                        ; movaps        %xmm11,%xmm12
   DB  69,15,89,228                        ; mulps         %xmm12,%xmm12
-  DB  68,15,40,45,172,18,0,0              ; movaps        0x12ac(%rip),%xmm13        # 59c0 <_sk_callback_sse2+0xe84>
+  DB  68,15,40,45,178,18,0,0              ; movaps        0x12b2(%rip),%xmm13        # 59f0 <_sk_callback_sse2+0xe8a>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
-  DB  68,15,88,45,176,18,0,0              ; addps         0x12b0(%rip),%xmm13        # 59d0 <_sk_callback_sse2+0xe94>
+  DB  68,15,88,45,182,18,0,0              ; addps         0x12b6(%rip),%xmm13        # 5a00 <_sk_callback_sse2+0xe9a>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
-  DB  68,15,88,45,180,18,0,0              ; addps         0x12b4(%rip),%xmm13        # 59e0 <_sk_callback_sse2+0xea4>
+  DB  68,15,88,45,186,18,0,0              ; addps         0x12ba(%rip),%xmm13        # 5a10 <_sk_callback_sse2+0xeaa>
   DB  69,15,89,236                        ; mulps         %xmm12,%xmm13
-  DB  68,15,88,45,184,18,0,0              ; addps         0x12b8(%rip),%xmm13        # 59f0 <_sk_callback_sse2+0xeb4>
+  DB  68,15,88,45,190,18,0,0              ; addps         0x12be(%rip),%xmm13        # 5a20 <_sk_callback_sse2+0xeba>
   DB  69,15,89,235                        ; mulps         %xmm11,%xmm13
   DB  69,15,194,202,1                     ; cmpltps       %xmm10,%xmm9
-  DB  68,15,40,21,183,18,0,0              ; movaps        0x12b7(%rip),%xmm10        # 5a00 <_sk_callback_sse2+0xec4>
+  DB  68,15,40,21,189,18,0,0              ; movaps        0x12bd(%rip),%xmm10        # 5a30 <_sk_callback_sse2+0xeca>
   DB  69,15,92,213                        ; subps         %xmm13,%xmm10
   DB  69,15,84,209                        ; andps         %xmm9,%xmm10
   DB  69,15,85,205                        ; andnps        %xmm13,%xmm9
   DB  69,15,86,202                        ; orps          %xmm10,%xmm9
   DB  68,15,194,192,1                     ; cmpltps       %xmm0,%xmm8
-  DB  68,15,40,21,170,18,0,0              ; movaps        0x12aa(%rip),%xmm10        # 5a10 <_sk_callback_sse2+0xed4>
+  DB  68,15,40,21,176,18,0,0              ; movaps        0x12b0(%rip),%xmm10        # 5a40 <_sk_callback_sse2+0xeda>
   DB  69,15,92,209                        ; subps         %xmm9,%xmm10
   DB  69,15,84,208                        ; andps         %xmm8,%xmm10
   DB  69,15,85,193                        ; andnps        %xmm9,%xmm8
   DB  69,15,86,194                        ; orps          %xmm10,%xmm8
   DB  68,15,40,201                        ; movaps        %xmm1,%xmm9
   DB  68,15,194,200,1                     ; cmpltps       %xmm0,%xmm9
-  DB  68,15,40,21,153,18,0,0              ; movaps        0x1299(%rip),%xmm10        # 5a20 <_sk_callback_sse2+0xee4>
+  DB  68,15,40,21,159,18,0,0              ; movaps        0x129f(%rip),%xmm10        # 5a50 <_sk_callback_sse2+0xeea>
   DB  69,15,92,208                        ; subps         %xmm8,%xmm10
   DB  69,15,84,209                        ; andps         %xmm9,%xmm10
   DB  69,15,85,200                        ; andnps        %xmm8,%xmm9
@@ -22284,7 +22318,7 @@ _sk_xy_to_radius_sse2 LABEL PROC
 PUBLIC _sk_save_xy_sse2
 _sk_save_xy_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,107,18,0,0               ; movaps        0x126b(%rip),%xmm8        # 5a30 <_sk_callback_sse2+0xef4>
+  DB  68,15,40,5,113,18,0,0               ; movaps        0x1271(%rip),%xmm8        # 5a60 <_sk_callback_sse2+0xefa>
   DB  15,17,0                             ; movups        %xmm0,(%rax)
   DB  68,15,40,200                        ; movaps        %xmm0,%xmm9
   DB  69,15,88,200                        ; addps         %xmm8,%xmm9
@@ -22292,7 +22326,7 @@ _sk_save_xy_sse2 LABEL PROC
   DB  69,15,91,210                        ; cvtdq2ps      %xmm10,%xmm10
   DB  69,15,40,217                        ; movaps        %xmm9,%xmm11
   DB  69,15,194,218,1                     ; cmpltps       %xmm10,%xmm11
-  DB  68,15,40,37,86,18,0,0               ; movaps        0x1256(%rip),%xmm12        # 5a40 <_sk_callback_sse2+0xf04>
+  DB  68,15,40,37,92,18,0,0               ; movaps        0x125c(%rip),%xmm12        # 5a70 <_sk_callback_sse2+0xf0a>
   DB  69,15,84,220                        ; andps         %xmm12,%xmm11
   DB  69,15,92,211                        ; subps         %xmm11,%xmm10
   DB  69,15,92,202                        ; subps         %xmm10,%xmm9
@@ -22335,8 +22369,8 @@ _sk_bilinear_nx_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,207,17,0,0                  ; addps         0x11cf(%rip),%xmm0        # 5a50 <_sk_callback_sse2+0xf14>
-  DB  68,15,40,13,215,17,0,0              ; movaps        0x11d7(%rip),%xmm9        # 5a60 <_sk_callback_sse2+0xf24>
+  DB  15,88,5,213,17,0,0                  ; addps         0x11d5(%rip),%xmm0        # 5a80 <_sk_callback_sse2+0xf1a>
+  DB  68,15,40,13,221,17,0,0              ; movaps        0x11dd(%rip),%xmm9        # 5a90 <_sk_callback_sse2+0xf2a>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -22347,7 +22381,7 @@ _sk_bilinear_px_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,198,17,0,0                  ; addps         0x11c6(%rip),%xmm0        # 5a70 <_sk_callback_sse2+0xf34>
+  DB  15,88,5,204,17,0,0                  ; addps         0x11cc(%rip),%xmm0        # 5aa0 <_sk_callback_sse2+0xf3a>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -22357,8 +22391,8 @@ _sk_bilinear_ny_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,184,17,0,0                 ; addps         0x11b8(%rip),%xmm1        # 5a80 <_sk_callback_sse2+0xf44>
-  DB  68,15,40,13,192,17,0,0              ; movaps        0x11c0(%rip),%xmm9        # 5a90 <_sk_callback_sse2+0xf54>
+  DB  15,88,13,190,17,0,0                 ; addps         0x11be(%rip),%xmm1        # 5ab0 <_sk_callback_sse2+0xf4a>
+  DB  68,15,40,13,198,17,0,0              ; movaps        0x11c6(%rip),%xmm9        # 5ac0 <_sk_callback_sse2+0xf5a>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -22369,7 +22403,7 @@ _sk_bilinear_py_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,174,17,0,0                 ; addps         0x11ae(%rip),%xmm1        # 5aa0 <_sk_callback_sse2+0xf64>
+  DB  15,88,13,180,17,0,0                 ; addps         0x11b4(%rip),%xmm1        # 5ad0 <_sk_callback_sse2+0xf6a>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -22379,13 +22413,13 @@ _sk_bicubic_n3x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,161,17,0,0                  ; addps         0x11a1(%rip),%xmm0        # 5ab0 <_sk_callback_sse2+0xf74>
-  DB  68,15,40,13,169,17,0,0              ; movaps        0x11a9(%rip),%xmm9        # 5ac0 <_sk_callback_sse2+0xf84>
+  DB  15,88,5,167,17,0,0                  ; addps         0x11a7(%rip),%xmm0        # 5ae0 <_sk_callback_sse2+0xf7a>
+  DB  68,15,40,13,175,17,0,0              ; movaps        0x11af(%rip),%xmm9        # 5af0 <_sk_callback_sse2+0xf8a>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,165,17,0,0              ; mulps         0x11a5(%rip),%xmm9        # 5ad0 <_sk_callback_sse2+0xf94>
-  DB  68,15,88,13,173,17,0,0              ; addps         0x11ad(%rip),%xmm9        # 5ae0 <_sk_callback_sse2+0xfa4>
+  DB  68,15,89,13,171,17,0,0              ; mulps         0x11ab(%rip),%xmm9        # 5b00 <_sk_callback_sse2+0xf9a>
+  DB  68,15,88,13,179,17,0,0              ; addps         0x11b3(%rip),%xmm9        # 5b10 <_sk_callback_sse2+0xfaa>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,128,0,0,0              ; movups        %xmm9,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -22396,16 +22430,16 @@ _sk_bicubic_n1x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,156,17,0,0                  ; addps         0x119c(%rip),%xmm0        # 5af0 <_sk_callback_sse2+0xfb4>
-  DB  68,15,40,13,164,17,0,0              ; movaps        0x11a4(%rip),%xmm9        # 5b00 <_sk_callback_sse2+0xfc4>
+  DB  15,88,5,162,17,0,0                  ; addps         0x11a2(%rip),%xmm0        # 5b20 <_sk_callback_sse2+0xfba>
+  DB  68,15,40,13,170,17,0,0              ; movaps        0x11aa(%rip),%xmm9        # 5b30 <_sk_callback_sse2+0xfca>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,168,17,0,0               ; movaps        0x11a8(%rip),%xmm8        # 5b10 <_sk_callback_sse2+0xfd4>
+  DB  68,15,40,5,174,17,0,0               ; movaps        0x11ae(%rip),%xmm8        # 5b40 <_sk_callback_sse2+0xfda>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,172,17,0,0               ; addps         0x11ac(%rip),%xmm8        # 5b20 <_sk_callback_sse2+0xfe4>
+  DB  68,15,88,5,178,17,0,0               ; addps         0x11b2(%rip),%xmm8        # 5b50 <_sk_callback_sse2+0xfea>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,176,17,0,0               ; addps         0x11b0(%rip),%xmm8        # 5b30 <_sk_callback_sse2+0xff4>
+  DB  68,15,88,5,182,17,0,0               ; addps         0x11b6(%rip),%xmm8        # 5b60 <_sk_callback_sse2+0xffa>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,180,17,0,0               ; addps         0x11b4(%rip),%xmm8        # 5b40 <_sk_callback_sse2+0x1004>
+  DB  68,15,88,5,186,17,0,0               ; addps         0x11ba(%rip),%xmm8        # 5b70 <_sk_callback_sse2+0x100a>
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -22413,17 +22447,17 @@ _sk_bicubic_n1x_sse2 LABEL PROC
 PUBLIC _sk_bicubic_p1x_sse2
 _sk_bicubic_p1x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,174,17,0,0               ; movaps        0x11ae(%rip),%xmm8        # 5b50 <_sk_callback_sse2+0x1014>
+  DB  68,15,40,5,180,17,0,0               ; movaps        0x11b4(%rip),%xmm8        # 5b80 <_sk_callback_sse2+0x101a>
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,72,64                      ; movups        0x40(%rax),%xmm9
   DB  65,15,88,192                        ; addps         %xmm8,%xmm0
-  DB  68,15,40,21,170,17,0,0              ; movaps        0x11aa(%rip),%xmm10        # 5b60 <_sk_callback_sse2+0x1024>
+  DB  68,15,40,21,176,17,0,0              ; movaps        0x11b0(%rip),%xmm10        # 5b90 <_sk_callback_sse2+0x102a>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,174,17,0,0              ; addps         0x11ae(%rip),%xmm10        # 5b70 <_sk_callback_sse2+0x1034>
+  DB  68,15,88,21,180,17,0,0              ; addps         0x11b4(%rip),%xmm10        # 5ba0 <_sk_callback_sse2+0x103a>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,170,17,0,0              ; addps         0x11aa(%rip),%xmm10        # 5b80 <_sk_callback_sse2+0x1044>
+  DB  68,15,88,21,176,17,0,0              ; addps         0x11b0(%rip),%xmm10        # 5bb0 <_sk_callback_sse2+0x104a>
   DB  68,15,17,144,128,0,0,0              ; movups        %xmm10,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -22433,11 +22467,11 @@ _sk_bicubic_p3x_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,0                             ; movups        (%rax),%xmm0
   DB  68,15,16,64,64                      ; movups        0x40(%rax),%xmm8
-  DB  15,88,5,157,17,0,0                  ; addps         0x119d(%rip),%xmm0        # 5b90 <_sk_callback_sse2+0x1054>
+  DB  15,88,5,163,17,0,0                  ; addps         0x11a3(%rip),%xmm0        # 5bc0 <_sk_callback_sse2+0x105a>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,157,17,0,0               ; mulps         0x119d(%rip),%xmm8        # 5ba0 <_sk_callback_sse2+0x1064>
-  DB  68,15,88,5,165,17,0,0               ; addps         0x11a5(%rip),%xmm8        # 5bb0 <_sk_callback_sse2+0x1074>
+  DB  68,15,89,5,163,17,0,0               ; mulps         0x11a3(%rip),%xmm8        # 5bd0 <_sk_callback_sse2+0x106a>
+  DB  68,15,88,5,171,17,0,0               ; addps         0x11ab(%rip),%xmm8        # 5be0 <_sk_callback_sse2+0x107a>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,128,0,0,0              ; movups        %xmm8,0x80(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -22448,13 +22482,13 @@ _sk_bicubic_n3y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,147,17,0,0                 ; addps         0x1193(%rip),%xmm1        # 5bc0 <_sk_callback_sse2+0x1084>
-  DB  68,15,40,13,155,17,0,0              ; movaps        0x119b(%rip),%xmm9        # 5bd0 <_sk_callback_sse2+0x1094>
+  DB  15,88,13,153,17,0,0                 ; addps         0x1199(%rip),%xmm1        # 5bf0 <_sk_callback_sse2+0x108a>
+  DB  68,15,40,13,161,17,0,0              ; movaps        0x11a1(%rip),%xmm9        # 5c00 <_sk_callback_sse2+0x109a>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
   DB  69,15,40,193                        ; movaps        %xmm9,%xmm8
   DB  69,15,89,192                        ; mulps         %xmm8,%xmm8
-  DB  68,15,89,13,151,17,0,0              ; mulps         0x1197(%rip),%xmm9        # 5be0 <_sk_callback_sse2+0x10a4>
-  DB  68,15,88,13,159,17,0,0              ; addps         0x119f(%rip),%xmm9        # 5bf0 <_sk_callback_sse2+0x10b4>
+  DB  68,15,89,13,157,17,0,0              ; mulps         0x119d(%rip),%xmm9        # 5c10 <_sk_callback_sse2+0x10aa>
+  DB  68,15,88,13,165,17,0,0              ; addps         0x11a5(%rip),%xmm9        # 5c20 <_sk_callback_sse2+0x10ba>
   DB  69,15,89,200                        ; mulps         %xmm8,%xmm9
   DB  68,15,17,136,160,0,0,0              ; movups        %xmm9,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -22465,16 +22499,16 @@ _sk_bicubic_n1y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,141,17,0,0                 ; addps         0x118d(%rip),%xmm1        # 5c00 <_sk_callback_sse2+0x10c4>
-  DB  68,15,40,13,149,17,0,0              ; movaps        0x1195(%rip),%xmm9        # 5c10 <_sk_callback_sse2+0x10d4>
+  DB  15,88,13,147,17,0,0                 ; addps         0x1193(%rip),%xmm1        # 5c30 <_sk_callback_sse2+0x10ca>
+  DB  68,15,40,13,155,17,0,0              ; movaps        0x119b(%rip),%xmm9        # 5c40 <_sk_callback_sse2+0x10da>
   DB  69,15,92,200                        ; subps         %xmm8,%xmm9
-  DB  68,15,40,5,153,17,0,0               ; movaps        0x1199(%rip),%xmm8        # 5c20 <_sk_callback_sse2+0x10e4>
+  DB  68,15,40,5,159,17,0,0               ; movaps        0x119f(%rip),%xmm8        # 5c50 <_sk_callback_sse2+0x10ea>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,157,17,0,0               ; addps         0x119d(%rip),%xmm8        # 5c30 <_sk_callback_sse2+0x10f4>
+  DB  68,15,88,5,163,17,0,0               ; addps         0x11a3(%rip),%xmm8        # 5c60 <_sk_callback_sse2+0x10fa>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,161,17,0,0               ; addps         0x11a1(%rip),%xmm8        # 5c40 <_sk_callback_sse2+0x1104>
+  DB  68,15,88,5,167,17,0,0               ; addps         0x11a7(%rip),%xmm8        # 5c70 <_sk_callback_sse2+0x110a>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
-  DB  68,15,88,5,165,17,0,0               ; addps         0x11a5(%rip),%xmm8        # 5c50 <_sk_callback_sse2+0x1114>
+  DB  68,15,88,5,171,17,0,0               ; addps         0x11ab(%rip),%xmm8        # 5c80 <_sk_callback_sse2+0x111a>
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -22482,17 +22516,17 @@ _sk_bicubic_n1y_sse2 LABEL PROC
 PUBLIC _sk_bicubic_p1y_sse2
 _sk_bicubic_p1y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
-  DB  68,15,40,5,159,17,0,0               ; movaps        0x119f(%rip),%xmm8        # 5c60 <_sk_callback_sse2+0x1124>
+  DB  68,15,40,5,165,17,0,0               ; movaps        0x11a5(%rip),%xmm8        # 5c90 <_sk_callback_sse2+0x112a>
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,72,96                      ; movups        0x60(%rax),%xmm9
   DB  65,15,88,200                        ; addps         %xmm8,%xmm1
-  DB  68,15,40,21,154,17,0,0              ; movaps        0x119a(%rip),%xmm10        # 5c70 <_sk_callback_sse2+0x1134>
+  DB  68,15,40,21,160,17,0,0              ; movaps        0x11a0(%rip),%xmm10        # 5ca0 <_sk_callback_sse2+0x113a>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,158,17,0,0              ; addps         0x119e(%rip),%xmm10        # 5c80 <_sk_callback_sse2+0x1144>
+  DB  68,15,88,21,164,17,0,0              ; addps         0x11a4(%rip),%xmm10        # 5cb0 <_sk_callback_sse2+0x114a>
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
   DB  69,15,88,208                        ; addps         %xmm8,%xmm10
   DB  69,15,89,209                        ; mulps         %xmm9,%xmm10
-  DB  68,15,88,21,154,17,0,0              ; addps         0x119a(%rip),%xmm10        # 5c90 <_sk_callback_sse2+0x1154>
+  DB  68,15,88,21,160,17,0,0              ; addps         0x11a0(%rip),%xmm10        # 5cc0 <_sk_callback_sse2+0x115a>
   DB  68,15,17,144,160,0,0,0              ; movups        %xmm10,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  255,224                             ; jmpq          *%rax
@@ -22502,11 +22536,11 @@ _sk_bicubic_p3y_sse2 LABEL PROC
   DB  72,173                              ; lods          %ds:(%rsi),%rax
   DB  15,16,72,32                         ; movups        0x20(%rax),%xmm1
   DB  68,15,16,64,96                      ; movups        0x60(%rax),%xmm8
-  DB  15,88,13,140,17,0,0                 ; addps         0x118c(%rip),%xmm1        # 5ca0 <_sk_callback_sse2+0x1164>
+  DB  15,88,13,146,17,0,0                 ; addps         0x1192(%rip),%xmm1        # 5cd0 <_sk_callback_sse2+0x116a>
   DB  69,15,40,200                        ; movaps        %xmm8,%xmm9
   DB  69,15,89,201                        ; mulps         %xmm9,%xmm9
-  DB  68,15,89,5,140,17,0,0               ; mulps         0x118c(%rip),%xmm8        # 5cb0 <_sk_callback_sse2+0x1174>
-  DB  68,15,88,5,148,17,0,0               ; addps         0x1194(%rip),%xmm8        # 5cc0 <_sk_callback_sse2+0x1184>
+  DB  68,15,89,5,146,17,0,0               ; mulps         0x1192(%rip),%xmm8        # 5ce0 <_sk_callback_sse2+0x117a>
+  DB  68,15,88,5,154,17,0,0               ; addps         0x119a(%rip),%xmm8        # 5cf0 <_sk_callback_sse2+0x118a>
   DB  69,15,89,193                        ; mulps         %xmm9,%xmm8
   DB  68,15,17,128,160,0,0,0              ; movups        %xmm8,0xa0(%rax)
   DB  72,173                              ; lods          %ds:(%rsi),%rax
@@ -22711,11 +22745,11 @@ ALIGN 16
   DB  128,191,0,0,128,191,0               ; cmpb          $0x0,-0x40800000(%rdi)
   DB  0,224                               ; add           %ah,%al
   DB  64,0,0                              ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4dc8 <.literal16+0x1d8>
+  DB  224,64                              ; loopne        4df8 <.literal16+0x1d8>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4dcc <.literal16+0x1dc>
+  DB  224,64                              ; loopne        4dfc <.literal16+0x1dc>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,64                              ; loopne        4dd0 <.literal16+0x1e0>
+  DB  224,64                              ; loopne        4e00 <.literal16+0x1e0>
   DB  154                                 ; (bad)
   DB  153                                 ; cltd
   DB  153                                 ; cltd
@@ -22735,13 +22769,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4df1 <.literal16+0x201>
+  DB  71,225,61                           ; rex.RXB       loope 4e21 <.literal16+0x201>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4df5 <.literal16+0x205>
+  DB  71,225,61                           ; rex.RXB       loope 4e25 <.literal16+0x205>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4df9 <.literal16+0x209>
+  DB  71,225,61                           ; rex.RXB       loope 4e29 <.literal16+0x209>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4dfd <.literal16+0x20d>
+  DB  71,225,61                           ; rex.RXB       loope 4e2d <.literal16+0x20d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -22766,13 +22800,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e31 <.literal16+0x241>
+  DB  71,225,61                           ; rex.RXB       loope 4e61 <.literal16+0x241>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e35 <.literal16+0x245>
+  DB  71,225,61                           ; rex.RXB       loope 4e65 <.literal16+0x245>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e39 <.literal16+0x249>
+  DB  71,225,61                           ; rex.RXB       loope 4e69 <.literal16+0x249>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e3d <.literal16+0x24d>
+  DB  71,225,61                           ; rex.RXB       loope 4e6d <.literal16+0x24d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -22797,13 +22831,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e71 <.literal16+0x281>
+  DB  71,225,61                           ; rex.RXB       loope 4ea1 <.literal16+0x281>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e75 <.literal16+0x285>
+  DB  71,225,61                           ; rex.RXB       loope 4ea5 <.literal16+0x285>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e79 <.literal16+0x289>
+  DB  71,225,61                           ; rex.RXB       loope 4ea9 <.literal16+0x289>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4e7d <.literal16+0x28d>
+  DB  71,225,61                           ; rex.RXB       loope 4ead <.literal16+0x28d>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -22828,13 +22862,13 @@ ALIGN 16
   DB  10,23                               ; or            (%rdi),%dl
   DB  63                                  ; (bad)
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4eb1 <.literal16+0x2c1>
+  DB  71,225,61                           ; rex.RXB       loope 4ee1 <.literal16+0x2c1>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4eb5 <.literal16+0x2c5>
+  DB  71,225,61                           ; rex.RXB       loope 4ee5 <.literal16+0x2c5>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4eb9 <.literal16+0x2c9>
+  DB  71,225,61                           ; rex.RXB       loope 4ee9 <.literal16+0x2c9>
   DB  174                                 ; scas          %es:(%rdi),%al
-  DB  71,225,61                           ; rex.RXB       loope 4ebd <.literal16+0x2cd>
+  DB  71,225,61                           ; rex.RXB       loope 4eed <.literal16+0x2cd>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -23063,13 +23097,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5099 <.literal16+0x4a9>
+  DB  224,7                               ; loopne        50c9 <.literal16+0x4a9>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        509d <.literal16+0x4ad>
+  DB  224,7                               ; loopne        50cd <.literal16+0x4ad>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        50a1 <.literal16+0x4b1>
+  DB  224,7                               ; loopne        50d1 <.literal16+0x4b1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        50a5 <.literal16+0x4b5>
+  DB  224,7                               ; loopne        50d5 <.literal16+0x4b5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -23134,11 +23168,11 @@ ALIGN 16
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            517b <.literal16+0x58b>
+  DB  127,67                              ; jg            51ab <.literal16+0x58b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            517f <.literal16+0x58f>
+  DB  127,67                              ; jg            51af <.literal16+0x58f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            5183 <.literal16+0x593>
+  DB  127,67                              ; jg            51b3 <.literal16+0x593>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,129,128,128,59           ; addb          $0x3b,-0x7f7f7ec5(%rax)
@@ -23153,16 +23187,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5174 <.literal16+0x584>
+  DB  127,0                               ; jg            51a4 <.literal16+0x584>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5178 <.literal16+0x588>
+  DB  127,0                               ; jg            51a8 <.literal16+0x588>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            517c <.literal16+0x58c>
+  DB  127,0                               ; jg            51ac <.literal16+0x58c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5180 <.literal16+0x590>
+  DB  127,0                               ; jg            51b0 <.literal16+0x590>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -23171,7 +23205,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5205 <.literal16+0x615>
+  DB  119,115                             ; ja            5235 <.literal16+0x615>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -23182,7 +23216,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           5169 <.literal16+0x579>
+  DB  117,191                             ; jne           5199 <.literal16+0x579>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -23194,7 +23228,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a391aa <_sk_callback_sse2+0xffffffffe9a3466e>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a391da <_sk_callback_sse2+0xffffffffe9a34674>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -23248,16 +23282,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5244 <.literal16+0x654>
+  DB  127,0                               ; jg            5274 <.literal16+0x654>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5248 <.literal16+0x658>
+  DB  127,0                               ; jg            5278 <.literal16+0x658>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            524c <.literal16+0x65c>
+  DB  127,0                               ; jg            527c <.literal16+0x65c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5250 <.literal16+0x660>
+  DB  127,0                               ; jg            5280 <.literal16+0x660>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -23266,7 +23300,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            52d5 <.literal16+0x6e5>
+  DB  119,115                             ; ja            5305 <.literal16+0x6e5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -23277,7 +23311,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           5239 <.literal16+0x649>
+  DB  117,191                             ; jne           5269 <.literal16+0x649>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -23289,7 +23323,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3927a <_sk_callback_sse2+0xffffffffe9a3473e>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a392aa <_sk_callback_sse2+0xffffffffe9a34744>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -23343,16 +23377,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5314 <.literal16+0x724>
+  DB  127,0                               ; jg            5344 <.literal16+0x724>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5318 <.literal16+0x728>
+  DB  127,0                               ; jg            5348 <.literal16+0x728>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            531c <.literal16+0x72c>
+  DB  127,0                               ; jg            534c <.literal16+0x72c>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            5320 <.literal16+0x730>
+  DB  127,0                               ; jg            5350 <.literal16+0x730>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -23361,7 +23395,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            53a5 <.literal16+0x7b5>
+  DB  119,115                             ; ja            53d5 <.literal16+0x7b5>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -23372,7 +23406,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           5309 <.literal16+0x719>
+  DB  117,191                             ; jne           5339 <.literal16+0x719>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -23384,7 +23418,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3934a <_sk_callback_sse2+0xffffffffe9a3480e>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3937a <_sk_callback_sse2+0xffffffffe9a34814>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -23438,16 +23472,16 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  52,255                              ; xor           $0xff,%al
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            53e4 <.literal16+0x7f4>
+  DB  127,0                               ; jg            5414 <.literal16+0x7f4>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            53e8 <.literal16+0x7f8>
+  DB  127,0                               ; jg            5418 <.literal16+0x7f8>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            53ec <.literal16+0x7fc>
+  DB  127,0                               ; jg            541c <.literal16+0x7fc>
   DB  255                                 ; (bad)
   DB  255                                 ; (bad)
-  DB  127,0                               ; jg            53f0 <.literal16+0x800>
+  DB  127,0                               ; jg            5420 <.literal16+0x800>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -23456,7 +23490,7 @@ ALIGN 16
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
-  DB  119,115                             ; ja            5475 <.literal16+0x885>
+  DB  119,115                             ; ja            54a5 <.literal16+0x885>
   DB  248                                 ; clc
   DB  194,119,115                         ; retq          $0x7377
   DB  248                                 ; clc
@@ -23467,7 +23501,7 @@ ALIGN 16
   DB  194,117,191                         ; retq          $0xbf75
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
-  DB  117,191                             ; jne           53d9 <.literal16+0x7e9>
+  DB  117,191                             ; jne           5409 <.literal16+0x7e9>
   DB  191,63,117,191,191                  ; mov           $0xbfbf753f,%edi
   DB  63                                  ; (bad)
   DB  249                                 ; stc
@@ -23479,7 +23513,7 @@ ALIGN 16
   DB  249                                 ; stc
   DB  68,180,62                           ; rex.R         mov $0x3e,%spl
   DB  163,233,220,63,163,233,220,63,163   ; movabs        %eax,0xa33fdce9a33fdce9
-  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3941a <_sk_callback_sse2+0xffffffffe9a348de>
+  DB  233,220,63,163,233                  ; jmpq          ffffffffe9a3944a <_sk_callback_sse2+0xffffffffe9a348e4>
   DB  220,63                              ; fdivrl        (%rdi)
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
@@ -23529,13 +23563,13 @@ ALIGN 16
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
   DB  200,66,0,0                          ; enterq        $0x42,$0x0
-  DB  127,67                              ; jg            54f7 <.literal16+0x907>
+  DB  127,67                              ; jg            5527 <.literal16+0x907>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            54fb <.literal16+0x90b>
+  DB  127,67                              ; jg            552b <.literal16+0x90b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            54ff <.literal16+0x90f>
+  DB  127,67                              ; jg            552f <.literal16+0x90f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            5503 <.literal16+0x913>
+  DB  127,67                              ; jg            5533 <.literal16+0x913>
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,195                               ; add           %al,%bl
   DB  0,0                                 ; add           %al,(%rax)
@@ -23582,16 +23616,16 @@ ALIGN 16
   DB  128,3,62                            ; addb          $0x3e,(%rbx)
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5583 <.literal16+0x993>
+  DB  118,63                              ; jbe           55b3 <.literal16+0x993>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           5587 <.literal16+0x997>
+  DB  118,63                              ; jbe           55b7 <.literal16+0x997>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           558b <.literal16+0x99b>
+  DB  118,63                              ; jbe           55bb <.literal16+0x99b>
   DB  31                                  ; (bad)
   DB  215                                 ; xlat          %ds:(%rbx)
-  DB  118,63                              ; jbe           558f <.literal16+0x99f>
+  DB  118,63                              ; jbe           55bf <.literal16+0x99f>
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
   DB  246,64,83,63                        ; testb         $0x3f,0x53(%rax)
@@ -23603,11 +23637,11 @@ ALIGN 16
   DB  128,59,0                            ; cmpb          $0x0,(%rbx)
   DB  0,127,67                            ; add           %bh,0x43(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            55cb <.literal16+0x9db>
+  DB  127,67                              ; jg            55fb <.literal16+0x9db>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            55cf <.literal16+0x9df>
+  DB  127,67                              ; jg            55ff <.literal16+0x9df>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            55d3 <.literal16+0x9e3>
+  DB  127,67                              ; jg            5603 <.literal16+0x9e3>
   DB  129,128,128,59,129,128,128,59,129,128; addl          $0x80813b80,-0x7f7ec480(%rax)
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,0,0,128,63               ; addb          $0x3f,-0x7fffffc5(%rax)
@@ -23647,13 +23681,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5619 <.literal16+0xa29>
+  DB  224,7                               ; loopne        5649 <.literal16+0xa29>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        561d <.literal16+0xa2d>
+  DB  224,7                               ; loopne        564d <.literal16+0xa2d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5621 <.literal16+0xa31>
+  DB  224,7                               ; loopne        5651 <.literal16+0xa31>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5625 <.literal16+0xa35>
+  DB  224,7                               ; loopne        5655 <.literal16+0xa35>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -23699,13 +23733,13 @@ ALIGN 16
   DB  132,55                              ; test          %dh,(%rdi)
   DB  8,33                                ; or            %ah,(%rcx)
   DB  132,55                              ; test          %dh,(%rdi)
-  DB  224,7                               ; loopne        5689 <.literal16+0xa99>
+  DB  224,7                               ; loopne        56b9 <.literal16+0xa99>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        568d <.literal16+0xa9d>
+  DB  224,7                               ; loopne        56bd <.literal16+0xa9d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5691 <.literal16+0xaa1>
+  DB  224,7                               ; loopne        56c1 <.literal16+0xaa1>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  224,7                               ; loopne        5695 <.literal16+0xaa5>
+  DB  224,7                               ; loopne        56c5 <.literal16+0xaa5>
   DB  0,0                                 ; add           %al,(%rax)
   DB  33,8                                ; and           %ecx,(%rax)
   DB  2,58                                ; add           (%rdx),%bh
@@ -23743,13 +23777,13 @@ ALIGN 16
   DB  65,0,0                              ; add           %al,(%r8)
   DB  248                                 ; clc
   DB  65,0,0                              ; add           %al,(%r8)
-  DB  124,66                              ; jl            5726 <.literal16+0xb36>
+  DB  124,66                              ; jl            5756 <.literal16+0xb36>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            572a <.literal16+0xb3a>
+  DB  124,66                              ; jl            575a <.literal16+0xb3a>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            572e <.literal16+0xb3e>
+  DB  124,66                              ; jl            575e <.literal16+0xb3e>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  124,66                              ; jl            5732 <.literal16+0xb42>
+  DB  124,66                              ; jl            5762 <.literal16+0xb42>
   DB  0,240                               ; add           %dh,%al
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,240                               ; add           %dh,%al
@@ -23839,13 +23873,13 @@ ALIGN 16
   DB  136,136,61,137,136,136              ; mov           %cl,-0x777776c3(%rax)
   DB  61,137,136,136,61                   ; cmp           $0x3d888889,%eax
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5835 <.literal16+0xc45>
+  DB  112,65                              ; jo            5865 <.literal16+0xc45>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5839 <.literal16+0xc49>
+  DB  112,65                              ; jo            5869 <.literal16+0xc49>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            583d <.literal16+0xc4d>
+  DB  112,65                              ; jo            586d <.literal16+0xc4d>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  112,65                              ; jo            5841 <.literal16+0xc51>
+  DB  112,65                              ; jo            5871 <.literal16+0xc51>
   DB  255,0                               ; incl          (%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  255,0                               ; incl          (%rax)
@@ -23867,11 +23901,11 @@ ALIGN 16
   DB  128,59,129                          ; cmpb          $0x81,(%rbx)
   DB  128,128,59,0,0,127,67               ; addb          $0x43,0x7f00003b(%rax)
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            588b <.literal16+0xc9b>
+  DB  127,67                              ; jg            58bb <.literal16+0xc9b>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            588f <.literal16+0xc9f>
+  DB  127,67                              ; jg            58bf <.literal16+0xc9f>
   DB  0,0                                 ; add           %al,(%rax)
-  DB  127,67                              ; jg            5893 <.literal16+0xca3>
+  DB  127,67                              ; jg            58c3 <.literal16+0xca3>
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,128,0,0,0,128                     ; add           %al,-0x80000000(%rax)
@@ -23947,13 +23981,13 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  255                                 ; (bad)
-  DB  127,71                              ; jg            597b <.literal16+0xd8b>
+  DB  127,71                              ; jg            59ab <.literal16+0xd8b>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            597f <.literal16+0xd8f>
+  DB  127,71                              ; jg            59af <.literal16+0xd8f>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            5983 <.literal16+0xd93>
+  DB  127,71                              ; jg            59b3 <.literal16+0xd93>
   DB  0,255                               ; add           %bh,%bh
-  DB  127,71                              ; jg            5987 <.literal16+0xd97>
+  DB  127,71                              ; jg            59b7 <.literal16+0xd97>
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,0                            ; cmpb          $0x0,(%rdi)
   DB  0,128,63,0,0,128                    ; add           %al,-0x7fffffc1(%rax)
@@ -24114,11 +24148,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         5af2 <.literal16+0xf02>
+  DB  62,114,28                           ; jb,pt         5b22 <.literal16+0xf02>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5af6 <.literal16+0xf06>
+  DB  62,114,28                           ; jb,pt         5b26 <.literal16+0xf06>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5afa <.literal16+0xf0a>
+  DB  62,114,28                           ; jb,pt         5b2a <.literal16+0xf0a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -24162,7 +24196,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e985 <_sk_callback_sse2+0x3d639e49>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e9b5 <_sk_callback_sse2+0x3d639e4f>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -24188,7 +24222,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e9c5 <_sk_callback_sse2+0x3d639e89>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63e9f5 <_sk_callback_sse2+0x3d639e8f>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -24197,13 +24231,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            5bbe <.literal16+0xfce>
+  DB  114,28                              ; jb            5bee <.literal16+0xfce>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5bc2 <.literal16+0xfd2>
+  DB  62,114,28                           ; jb,pt         5bf2 <.literal16+0xfd2>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5bc6 <.literal16+0xfd6>
+  DB  62,114,28                           ; jb,pt         5bf6 <.literal16+0xfd6>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5bca <.literal16+0xfda>
+  DB  62,114,28                           ; jb,pt         5bfa <.literal16+0xfda>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -24224,11 +24258,11 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  128,63,114                          ; cmpb          $0x72,(%rdi)
   DB  28,199                              ; sbb           $0xc7,%al
-  DB  62,114,28                           ; jb,pt         5c02 <.literal16+0x1012>
+  DB  62,114,28                           ; jb,pt         5c32 <.literal16+0x1012>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5c06 <.literal16+0x1016>
+  DB  62,114,28                           ; jb,pt         5c36 <.literal16+0x1016>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5c0a <.literal16+0x101a>
+  DB  62,114,28                           ; jb,pt         5c3a <.literal16+0x101a>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
@@ -24272,7 +24306,7 @@ ALIGN 16
   DB  0,0                                 ; add           %al,(%rax)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63ea95 <_sk_callback_sse2+0x3d639f59>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63eac5 <_sk_callback_sse2+0x3d639f5f>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  0,63                                ; add           %bh,(%rdi)
   DB  0,0                                 ; add           %al,(%rax)
@@ -24298,7 +24332,7 @@ ALIGN 16
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
   DB  57,142,99,61,57,142                 ; cmp           %ecx,-0x71c6c29d(%rsi)
-  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63ead5 <_sk_callback_sse2+0x3d639f99>
+  DB  99,61,57,142,99,61                  ; movslq        0x3d638e39(%rip),%edi        # 3d63eb05 <_sk_callback_sse2+0x3d639f9f>
   DB  57,142,99,61,0,0                    ; cmp           %ecx,0x3d63(%rsi)
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
@@ -24307,13 +24341,13 @@ ALIGN 16
   DB  192,63,0                            ; sarb          $0x0,(%rdi)
   DB  0,192                               ; add           %al,%al
   DB  63                                  ; (bad)
-  DB  114,28                              ; jb            5cce <.literal16+0x10de>
+  DB  114,28                              ; jb            5cfe <.literal16+0x10de>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5cd2 <_sk_callback_sse2+0x1196>
+  DB  62,114,28                           ; jb,pt         5d02 <_sk_callback_sse2+0x119c>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5cd6 <_sk_callback_sse2+0x119a>
+  DB  62,114,28                           ; jb,pt         5d06 <_sk_callback_sse2+0x11a0>
   DB  199                                 ; (bad)
-  DB  62,114,28                           ; jb,pt         5cda <_sk_callback_sse2+0x119e>
+  DB  62,114,28                           ; jb,pt         5d0a <_sk_callback_sse2+0x11a4>
   DB  199                                 ; (bad)
   DB  62,171                              ; ds            stos %eax,%es:(%rdi)
   DB  170                                 ; stos          %al,%es:(%rdi)
index a486403..aa161e9 100644 (file)
@@ -320,6 +320,10 @@ STAGE(dither) {
     r += c->rate*dither;
     g += c->rate*dither;
     b += c->rate*dither;
+
+    r = max(0, min(r, a));
+    g = max(0, min(g, a));
+    b = max(0, min(b, a));
 }
 
 // load 4 floats from memory, and splat them into r,g,b,a