Commit 7b4c3a120fa for php

commit 7b4c3a120fade89d54e2396365ff3acea30b951d
Author: Dmitry Stogov <dmitry@php.net>
Date:   Tue Sep 29 13:46:20 2026 +0300

    Update IR (#23861)

    IR commit: 00bec1ca490b8bc51f636a42040c5e9f64c1a91b

diff --git a/ext/opcache/jit/ir/dynasm/dasm_arm64.h b/ext/opcache/jit/ir/dynasm/dasm_arm64.h
index 1257fb2011e..8a22974ad3f 100644
--- a/ext/opcache/jit/ir/dynasm/dasm_arm64.h
+++ b/ext/opcache/jit/ir/dynasm/dasm_arm64.h
@@ -152,14 +152,17 @@ void dasm_setup(Dst_DECL, const void *actionlist)
   }
 }

+#ifndef DASM_ABORT
+# define DASM_ABORT
+#endif

 #ifdef DASM_CHECKS
 #define CK(x, st) \
   do { if (!(x)) { \
-    D->status = DASM_S_##st|(int)(p-D->actionlist-1); return; } } while (0)
+    D->status = DASM_S_##st|(int)(p-D->actionlist-1); DASM_ABORT; return; } } while (0)
 #define CKPL(kind, st) \
   do { if ((size_t)((char *)pl-(char *)D->kind##labels) >= D->kind##size) { \
-    D->status = DASM_S_RANGE_##st|(int)(p-D->actionlist-1); return; } } while (0)
+    D->status = DASM_S_RANGE_##st|(int)(p-D->actionlist-1); DASM_ABORT; return; } } while (0)
 #else
 #define CK(x, st)	((void)0)
 #define CKPL(kind, st)	((void)0)
diff --git a/ext/opcache/jit/ir/dynasm/dasm_arm64.lua b/ext/opcache/jit/ir/dynasm/dasm_arm64.lua
index 05ea3e228cc..501c4bd2abf 100644
--- a/ext/opcache/jit/ir/dynasm/dasm_arm64.lua
+++ b/ext/opcache/jit/ir/dynasm/dasm_arm64.lua
@@ -801,8 +801,8 @@ map_op = {
   ["strh_*"]  = "78000000DwL",
   ["ldrh_*"]  = "78400000DwL",
   ["ldrsh_*"] = "78c00000DwL|78800000DxL",
-  ["str_*"]   = "b8000000DwL|f8000000DxL|bc000000DsL|fc000000DdL",
-  ["ldr_*"]   = "18000000DwB|58000000DxB|1c000000DsB|5c000000DdB|b8400000DwL|f8400000DxL|bc400000DsL|fc400000DdL",
+  ["str_*"]   = "b8000000DwL|f8000000DxL|bc000000DsL|fc000000DdL|3c800000DqL",
+  ["ldr_*"]   = "18000000DwB|58000000DxB|1c000000DsB|5c000000DdB|9c000000DqB|b8400000DwL|f8400000DxL|bc400000DsL|fc400000DdL|3cc00000DqL",
   ["ldrsw_*"] = "98000000DxB|b8800000DxL",
   -- NOTE: ldur etc. are handled by ldr et al.

diff --git a/ext/opcache/jit/ir/dynasm/dasm_x86.h b/ext/opcache/jit/ir/dynasm/dasm_x86.h
index 13fb74ccbe7..d50299b13c1 100644
--- a/ext/opcache/jit/ir/dynasm/dasm_x86.h
+++ b/ext/opcache/jit/ir/dynasm/dasm_x86.h
@@ -148,14 +148,17 @@ void dasm_setup(Dst_DECL, const void *actionlist)
   }
 }

+#ifndef DASM_ABORT
+# define DASM_ABORT
+#endif

 #ifdef DASM_CHECKS
 #define CK(x, st) \
   do { if (!(x)) { \
-    D->status = DASM_S_##st|(int)(p-D->actionlist-1); return; } } while (0)
+    D->status = DASM_S_##st|(int)(p-D->actionlist-1); DASM_ABORT; return; } } while (0)
 #define CKPL(kind, st) \
   do { if ((size_t)((char *)pl-(char *)D->kind##labels) >= D->kind##size) { \
-    D->status=DASM_S_RANGE_##st|(int)(p-D->actionlist-1); return; } } while (0)
+    D->status=DASM_S_RANGE_##st|(int)(p-D->actionlist-1); DASM_ABORT; return; } } while (0)
 #else
 #define CK(x, st)	((void)0)
 #define CKPL(kind, st)	((void)0)
diff --git a/ext/opcache/jit/ir/dynasm/dasm_x86.lua b/ext/opcache/jit/ir/dynasm/dasm_x86.lua
index 7c789f8216d..0467bf91329 100644
--- a/ext/opcache/jit/ir/dynasm/dasm_x86.lua
+++ b/ext/opcache/jit/ir/dynasm/dasm_x86.lua
@@ -1310,7 +1310,7 @@ local map_op = {
   mfence_0 =	"0FAEF0",
   movapd_2 =	"rmo:660F28rM|mro:660F29Rm",
   movaps_2 =	"rmo:0F28rM|mro:0F29Rm",
-  movd_2 =	"rm/od:660F6ErM|rm/oq:660F6ErXM|mr/do:660F7ERm|mr/qo:",
+  movd_2 =	"rm/od:660F6ErM|mr/do:660F7ERm",
   movdqa_2 =	"rmo:660F6FrM|mro:660F7FRm",
   movdqu_2 =	"rmo:F30F6FrM|mro:F30F7FRm",
   movhlps_2 =	"rro:0F12rM",
@@ -1325,7 +1325,7 @@ local map_op = {
   movnti_2 =	"xrqd:0FC3Rm",
   movntpd_2 =	"xro:660F2BRm",
   movntps_2 =	"xro:0F2BRm",
-  movq_2 =	"rro:F30F7ErM|rx/oq:|xr/qo:n660FD6Rm",
+  movq_2 =	x64 and "rro:F30F7ErM|rx/oq:|xr/qo:n660FD6Rm|rm/oq:660F6ErXM|mr/qo:660F7ERm" or "rro:F30F7ErM|rx/oq:|xr/qo:n660FD6Rm",
   movsd_2 =	"rro:F20F10rM|rx/oq:|xr/qo:nF20F11Rm",
   movss_2 =	"rro:F30F10rM|rx/od:|xr/do:F30F11Rm",
   movupd_2 =	"rmo:660F10rM|mro:660F11Rm",
@@ -1409,7 +1409,7 @@ local map_op = {
   dppd_3 =	"rmio:660F3A41rMU",
   dpps_3 =	"rmio:660F3A40rMU",
   extractps_3 =	"mri/do:660F3A17RmU|rri/qo:660F3A17RXmU",
-  insertps_3 =	"rrio:660F3A41rMU|rxi/od:",
+  insertps_3 =	"rrio:660F3A21rMU|rxi/od:",
   movntdqa_2 =	"rxo:660F382ArM",
   mpsadbw_3 =	"rmio:660F3A42rMU",
   packusdw_2 =	"rmo:660F382BrM",
@@ -1529,8 +1529,8 @@ local map_op = {
   vmaskmovpd_3 = "rrxoy:660F38V2DrM|xrroy:660F38V2FRm",
   vmovapd_2 =	"rmoy:660Fu28rM|mroy:660Fu29Rm",
   vmovaps_2 =	"rmoy:0Fu28rM|mroy:0Fu29Rm",
-  vmovd_2 =	"rm/od:660Fu6ErM|rm/oq:660FuX6ErM|mr/do:660Fu7ERm|mr/qo:",
-  vmovq_2 =	"rro:F30Fu7ErM|rx/oq:|xr/qo:660FuD6Rm",
+  vmovd_2 =	"rm/od:660Fu6ErM|mr/do:660Fu7ERm",
+  vmovq_2 =	x64 and "rro:F30Fu7ErM|rx/oq:|xr/qo:660FuD6Rm|rm/oq:660FuX6ErM|mr/qo:660Fu7ERm" or "rro:F30Fu7ErM|rx/oq:|xr/qo:660FuD6Rm",
   vmovddup_2 =	"rmy:F20Fu12rM|rro:|rx/oq:",
   vmovhlps_3 =	"rrro:0FV12rM",
   vmovhpd_2 =	"xr/qo:660Fu17Rm",
diff --git a/ext/opcache/jit/ir/gen_ir_fold_hash.c b/ext/opcache/jit/ir/gen_ir_fold_hash.c
index 800da27bdfa..1490ba8794b 100644
--- a/ext/opcache/jit/ir/gen_ir_fold_hash.c
+++ b/ext/opcache/jit/ir/gen_ir_fold_hash.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Folding engine generator)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * Based on Mike Pall's implementation for LuaJIT.
diff --git a/ext/opcache/jit/ir/ir.c b/ext/opcache/jit/ir/ir.c
index f6a0cb60af9..120722b2dca 100644
--- a/ext/opcache/jit/ir/ir.c
+++ b/ext/opcache/jit/ir/ir.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (IR construction, folding, utilities)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * The logical IR representation is based on Cliff Click's Sea of Nodes.
@@ -42,7 +42,7 @@
 # include <valgrind/valgrind.h>
 #endif

-#define IR_TYPE_FLAGS(name, type, field, flags) ((flags)|sizeof(type)),
+#define IR_TYPE_FLAGS(name, type, field, flags) (flags),
 #define IR_TYPE_NAME(name, type, field, flags)  #name,
 #define IR_TYPE_CNAME(name, type, field, flags) #type,
 #define IR_TYPE_SIZE(name, type, field, flags)  sizeof(type),
@@ -114,10 +114,44 @@ void ir_print_escaped_str(const char *s, size_t len, FILE *f)
 	}
 }

-void ir_print_const(const ir_ctx *ctx, const ir_insn *insn, FILE *f, bool quoted)
+static void ir_print_double(double v, FILE *f)
 {
 	char buf[128];

+	if (isnan(v)) {
+		fprintf(f, "nan");
+	} else {
+		snprintf(buf, sizeof(buf), "%g", v);
+		if (strtod(buf, NULL) != v) {
+			snprintf(buf, sizeof(buf), "%.53e", v);
+			if (strtod(buf, NULL) != v) {
+				IR_ASSERT(0 && "can't format double");
+			}
+		}
+		fprintf(f, "%s", buf);
+	}
+}
+
+static void ir_print_float(float v, FILE *f)
+{
+	char buf[128];
+
+	if (isnan(v)) {
+		fprintf(f, "nan");
+	} else {
+		snprintf(buf, sizeof(buf), "%g", v);
+		if (strtod(buf, NULL) != v) {
+			snprintf(buf, sizeof(buf), "%.24e", v);
+			if (strtod(buf, NULL) != v) {
+				IR_ASSERT(0 && "can't format float");
+			}
+		}
+		fprintf(f, "%s", buf);
+	}
+}
+
+void ir_print_const(const ir_ctx *ctx, const ir_insn *insn, FILE *f, bool quoted)
+{
 	if (insn->op == IR_FUNC || insn->op == IR_SYM || insn->op == IR_LABEL) {
 		fprintf(f, "%s", ir_get_str(ctx, insn->val.name));
 		return;
@@ -134,6 +168,95 @@ void ir_print_const(const ir_ctx *ctx, const ir_insn *insn, FILE *f, bool quoted
 		}
 		return;
 	}
+
+	if (IR_IS_TYPE_VECTOR(insn->type)) {
+		ir_type t = IR_VECTOR_BASE_TYPE(insn->type);
+		uint32_t n = IR_VECTOR_LENGTH(insn->type);
+		const void *p = insn + 1;
+
+		fprintf(f, "{");
+		switch (t) {
+			case IR_I8:
+			case IR_CHAR:
+				fprintf(f, "%d", *(int8_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(int8_t);;
+					fprintf(f, ", %d", *(int8_t*)p);
+				}
+				break;
+			case IR_I16:
+				fprintf(f, "%d", *(int16_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(int16_t);
+					fprintf(f, ", %d", *(int16_t*)p);
+				}
+				break;
+			case IR_I32:
+				fprintf(f, "%d", *(int32_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(int32_t);
+					fprintf(f, ", %d", *(int32_t*)p);
+				}
+				break;
+			case IR_I64:
+				fprintf(f, "%" PRIi64, *(int64_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(int64_t);
+					fprintf(f, ", %" PRIi64, *(int64_t*)p);
+				}
+				break;
+			case IR_U8:
+				fprintf(f, "%u", *(uint8_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(uint8_t);
+					fprintf(f, ", %u", *(uint8_t*)p);
+				}
+				break;
+			case IR_U16:
+				fprintf(f, "%u", *(uint16_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(uint16_t);
+					fprintf(f, ", %u", *(uint16_t*)p);
+				}
+				break;
+			case IR_U32:
+				fprintf(f, "%u", *(uint32_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(uint32_t);
+					fprintf(f, ", %u", *(uint32_t*)p);
+				}
+				break;
+			case IR_U64:
+				fprintf(f, "%" PRIu64, *(uint64_t*)p);
+				while (--n) {
+					p = (char*)p + sizeof(uint64_t);
+					fprintf(f, ", %" PRIu64, *(uint64_t*)p);
+				}
+				break;
+			case IR_DOUBLE:
+				ir_print_double(*(double*)p, f);
+				while (--n) {
+					p = (char*)p + sizeof(double);
+					fprintf(f, ", ");
+					ir_print_double(*(double*)p, f);
+				}
+				break;
+			case IR_FLOAT:
+				ir_print_float(*(float*)p, f);
+				while (--n) {
+					p = (char*)p + sizeof(float);
+					fprintf(f, ", ");
+					ir_print_float(*(float*)p, f);
+				}
+				break;
+			default:
+				IR_ASSERT(0);
+				break;
+		}
+		fprintf(f, "}");
+		return;
+	}
+
 	IR_ASSERT(IR_IS_CONST_OP(insn->op) || insn->op == IR_FUNC_ADDR);
 	switch (insn->type) {
 		case IR_BOOL:
@@ -190,32 +313,10 @@ void ir_print_const(const ir_ctx *ctx, const ir_insn *insn, FILE *f, bool quoted
 			fprintf(f, "%" PRIi64, insn->val.i64);
 			break;
 		case IR_DOUBLE:
-			if (isnan(insn->val.d)) {
-				fprintf(f, "nan");
-			} else {
-				snprintf(buf, sizeof(buf), "%g", insn->val.d);
-				if (strtod(buf, NULL) != insn->val.d) {
-					snprintf(buf, sizeof(buf), "%.53e", insn->val.d);
-					if (strtod(buf, NULL) != insn->val.d) {
-						IR_ASSERT(0 && "can't format double");
-					}
-				}
-				fprintf(f, "%s", buf);
-			}
+			ir_print_double(insn->val.d, f);
 			break;
 		case IR_FLOAT:
-			if (isnan(insn->val.f)) {
-				fprintf(f, "nan");
-			} else {
-				snprintf(buf, sizeof(buf), "%g", insn->val.f);
-				if (strtod(buf, NULL) != insn->val.f) {
-					snprintf(buf, sizeof(buf), "%.24e", insn->val.f);
-					if (strtod(buf, NULL) != insn->val.f) {
-						IR_ASSERT(0 && "can't format float");
-					}
-				}
-				fprintf(f, "%s", buf);
-			}
+			ir_print_float(insn->val.f, f);
 			break;
 		default:
 			IR_ASSERT(0);
@@ -464,6 +565,9 @@ void ir_free(ir_ctx *ctx)
 	}
 	if (ctx->regs) {
 		ir_mem_free(ctx->regs);
+		if (ctx->tmp_regs) {
+			ir_mem_free(ctx->tmp_regs);
+		}
 		if (ctx->fused_regs) {
 			ir_strtab_free(ctx->fused_regs);
 			ir_mem_free(ctx->fused_regs);
@@ -500,7 +604,7 @@ ir_ref ir_unique_const_addr(ir_ctx *ctx, uintptr_t addr)

 IR_ALWAYS_INLINE uintptr_t ir_const_hash(ir_val val, uint32_t optx)
 {
-	return (val.u64 ^ (val.u64 >> 32) ^ optx);
+	return (uintptr_t)(val.u64 ^ (val.u64 >> 32) ^ optx);
 }

 static IR_NEVER_INLINE void ir_const_hash_rehash(ir_ctx *ctx)
@@ -514,11 +618,18 @@ static IR_NEVER_INLINE void ir_const_hash_rehash(ir_ctx *ctx)
 	}
 	ctx->const_hash_mask = (ctx->const_hash_mask + 1) * 2 - 1;
 	ctx->const_hash = ir_mem_calloc(ctx->const_hash_mask + 1, sizeof(ir_ref));
-	for (ref = IR_TRUE - 1; ref > -ctx->consts_count; ref--) {
-		insn = &ctx->ir_base[ref];
-		hash = ir_const_hash(insn->val, insn->optx) & ctx->const_hash_mask;
-		insn->prev_const = ctx->const_hash[hash];
-		ctx->const_hash[hash] = ref;
+	for (ref = 1 - ctx->consts_count, insn = ctx->ir_base + ref; ref < IR_TRUE; ref++, insn++) {
+		if (insn->op == IR_LONG_CONST) {
+			hash = insn->val.u64;
+			insn->prev_const = ctx->const_hash[hash & ctx->const_hash_mask];
+			ctx->const_hash[hash & ctx->const_hash_mask] = ref;
+			ref += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+			insn += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+		} else {
+			hash = ir_const_hash(insn->val, insn->optx) & ctx->const_hash_mask;
+			insn->prev_const = ctx->const_hash[hash];
+			ctx->const_hash[hash] = ref;
+		}
 	}
 }

@@ -660,7 +771,7 @@ ir_ref ir_const_addr(ir_ctx *ctx, uintptr_t c)
 	return ir_const(ctx, val, IR_ADDR);
 }

-ir_ref ir_const_func_addr(ir_ctx *ctx, uintptr_t c, ir_ref proto)
+ir_ref ir_const_func_addr(ir_ctx *ctx, uintptr_t c, ir_str proto)
 {
 	if (c == 0) {
 		return IR_NULL;
@@ -671,7 +782,7 @@ ir_ref ir_const_func_addr(ir_ctx *ctx, uintptr_t c, ir_ref proto)
 	return ir_const_ex(ctx, val, IR_ADDR, IR_OPTX(IR_FUNC_ADDR, IR_ADDR, proto));
 }

-ir_ref ir_const_func(ir_ctx *ctx, ir_ref str, ir_ref proto)
+ir_ref ir_const_func(ir_ctx *ctx, ir_str str, ir_str proto)
 {
 	ir_val val;
 	val.u64 = str;
@@ -679,28 +790,114 @@ ir_ref ir_const_func(ir_ctx *ctx, ir_ref str, ir_ref proto)
 	return ir_const_ex(ctx, val, IR_ADDR, IR_OPTX(IR_FUNC, IR_ADDR, proto));
 }

-ir_ref ir_const_sym(ir_ctx *ctx, ir_ref str)
+ir_ref ir_const_sym(ir_ctx *ctx, ir_str str)
 {
 	ir_val val;
 	val.u64 = str;
 	return ir_const_ex(ctx, val, IR_ADDR, IR_OPTX(IR_SYM, IR_ADDR, 0));
 }

-ir_ref ir_const_str(ir_ctx *ctx, ir_ref str)
+ir_ref ir_const_str(ir_ctx *ctx, ir_str str)
 {
 	ir_val val;
 	val.u64 = str;
 	return ir_const_ex(ctx, val, IR_ADDR, IR_OPTX(IR_STR, IR_ADDR, 0));
 }

-ir_ref ir_const_label(ir_ctx *ctx, ir_ref str)
+ir_ref ir_const_label(ir_ctx *ctx, ir_str str)
 {
 	ir_val val;
 	val.u64 = str;
 	return ir_const_ex(ctx, val, IR_ADDR, IR_OPTX(IR_LABEL, IR_ADDR, 0));
 }

-ir_ref ir_str(ir_ctx *ctx, const char *s)
+ir_ref ir_long_const(ir_ctx *ctx, ir_type type, size_t size)
+{
+	ir_ref ref = ctx->consts_count;
+	ir_insn *insn;
+
+	IR_ASSERT(size <= 0xfff0);
+
+	ref = ctx->consts_count + IR_ALIGNED_SIZE(size, sizeof(ir_insn)) / sizeof(ir_insn);
+	while (UNEXPECTED(ref >= ctx->consts_limit)) {
+		ir_grow_bottom(ctx);
+	}
+	ctx->consts_count = ref + 1;
+	ref = -ref;
+
+	insn = &ctx->ir_base[ref];
+	insn->optx = IR_OPTX(IR_LONG_CONST, type, size);
+	insn->op1 = IR_UNUSED;
+	insn->val.u64 = 0;
+
+	ctx->flags2 |= IR_HAS_LONG_CONSTANTS;
+
+	return ref;
+}
+
+void *ir_long_const_ptr(ir_ctx *ctx, ir_ref ref)
+{
+	IR_ASSERT(IR_IS_CONST_REF(ref));
+	return (void*)&ctx->ir_base[ref + 1];
+}
+
+IR_ALWAYS_INLINE uintptr_t ir_long_const_hash(uint32_t optx, const void *ptr, size_t len)
+{
+	size_t i;
+	const uint8_t *str = ptr;
+	uint32_t h = 5381;
+
+    for (i = 0; i < len; i++) {
+        h = ((h << 5) + h) + *str;
+        str++;
+    }
+	return (uintptr_t)(h ^ optx);
+}
+
+ir_ref ir_long_const_commit(ir_ctx *ctx, ir_ref const_ref)
+{
+	ir_ref ref;
+	uintptr_t hash, n;
+	ir_insn *insn = &ctx->ir_base[const_ref];
+	uint32_t optx = insn->optx;
+	size_t size = insn->long_const_size;
+	const void *ptr = insn + 1;
+
+	IR_ASSERT(ctx->consts_count == 1 - const_ref && "ir_long_const_commit() argument must be result of the last ir_long_const()");
+
+	/* check if we already have the same constant */
+	hash = ir_long_const_hash(optx, ptr, size);
+	ref = ctx->const_hash[hash & ctx->const_hash_mask];
+	while (ref) {
+		insn = &ctx->ir_base[ref];
+		if (insn->val.u64 == hash && insn->optx == optx && memcmp(ptr, insn + 1, size) == 0) {
+			/* rollback */
+			ctx->consts_count -= (IR_ALIGNED_SIZE(size, sizeof(ir_insn)) / sizeof(ir_insn)) + 1;
+			return ref;
+		}
+		ref = insn->prev_const;
+	}
+
+	if ((uintptr_t)ctx->consts_count > ctx->const_hash_mask) {
+		ir_const_hash_rehash(ctx);
+	}
+
+	n = hash & ctx->const_hash_mask;
+	insn = &ctx->ir_base[const_ref];
+	insn->prev_const = ctx->const_hash[n];
+	insn->val.u64 = hash;
+	ctx->const_hash[n] = const_ref;
+
+	return const_ref;
+}
+
+ir_ref ir_const_vector(ir_ctx *ctx, ir_type type)
+{
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	return ir_long_const(ctx, type, IR_VECTOR_SIZE(type));
+}
+
+ir_str ir_string(ir_ctx *ctx, const char *s)
 {
 	size_t len;

@@ -712,7 +909,7 @@ ir_ref ir_str(ir_ctx *ctx, const char *s)
 	return ir_strtab_lookup(&ctx->strtab, s, (uint32_t)len, ir_strtab_count(&ctx->strtab) + 1);
 }

-ir_ref ir_strl(ir_ctx *ctx, const char *s, size_t len)
+ir_str ir_stringl(ir_ctx *ctx, const char *s, size_t len)
 {
 	if (!ctx->strtab.data) {
 		ir_strtab_init(&ctx->strtab, 64, 4096);
@@ -721,29 +918,35 @@ ir_ref ir_strl(ir_ctx *ctx, const char *s, size_t len)
 	return ir_strtab_lookup(&ctx->strtab, s, (uint32_t)len, ir_strtab_count(&ctx->strtab) + 1);
 }

-const char *ir_get_str(const ir_ctx *ctx, ir_ref idx)
+const char *ir_get_str(const ir_ctx *ctx, ir_str idx)
 {
+	if (IR_IS_EXT_STR(idx)) {
+		return ctx->loader->get_str(ctx->loader, idx);
+	}
 	IR_ASSERT(ctx->strtab.data);
 	return ir_strtab_str(&ctx->strtab, idx - 1);
 }

-const char *ir_get_strl(const ir_ctx *ctx, ir_ref idx, size_t *len)
+const char *ir_get_strl(const ir_ctx *ctx, ir_str idx, size_t *len)
 {
+	if (IR_IS_EXT_STR(idx)) {
+		return ctx->loader->get_strl(ctx->loader, idx, len);
+	}
 	IR_ASSERT(ctx->strtab.data);
 	return ir_strtab_strl(&ctx->strtab, idx - 1, len);
 }

-ir_ref ir_proto_0(ir_ctx *ctx, uint8_t flags, ir_type ret_type)
+ir_str ir_proto_0(ir_ctx *ctx, uint8_t flags, ir_type ret_type)
 {
 	ir_proto_t proto;

 	proto.flags = flags;
 	proto.ret_type = ret_type;
 	proto.params_count = 0;
-	return ir_strl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 0);
+	return ir_stringl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 0);
 }

-ir_ref ir_proto_1(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1)
+ir_str ir_proto_1(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1)
 {
 	ir_proto_t proto;

@@ -751,10 +954,10 @@ ir_ref ir_proto_1(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1)
 	proto.ret_type = ret_type;
 	proto.params_count = 1;
 	proto.param_types[0] = t1;
-	return ir_strl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 1);
+	return ir_stringl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 1);
 }

-ir_ref ir_proto_2(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2)
+ir_str ir_proto_2(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2)
 {
 	ir_proto_t proto;

@@ -763,10 +966,10 @@ ir_ref ir_proto_2(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_t
 	proto.params_count = 2;
 	proto.param_types[0] = t1;
 	proto.param_types[1] = t2;
-	return ir_strl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 2);
+	return ir_stringl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 2);
 }

-ir_ref ir_proto_3(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3)
+ir_str ir_proto_3(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3)
 {
 	ir_proto_t proto;

@@ -776,10 +979,10 @@ ir_ref ir_proto_3(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_t
 	proto.param_types[0] = t1;
 	proto.param_types[1] = t2;
 	proto.param_types[2] = t3;
-	return ir_strl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 3);
+	return ir_stringl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 3);
 }

-ir_ref ir_proto_4(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
+ir_str ir_proto_4(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
                                                                 ir_type t4)
 {
 	ir_proto_t proto;
@@ -791,10 +994,10 @@ ir_ref ir_proto_4(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_t
 	proto.param_types[1] = t2;
 	proto.param_types[2] = t3;
 	proto.param_types[3] = t4;
-	return ir_strl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 4);
+	return ir_stringl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 4);
 }

-ir_ref ir_proto_5(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
+ir_str ir_proto_5(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
                                                                 ir_type t4, ir_type t5)
 {
 	ir_proto_t proto;
@@ -807,10 +1010,10 @@ ir_ref ir_proto_5(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_t
 	proto.param_types[2] = t3;
 	proto.param_types[3] = t4;
 	proto.param_types[4] = t5;
-	return ir_strl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 5);
+	return ir_stringl(ctx, (const char *)&proto, offsetof(ir_proto_t, param_types) + 5);
 }

-ir_ref ir_proto(ir_ctx *ctx, uint8_t flags, ir_type ret_type, uint32_t params_count, uint8_t *param_types)
+ir_str ir_proto(ir_ctx *ctx, uint8_t flags, ir_type ret_type, uint32_t params_count, uint8_t *param_types)
 {
 	ir_proto_t *proto = alloca(offsetof(ir_proto_t, param_types) + params_count);

@@ -821,7 +1024,7 @@ ir_ref ir_proto(ir_ctx *ctx, uint8_t flags, ir_type ret_type, uint32_t params_co
 	if (params_count) {
 		memcpy(proto->param_types, param_types, params_count);
 	}
-	return ir_strl(ctx, (const char *)proto, offsetof(ir_proto_t, param_types) + params_count);
+	return ir_stringl(ctx, (const char *)proto, offsetof(ir_proto_t, param_types) + params_count);
 }

 /* IR construction */
@@ -896,8 +1099,12 @@ IR_ALWAYS_INLINE ir_ref _ir_fold_cast(ir_ctx *ctx, ir_ref ref, ir_type type)
 		return ref;
 	} else if (IR_IS_CONST_REF(ref) && !IR_IS_SYM_CONST(ctx->ir_base[ref].op)) {
 		return ir_const(ctx, ctx->ir_base[ref].val, type);
-	} else {
+	} else if (EXPECTED(!ctx->use_lists)) {
 		return ir_emit1(ctx, IR_OPT(IR_BITCAST, type), ref);
+	} else {
+		ir_ref ret = ir_emit1(ctx, IR_OPTX(IR_BITCAST, type, 1), ref);
+		ir_use_list_add(ctx, ref, ret);
+		return ret;
 	}
 }

@@ -1063,12 +1270,27 @@ ir_ref ir_folding(ir_ctx *ctx, uint32_t opt, ir_ref op1, ir_ref op2, ir_ref op3,
 		return IR_FOLD_DO_COPY;
 	}
 ir_fold_const:
-	if (!ctx->use_lists) {
-		return ir_const(ctx, val, IR_OPT_TYPE(opt));
-	} else {
-		ctx->fold_insn.opt = IR_OPT(IR_OPT_TYPE(opt), IR_OPT_TYPE(opt));
-		ctx->fold_insn.val.u64 = val.u64;
-		return IR_FOLD_DO_CONST;
+	{
+		ir_type type = IR_OPT_TYPE(opt);
+
+		/* Extend a narrow integer result to its type width so no
+		 * fold rule sees garbage in the upper bits. */
+		if (IR_IS_TYPE_INT(type) && ir_type_size[type] < 8) {
+			uint32_t shift = (8 - ir_type_size[type]) * 8;
+			if (IR_IS_TYPE_SIGNED(type)) {
+				val.i64 = (int64_t)(val.u64 << shift) >> shift;
+			} else {
+				val.u64 = (val.u64 << shift) >> shift;
+			}
+		}
+
+		if (!ctx->use_lists) {
+			return ir_const(ctx, val, type);
+		} else {
+			ctx->fold_insn.opt = IR_OPT(type, type);
+			ctx->fold_insn.val.u64 = val.u64;
+			return IR_FOLD_DO_CONST;
+		}
 	}
 }

@@ -1159,13 +1381,25 @@ ir_ref ir_get_op(const ir_ctx *ctx, ir_ref ref, int32_t n)
 ir_ref ir_param(ir_ctx *ctx, ir_type type, ir_ref region, const char *name, int pos)
 {
 	IR_ASSERT(ctx->ir_base[region].op == IR_START);
-	return ir_emit(ctx, IR_OPT(IR_PARAM, type), region, ir_str(ctx, name), pos);
+	return ir_emit(ctx, IR_OPT(IR_PARAM, type), region, ir_string(ctx, name), pos);
+}
+
+ir_ref ir_param_ex(ir_ctx *ctx, ir_type type, ir_ref region, ir_str name, int pos)
+{
+	IR_ASSERT(ctx->ir_base[region].op == IR_START);
+	return ir_emit(ctx, IR_OPT(IR_PARAM, type), region, name, pos);
 }

 ir_ref ir_var(ir_ctx *ctx, ir_type type, ir_ref region, const char *name)
 {
 	IR_ASSERT(IR_IS_BB_START(ctx->ir_base[region].op));
-	return ir_emit(ctx, IR_OPT(IR_VAR, type), region, ir_str(ctx, name), IR_UNUSED);
+	return ir_emit(ctx, IR_OPT(IR_VAR, type), region, ir_string(ctx, name), IR_UNUSED);
+}
+
+ir_ref ir_var_ex(ir_ctx *ctx, ir_type type, ir_ref region, ir_str name)
+{
+	IR_ASSERT(IR_IS_BB_START(ctx->ir_base[region].op));
+	return ir_emit(ctx, IR_OPT(IR_VAR, type), region, name, IR_UNUSED);
 }

 ir_ref ir_bind(ir_ctx *ctx, ir_ref var, ir_ref def)
@@ -1287,7 +1521,7 @@ void ir_build_def_use_lists(ir_ctx *ctx)
 					/* form a linked list of "uses" (like in binsort) */
 					linked_lists[linked_lists_top] = i; /* store the "use" */
 					linked_lists[linked_lists_top + 1] = use_list->refs; /* store list next */
-					use_list->refs = -(linked_lists_top + 1); /* store a head of the list using a negative number */
+					use_list->refs = -(ir_ref)(linked_lists_top + 1); /* store a head of the list using a negative number */
 					linked_lists_top += 2;
 					use_list->count++;
 				}
@@ -1298,7 +1532,8 @@ void ir_build_def_use_lists(ir_ctx *ctx)
 		insn += n;
 	}

-	ctx->use_edges_count = edges_count;
+	IR_ASSERT(edges_count <= 0x7fffffff);
+	ctx->use_edges_count = (ir_ref)edges_count;
 	edges = ir_mem_malloc(IR_ALIGNED_SIZE(edges_count * sizeof(ir_ref), 4096));
 	for (use_list = lists + ctx->insns_count - 1; use_list != lists; use_list--) {
 		n = use_list->refs;
@@ -1311,7 +1546,7 @@ void ir_build_def_use_lists(ir_ctx *ctx)
 			}
 			IR_ASSERT(n > 0);
 			edges[--edges_count] = n;
-			use_list->refs = edges_count;
+			use_list->refs = (ir_ref)edges_count;
 		}
 	}

@@ -1339,7 +1574,7 @@ void ir_use_list_remove_all(ir_ctx *ctx, ir_ref from, ir_ref ref)
 		}
 	}
 	if (p != q) {
-		use_list->count -= (p - q);
+		use_list->count -= (ir_ref)(p - q);
 		do {
 			*q = IR_UNUSED;
 			q++;
@@ -1981,133 +2216,146 @@ typedef enum _ir_alias {
 	IR_MUST_ALIAS =  1,
 } ir_alias;

-#if 0
-static ir_alias ir_check_aliasing(ir_ctx *ctx, ir_ref addr1, ir_ref addr2)
+IR_ALWAYS_INLINE const ir_insn *ir_decompose_addr(const ir_ctx *ctx, ir_ref addr, ir_ref *base, ir_ref *index, intptr_t *offset)
 {
-	ir_insn *insn1, *insn2;
+	const ir_insn *insn = &ctx->ir_base[addr];
+	ir_ref idx = IR_UNUSED;
+	intptr_t off = 0;

-	if (addr1 == addr2) {
-		return IR_MUST_ALIAS;
+	while (1) {
+		if (insn->op == IR_ADD) {
+			ir_ref op1 = insn->op1;
+			ir_ref op2 = insn->op2;
+			const ir_insn *op1_insn = &ctx->ir_base[op1];
+			const ir_insn *op2_insn = &ctx->ir_base[op2];
+
+			if ((op2_insn->type == IR_ADDR && op1_insn->type != IR_ADDR)
+			 || op2_insn->op == IR_SYM
+			 || op2_insn->op == IR_ALLOCA
+			 || op2_insn->op == IR_VADDR) {
+				const ir_insn *tmp = op1_insn;
+				op1_insn = op2_insn;
+				op2_insn = tmp;
+				SWAP_REFS(op1, op2);
+		    }
+			if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(op2_insn->op)) {
+				off += op2_insn->val.addr;
+				addr = op1;
+				insn = op1_insn;
+			} else if (!idx) {
+				addr = op1;
+				insn = op1_insn;
+				idx = op2;
+			} else {
+				goto exit;
+			}
+		} else if (insn->op == IR_SUB
+		 && IR_IS_CONST_REF(insn->op2)
+		 && !IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op)) {
+			off -= ctx->ir_base[insn->op2].val.addr;
+			addr = insn->op1;
+			insn = &ctx->ir_base[insn->op1];
+		} else {
+			break;
+		}
 	}

-	insn1 = &ctx->ir_base[addr1];
-	insn2 = &ctx->ir_base[addr2];
-	if (insn1->op == IR_ADD && IR_IS_CONST_REF(insn1->op2)) {
-		if (insn1->op1 == addr2) {
-			uintptr_t offset1 = ctx->ir_base[insn1->op2].val.u64;
-			return (offset1 != 0) ? IR_MUST_ALIAS : IR_NO_ALIAS;
-		} else if (insn2->op == IR_ADD && IR_IS_CONST_REF(insn1->op2) && insn1->op1 == insn2->op1) {
-			if (insn1->op2 == insn2->op2) {
-				return IR_MUST_ALIAS;
-			} else if (IR_IS_CONST_REF(insn1->op2) && IR_IS_CONST_REF(insn2->op2)) {
-				uintptr_t offset1 = ctx->ir_base[insn1->op2].val.u64;
-				uintptr_t offset2 = ctx->ir_base[insn2->op2].val.u64;
+	if (idx) {
+		while (1) {
+			const ir_insn *insn = &ctx->ir_base[idx];

-				return (offset1 == offset2) ? IR_MUST_ALIAS : IR_NO_ALIAS;
+			if (insn->op == IR_ADD) {
+				if (IR_IS_CONST_REF(insn->op2)
+				 && !IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op)) {
+					off += ctx->ir_base[insn->op2].val.addr;
+					idx = insn->op1;
+				} else {
+					break;
+				}
+			} else if (insn->op == IR_SUB
+			 && IR_IS_CONST_REF(insn->op2)
+			 && !IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op)) {
+				off -= ctx->ir_base[insn->op2].val.addr;
+				idx = insn->op1;
+			} else {
+				break;
 			}
 		}
-	} else if (insn2->op == IR_ADD && IR_IS_CONST_REF(insn2->op2)) {
-		if (insn2->op1 == addr1) {
-			uintptr_t offset2 = ctx->ir_base[insn2->op2].val.u64;
+	}
+
+exit:
+	*base = addr;
+	*index = idx;
+	*offset = off;

-			return (offset2 != 0) ? IR_MUST_ALIAS : IR_NO_ALIAS;
+	return insn;
+}
+
+IR_ALWAYS_INLINE const ir_insn *ir_get_base_addr(const ir_ctx *ctx, const ir_insn *insn)
+{
+	while (1) {
+		if (insn->op == IR_ADD) {
+			ir_ref op1 = insn->op1;
+			ir_ref op2 = insn->op2;
+			const ir_insn *op1_insn = &ctx->ir_base[op1];
+			const ir_insn *op2_insn = &ctx->ir_base[op2];
+
+			if (op2_insn->op == IR_SYM || op2_insn->op == IR_ALLOCA || op2_insn->op == IR_VADDR) {
+				return op2_insn;
+			} else if (op2_insn->type == IR_ADDR && op1_insn->type != IR_ADDR) {
+				insn = op2_insn;
+		    } else {
+				insn = op1_insn;
+			}
+		} else if (insn->op == IR_SUB) {
+			insn = &ctx->ir_base[insn->op1];
+		} else {
+			break;
 		}
 	}
-	return IR_MAY_ALIAS;
+	return insn;
 }
-#endif

-ir_alias ir_check_partial_aliasing(const ir_ctx *ctx, ir_ref addr1, ir_ref addr2, ir_type type1, ir_type type2)
+static ir_alias ir_check_aliasing(const ir_ctx *ctx, ir_ref addr1, ir_ref addr2, ir_type type1, ir_type type2)
 {
 	const ir_insn *insn1, *insn2;
-	ir_ref base1, base2, off1, off2;
+	ir_ref base1, base2, index1, index2;
+	intptr_t offset1, offset2;

 	/* this must be already check */
 	IR_ASSERT(addr1 != addr2);

-	insn1 = &ctx->ir_base[addr1];
-	insn2 = &ctx->ir_base[addr2];
-	if (insn1->op != IR_ADD) {
-		base1 = addr1;
-		off1 = IR_UNUSED;
-	} else if (ctx->ir_base[insn1->op2].op == IR_SYM
-			|| ctx->ir_base[insn1->op2].op == IR_ALLOCA
-			|| ctx->ir_base[insn1->op2].op == IR_VADDR) {
-		base1 = insn1->op2;
-		off1 = insn1->op1;
-	} else {
-		base1 = insn1->op1;
-		off1 = insn1->op2;
-	}
-	if (insn2->op != IR_ADD) {
-		base2 = addr2;
-		off2 = IR_UNUSED;
-	} else if (ctx->ir_base[insn2->op2].op == IR_SYM
-			|| ctx->ir_base[insn2->op2].op == IR_ALLOCA
-			|| ctx->ir_base[insn2->op2].op == IR_VADDR) {
-		base2 = insn2->op2;
-		off2 = insn2->op1;
-	} else {
-		base2 = insn2->op1;
-		off2 = insn2->op2;
-	}
-	if (base1 == base2) {
-		uintptr_t offset1, offset2;
+	/* check if addresses overlap */
+	insn1 = ir_decompose_addr(ctx, addr1, &base1, &index1, &offset1);
+	insn2 = ir_decompose_addr(ctx, addr2, &base2, &index2, &offset2);

-		if (!off1) {
-			offset1 = 0;
-		} else if (IR_IS_CONST_REF(off1) && !IR_IS_SYM_CONST(ctx->ir_base[off1].op)) {
-			offset1 = ctx->ir_base[off1].val.addr;
-		} else {
-			return IR_MAY_ALIAS;
-		}
-		if (!off2) {
-			offset2 = 0;
-		} else if (IR_IS_CONST_REF(off2) && !IR_IS_SYM_CONST(ctx->ir_base[off2].op)) {
-			offset2 = ctx->ir_base[off2].val.addr;
-		} else {
+	if (base1 == base2) {
+		if (index1 != index2) {
 			return IR_MAY_ALIAS;
-		}
-		if (offset1 == offset2) {
+		} else if (offset1 == offset2) {
 			return IR_MUST_ALIAS;
 		} else if (offset1 < offset2) {
-			return offset1 + ir_type_size[type1] <= offset2 ? IR_NO_ALIAS : IR_MUST_ALIAS;
+			return offset1 + (intptr_t)ir_get_type_size(type1) <= offset2 ? IR_NO_ALIAS : IR_MUST_ALIAS;
 		} else {
-			return offset2 + ir_type_size[type2] <= offset1 ? IR_NO_ALIAS : IR_MUST_ALIAS;
-		}
-	} else {
-		insn1 = &ctx->ir_base[base1];
-		insn2 = &ctx->ir_base[base2];
-		while (insn1->op == IR_ADD) {
-			insn1 = &ctx->ir_base[insn1->op2];
-			if (insn1->op == IR_SYM
-			 || insn1->op == IR_ALLOCA
-			 || insn1->op == IR_VADDR) {
-				break;
-			} else {
-				insn1 = &ctx->ir_base[insn1->op1];
-			}
-		}
-		while (insn2->op == IR_ADD) {
-			insn2 = &ctx->ir_base[insn2->op2];
-			if (insn2->op == IR_SYM
-			 || insn2->op == IR_ALLOCA
-			 || insn2->op == IR_VADDR) {
-				break;
-			} else {
-				insn2 = &ctx->ir_base[insn2->op1];
-			}
-		}
-		if (insn1 == insn2) {
-			return IR_MAY_ALIAS;
-		}
-		if ((insn1->op == IR_ALLOCA && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR || insn2->op == IR_SYM || insn2->op == IR_PARAM))
-		 || (insn1->op == IR_VADDR && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR || insn2->op == IR_SYM || insn2->op == IR_PARAM))
-		 || (insn1->op == IR_SYM && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR || insn2->op == IR_SYM))
-		 || (insn1->op == IR_PARAM && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR))) {
-			return IR_NO_ALIAS;
+			return offset2 + (intptr_t)ir_get_type_size(type2) <= offset1 ? IR_NO_ALIAS : IR_MUST_ALIAS;
 		}
 	}
+
+	/* check if addresses lay in different memory areas (e.g. local variables cannot alias with arguments) */
+	insn1 = ir_get_base_addr(ctx, insn1);
+	insn2 = ir_get_base_addr(ctx, insn2);
+
+	if (insn1 == insn2 || insn1->type != IR_ADDR || insn2->type != IR_ADDR) {
+		return IR_MAY_ALIAS;
+	}
+
+	if ((insn1->op == IR_ALLOCA && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR || insn2->op == IR_SYM || insn2->op == IR_PARAM))
+	 || (insn1->op == IR_VADDR && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR || insn2->op == IR_SYM || insn2->op == IR_PARAM))
+	 || (insn1->op == IR_SYM && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR || insn2->op == IR_SYM))
+	 || (insn1->op == IR_PARAM && (insn2->op == IR_ALLOCA || insn2->op == IR_VADDR))) {
+		return IR_NO_ALIAS;
+	}
+
 	return IR_MAY_ALIAS;
 }

@@ -2122,9 +2370,9 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_load_i(const ir_ctx *ctx, ir_ref ref, i
 			if (insn->op2 == addr) {
 				if (insn->type == type) {
 					return ref; /* load forwarding (L2L) */
-				} else if (ir_type_size[insn->type] == ir_type_size[type]) {
+				} else if (ir_get_type_size(insn->type) == ir_get_type_size(type)) {
 					return ref; /* load forwarding with bitcast (L2L) */
-				} else if (ir_type_size[insn->type] > ir_type_size[type]
+				} else if (ir_get_type_size(insn->type) > ir_get_type_size(type)
 						&& IR_IS_TYPE_INT(type) && IR_IS_TYPE_INT(insn->type)) {
 					return ref; /* partial load forwarding (L2L) */
 				}
@@ -2139,15 +2387,15 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_load_i(const ir_ctx *ctx, ir_ref ref, i
 					return IR_UNUSED;
 				} else if (type2 == type) {
 					return insn->op3; /* store forwarding (S2L) */
-				} else if (ir_type_size[type2] == ir_type_size[type]) {
+				} else if (ir_get_type_size(type2) == ir_get_type_size(type)) {
 					return insn->op3; /* store forwarding with bitcast (S2L) */
-				} else if (ir_type_size[type2] > ir_type_size[type]
+				} else if (ir_get_type_size(type2) > ir_get_type_size(type)
 						&& IR_IS_TYPE_INT(type) && IR_IS_TYPE_INT(type2)) {
 					return insn->op3; /* partial store forwarding (S2L) */
 				} else {
 					return IR_UNUSED;
 				}
-			} else if (ir_check_partial_aliasing(ctx, addr, insn->op2, type, type2) != IR_NO_ALIAS) {
+			} else if (ir_check_aliasing(ctx, addr, insn->op2, type, type2) != IR_NO_ALIAS) {
 				return IR_UNUSED;
 			}
 		} else if (insn->op == IR_RSTORE) {
@@ -2198,9 +2446,9 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_vload_i(const ir_ctx *ctx, ir_ref ref,
 			if (insn->op2 == var) {
 				if (insn->type == type) {
 					return ref; /* load forwarding (L2L) */
-				} else if (ir_type_size[insn->type] == ir_type_size[type]) {
+				} else if (ir_get_type_size(insn->type) > ir_get_type_size(type)) {
 					return ref; /* load forwarding with bitcast (L2L) */
-				} else if (ir_type_size[insn->type] > ir_type_size[type]
+				} else if (ir_get_type_size(insn->type) > ir_get_type_size(type)
 						&& IR_IS_TYPE_INT(type) && IR_IS_TYPE_INT(insn->type)) {
 					return ref; /* partial load forwarding (L2L) */
 				}
@@ -2211,9 +2459,9 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_vload_i(const ir_ctx *ctx, ir_ref ref,
 			if (insn->op2 == var) {
 				if (type2 == type) {
 					return insn->op3; /* store forwarding (S2L) */
-				} else if (ir_type_size[type2] == ir_type_size[type]) {
+				} else if (ir_get_type_size(type2) == ir_get_type_size(type)) {
 					return insn->op3; /* store forwarding with bitcast (S2L) */
-				} else if (ir_type_size[type2] > ir_type_size[type]
+				} else if (ir_get_type_size(type2) > ir_get_type_size(type)
 						&& IR_IS_TYPE_INT(type) && IR_IS_TYPE_INT(type2)) {
 					return insn->op3; /* partial store forwarding (S2L) */
 				} else {
@@ -2303,9 +2551,15 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_store_i(ir_ctx *ctx, ir_ref ref, ir_ref
 								ir_use_list_replace_one(ctx, prev, ref, next);
 								if (!IR_IS_CONST_REF(insn->op2)) {
 									ir_use_list_remove_one(ctx, insn->op2, ref);
+									if (ctx->iter_worklist && ctx->use_lists[insn->op2].count == 0) {
+										ir_bitqueue_add(ctx->iter_worklist, insn->op2);
+									}
 								}
 								if (!IR_IS_CONST_REF(insn->op3)) {
 									ir_use_list_remove_one(ctx, insn->op3, ref);
+									if (ctx->iter_worklist && ctx->use_lists[insn->op3].count == 0) {
+										ir_bitqueue_add(ctx->iter_worklist, insn->op3);
+									}
 								}
 								insn->op1 = IR_UNUSED;
 							}
@@ -2330,7 +2584,7 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_store_i(ir_ctx *ctx, ir_ref ref, ir_ref
 			}
 			type2 = insn->type;
 check_aliasing:
-			if (ir_check_partial_aliasing(ctx, addr, insn->op2, type, type2) != IR_NO_ALIAS) {
+			if (ir_check_aliasing(ctx, addr, insn->op2, type, type2) != IR_NO_ALIAS) {
 				break;
 			}
 		} else if (insn->op == IR_GUARD || insn->op == IR_GUARD_NOT) {
@@ -2406,6 +2660,9 @@ IR_ALWAYS_INLINE ir_ref ir_find_aliasing_vstore_i(ir_ctx *ctx, ir_ref ref, ir_re
 							}
 							if (!IR_IS_CONST_REF(insn->op3)) {
 								ir_use_list_remove_one(ctx, insn->op3, ref);
+								if (ctx->iter_worklist && ctx->use_lists[insn->op3].count == 0) {
+									ir_bitqueue_add(ctx->iter_worklist, insn->op3);
+								}
 							}
 							insn->op1 = IR_UNUSED;
 						}
@@ -3299,7 +3556,7 @@ ir_ref _ir_VLOAD(ir_ctx *ctx, ir_type type, ir_ref var)

 			if (insn->type == type) {
 				return ref;
-			} else if (ir_type_size[insn->type] == ir_type_size[type]) {
+			} else if (ir_get_type_size(insn->type) == ir_get_type_size(type)) {
 				return ir_fold1(ctx, IR_OPT(IR_BITCAST, type), ref); /* load forwarding with bitcast (L2L) */
 			} else {
 				return ir_fold1(ctx, IR_OPT(IR_TRUNC, type), ref); /* partial load forwarding (L2L) */
@@ -3333,10 +3590,10 @@ void _ir_VSTORE_v(ir_ctx *ctx, ir_ref var, ir_ref val)
 	ctx->control = ir_emit3(ctx, IR_VSTORE_v, ctx->control, var, val);
 }

-ir_ref _ir_TLS(ir_ctx *ctx, ir_ref index, ir_ref offset)
+ir_ref _ir_TLS_ADDR(ir_ctx *ctx, ir_ref index, ir_ref offset)
 {
 	IR_ASSERT(ctx->control);
-	return ctx->control = ir_emit3(ctx, IR_OPT(IR_TLS, IR_ADDR), ctx->control, index, offset);
+	return ctx->control = ir_emit3(ctx, IR_OPT(IR_TLS_ADDR, IR_ADDR), ctx->control, index, offset);
 }

 ir_ref _ir_RLOAD(ir_ctx *ctx, ir_type type, ir_ref reg)
@@ -3366,7 +3623,7 @@ ir_ref _ir_LOAD(ir_ctx *ctx, ir_type type, ir_ref addr)

 			if (insn->type == type) {
 				return ref;
-			} else if (ir_type_size[insn->type] == ir_type_size[type]) {
+			} else if (ir_get_type_size(insn->type) == ir_get_type_size(type)) {
 				return ir_fold1(ctx, IR_OPT(IR_BITCAST, type), ref); /* load forwarding with bitcast (L2L) */
 			} else {
 				return ir_fold1(ctx, IR_OPT(IR_TRUNC, type), ref); /* partial load forwarding (L2L) */
diff --git a/ext/opcache/jit/ir/ir.h b/ext/opcache/jit/ir/ir.h
index 01db4ecf6b1..1470606a101 100644
--- a/ext/opcache/jit/ir/ir.h
+++ b/ext/opcache/jit/ir/ir.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Public API)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -53,18 +53,52 @@ extern "C" {
 # endif
 #endif

-#if defined(IR_TARGET_X86)
-# define IR_TARGET "x86"
-#elif defined(IR_TARGET_X64)
-# ifdef _WIN64
-#  define IR_TARGET "Windows-x86_64" /* 64-bit Windows use different ABI and calling convention */
+#ifndef IR_TARGET_TRIPLET
+# if defined(IR_TARGET_X64)
+#  if defined(_WIN32)
+#   define IR_TARGET_TRIPLET "x86-windows-msvc"
+#  elif defined(__APPLE__)
+#   define IR_TARGET_TRIPLET "x86-darwin"
+#  elif defined(__linux__)
+#   define IR_TARGET_TRIPLET "x86-linux-sysv"
+#  elif defined(__FreeBSD__)
+#   define IR_TARGET_TRIPLET "x86-freebsd-sysv"
+#  elif defined(__NetBSD__)
+#   define IR_TARGET_TRIPLET "x86-netbsd-sysv"
+#  else
+#   define IR_TARGET_TRIPLET "x86-unknown-sysv"
+#  endif
+# elif defined(IR_TARGET_X86)
+#  if defined(_WIN32)
+#   define IR_TARGET_TRIPLET "x86_64-windows-msvc"
+#  elif defined(__APPLE__)
+#   define IR_TARGET_TRIPLET "x86_64-darwin"
+#  elif defined(__linux__)
+#   define IR_TARGET_TRIPLET "x86_64-linux-sysv"
+#  elif defined(__FreeBSD__)
+#   define IR_TARGET_TRIPLET "x86_64-freebsd-sysv"
+#  elif defined(__NetBSD__)
+#   define IR_TARGET_TRIPLET "x86_64-netbsd-sysv"
+#  else
+#   define IR_TARGET_TRIPLET "x86_64-unknown-sysv"
+#  endif
+# elif defined(IR_TARGET_AARCH64)
+#  if defined(_WIN32)
+#   define IR_TARGET_TRIPLET "aarch64-windows"
+#  elif defined(__APPLE__)
+#   define IR_TARGET_TRIPLET "aarch64-darwin"
+#  elif defined(__linux__)
+#   define IR_TARGET_TRIPLET "aarch64-linux-sysv"
+#  elif defined(__FreeBSD__)
+#   define IR_TARGET_TRIPLET "aarch64-freebsd-sysv"
+#  elif defined(__NetBSD__)
+#   define IR_TARGET_TRIPLET "aarch64-netbsd-sysv"
+#  else
+#   define IR_TARGET_TRIPLET "aarch64-unknown-sysv"
+#  endif
 # else
-#  define IR_TARGET "x86_64"
+#  error "Unknown IR_TARGET"
 # endif
-#elif defined(IR_TARGET_AARCH64)
-# define IR_TARGET "aarch64"
-#else
-# error "Unknown IR target"
 #endif

 #if defined(__SIZEOF_SIZE_T__)
@@ -115,11 +149,23 @@ extern "C" {
 # include "ir_php.h"
 #endif

-/* IR Type flags (low 4 bits are used for type size) */
-#define IR_TYPE_SIGNED     (1<<4)
-#define IR_TYPE_UNSIGNED   (1<<5)
-#define IR_TYPE_FP         (1<<6)
-#define IR_TYPE_SPECIAL    (1<<7)
+#ifdef IR_TARGET_X86
+# ifndef IR_X86_I64
+#  define IR_X86_I64 1
+# endif
+#else
+# define IR_X86_I64 0
+#endif
+
+#ifndef IR_SIMD
+# define IR_SIMD 1
+#endif
+
+/* IR Type flags */
+#define IR_TYPE_SIGNED     (1<<0)
+#define IR_TYPE_UNSIGNED   (1<<1)
+#define IR_TYPE_FP         (1<<2)
+#define IR_TYPE_SPECIAL    (1<<3)
 #define IR_TYPE_BOOL       (IR_TYPE_SPECIAL|IR_TYPE_UNSIGNED)
 #define IR_TYPE_ADDR       (IR_TYPE_SPECIAL|IR_TYPE_UNSIGNED)
 #define IR_TYPE_CHAR       (IR_TYPE_SPECIAL|IR_TYPE_SIGNED)
@@ -143,16 +189,33 @@ extern "C" {
 #define IR_IS_TYPE_UNSIGNED(t) ((t) < IR_CHAR)
 #define IR_IS_TYPE_SIGNED(t)   ((t) >= IR_CHAR && (t) < IR_DOUBLE)
 #define IR_IS_TYPE_INT(t)      ((t) < IR_DOUBLE)
-#define IR_IS_TYPE_FP(t)       ((t) >= IR_DOUBLE)
+#define IR_IS_TYPE_FP(t)       ((t) >= IR_DOUBLE && (t) <= IR_FLOAT)

 #define IR_TYPE_ENUM(name, type, field, flags) IR_ ## name,

 typedef enum _ir_type {
 	IR_VOID,
 	IR_TYPES(IR_TYPE_ENUM)
-	IR_LAST_TYPE
+	IR_LAST_TYPE,
+
+	IR_BASE_TYPE_MASK = 0x0f,
+	IR_VECTOR_MASK    = 0x70,
+
+	IR_VECTOR_1       = 0x10,
+	IR_VECTOR_2       = 0x20,
+	IR_VECTOR_4       = 0x30,
+	IR_VECTOR_8       = 0x40,
+	IR_VECTOR_16      = 0x50,
+	IR_VECTOR_32      = 0x60,
+	IR_VECTOR_64      = 0x70,
 } ir_type;

+#define IR_IS_TYPE_SCALAR(t)   (((t) & IR_VECTOR_MASK) == 0)
+#define IR_IS_TYPE_VECTOR(t)   (((t) & IR_VECTOR_MASK) != 0)
+
+#define IR_VECTOR_BASE_TYPE(t) ((t) & IR_BASE_TYPE_MASK)
+#define IR_VECTOR_LENGTH(t)    (1U << ((((t) & IR_VECTOR_MASK) >> 4) - 1))
+
 #ifdef IR_64
 # define IR_SIZE_T          IR_U64
 # define IR_SSIZE_T         IR_I64
@@ -308,6 +371,12 @@ typedef enum _ir_type {
 	_(MAX,	        d2C,  def, def, ___) /* max(op1, op2)               */ \
 	_(COND,	        d3,   def, def, def) /* op1 ? op2 : op3             */ \
 	\
+	/* SIMD vector ops                                                  */ \
+	_(EXTRACT,      d2,   def, def, ___) /* get element of vector       */ \
+	_(REPLACE,      d3,   def, def, def) /* set element of vector       */ \
+	_(SPLAT,        d1,   def, ___, ___) /* set all elements of vector  */ \
+	_(SHUFFLE,      d3,   def, def, def) /* shuffle elements of vectors */ \
+	\
 	/* data-flow and miscellaneous ops                                  */ \
 	_(VADDR,        d1,   var, ___, ___) /* load address of local var   */ \
 	_(FRAME_ADDR,   d0,   ___, ___, ___) /* function frame address      */ \
@@ -326,6 +395,7 @@ typedef enum _ir_type {
 	_(SYM,          r0,   ___, ___, ___) /* constant symbol ref         */ \
 	_(LABEL,        r0,   ___, ___, ___) /* label address ref           */ \
 	_(STR,          r0,   ___, ___, ___) /* constant str ref            */ \
+	_(LONG_CONST,   r0,   ___, ___, ___) /* long constant (vector)      */ \
 	\
 	/* call ops                                                         */ \
 	_(CALL,         xN,   src, def, def) /* CALL(src, func, args...)    */ \
@@ -346,7 +416,8 @@ typedef enum _ir_type {
 	_(LOAD_v,       l2,   src, ref, ___) /* volatile variant of VLOAD   */ \
 	_(STORE,        s3,   src, ref, def) /* store to memory             */ \
 	_(STORE_v,      s3,   src, ref, def) /* volatile variant of VSTORE  */ \
-	_(TLS,          l1X2, src, num, num) /* thread local variable       */ \
+	_(TLS_ADDR,     l1X2, src, num, num) /* TLS_ADDR(_, module, offset) */ \
+	                                     /* for static TLS module is -1 */ \
 	_(TRAP,         x1,   src, ___, ___) /* DebugBreak                  */ \
 	/* memory reference ops (A, H, U, S, TMP, STR, NEW, X, V) ???       */ \
 	\
@@ -423,11 +494,17 @@ typedef enum _ir_op {
 #define IR_VA_ARG_ALIGN(op3) (1U << ((uint32_t)(op3) & 0x7))
 #define IR_VA_ARG_OP3(s, a)  (((s) << 3) | ir_ntzl(a))

-/* IR References */
+/* IR Reference: index of ir_insn in ir_ctx.ir_base[], positive - instructions, negaive - constants */
 typedef int32_t ir_ref;

 #define IR_IS_CONST_REF(ref) ((ref) < 0)

+/* IR String: string index; positive - index in ir_strtab, negative - resolved through loader.get_str() */
+typedef int32_t ir_str;
+
+#define IR_IS_EXT_STR(str)   ((str) < 0)
+#define IR_EXT_STR(str)      (-(str))
+
 /* IR Constant Value */
 #define IR_UNUSED            0
 #define IR_NULL              (-1)
@@ -459,8 +536,8 @@ typedef union _ir_val {
 			int32_t                    i32;
 			float                      f;
 			ADDR_MEMBER
-			ir_ref                     name;
-			ir_ref                     str;
+			ir_str                     name;
+			ir_str                     str;
 			IR_STRUCT_LOHI(
 				union {
 					uint16_t           u16;
@@ -499,6 +576,7 @@ typedef struct _ir_insn {
 					uint16_t           inputs_count;       /* number of input control edges for MERGE, PHI, CALL, TAILCALL */
 					uint16_t           prev_insn_offset;   /* 16-bit backward offset from current instruction for CSE */
 					uint16_t           proto;
+					uint16_t           long_const_size;
 				}
 			);
 			uint32_t                   optx;
@@ -536,14 +614,14 @@ typedef struct _ir_strtab {

 #define ir_strtab_count(strtab) (strtab)->count

-typedef void (*ir_strtab_apply_t)(const char *str, uint32_t len, ir_ref val);
+typedef void (*ir_strtab_apply_t)(const char *str, uint32_t len, ir_str val);

 void ir_strtab_init(ir_strtab *strtab, uint32_t count, uint32_t buf_size);
-ir_ref ir_strtab_lookup(ir_strtab *strtab, const char *str, uint32_t len, ir_ref val);
-ir_ref ir_strtab_find(const ir_strtab *strtab, const char *str, uint32_t len);
-ir_ref ir_strtab_update(ir_strtab *strtab, const char *str, uint32_t len, ir_ref val);
-const char *ir_strtab_str(const ir_strtab *strtab, ir_ref idx);
-const char *ir_strtab_strl(const ir_strtab *strtab, ir_ref idx, size_t *len);
+ir_str ir_strtab_lookup(ir_strtab *strtab, const char *str, uint32_t len, ir_str val);
+ir_str ir_strtab_find(const ir_strtab *strtab, const char *str, uint32_t len);
+ir_str ir_strtab_update(ir_strtab *strtab, const char *str, uint32_t len, ir_str val);
+const char *ir_strtab_str(const ir_strtab *strtab, ir_str idx);
+const char *ir_strtab_strl(const ir_strtab *strtab, ir_str idx, size_t *len);
 void ir_strtab_apply(const ir_strtab *strtab, ir_strtab_apply_t func);
 void ir_strtab_free(ir_strtab *strtab);

@@ -578,6 +656,7 @@ void ir_strtab_free(ir_strtab *strtab);
 #define IR_OPT_CFG             (1<<21) /* merge BBs, by remove END->BEGIN nodes during CFG construction */
 #define IR_OPT_MEM2SSA         (1<<22)
 #define IR_OPT_CODEGEN         (1<<23)
+#define IR_OPT_TAILCALL        (1<<24)

 /* debug related */
 #ifdef IR_DEBUG
@@ -636,6 +715,8 @@ typedef struct {
 	int   offset;
 } ir_value_param;

+typedef struct _ir_bitqueue ir_bitqueue;
+
 #define IR_CONST_HASH_SIZE 64

 struct _ir_ctx {
@@ -653,6 +734,7 @@ struct _ir_ctx {
 	int32_t            status;                  /* non-zero error code (see IR_ERROR_... macros), app may use negative codes */
 	ir_ref             fold_cse_limit;          /* CSE finds identical insns backward from "insn_count" to "fold_cse_limit" */
 	ir_insn            fold_insn;               /* temporary storage for folding engine */
+	ir_bitqueue       *iter_worklist;
 	ir_value_param    *value_params;            /* information about "by-val" struct parameters */
 	ir_hashtab        *binding;
 	ir_use_list       *use_lists;               /* def->use lists for each instruction */
@@ -667,6 +749,7 @@ struct _ir_ctx {
 	uint32_t          *rules;                   /* array of target specific code-generation rules (for each instruction) */
 	uint32_t          *vregs;
 	ir_ref             vregs_count;
+	ir_str             func_name;               /* Function name (should be set through ir_string()/ir_stringl()) */
 	int32_t            spill_base;              /* base register for special spill area (e.g. PHP VM frame pointer) */
 	uint64_t           fixed_regset;            /* fixed registers, excluded for regular register allocation */
 	int32_t            fixed_stack_red_zone;    /* reusable stack allocated by caller (default 0) */
@@ -681,6 +764,7 @@ struct _ir_ctx {
 	ir_arena          *arena;
 	ir_live_range     *unused_ranges;
 	ir_regs           *regs;
+	int8_t            *tmp_regs;                /* additional tmp registers, used for COND(I64, _, _) and SIMD */
 	ir_strtab         *fused_regs;
 	ir_ref            *prev_ref;
 	union {
@@ -735,20 +819,26 @@ ir_ref ir_const_float(ir_ctx *ctx, float c);
 ir_ref ir_const_double(ir_ctx *ctx, double c);
 ir_ref ir_const_addr(ir_ctx *ctx, uintptr_t c);

-ir_ref ir_const_func_addr(ir_ctx *ctx, uintptr_t c, ir_ref proto);
-ir_ref ir_const_func(ir_ctx *ctx, ir_ref str, ir_ref proto);
-ir_ref ir_const_sym(ir_ctx *ctx, ir_ref str);
-ir_ref ir_const_str(ir_ctx *ctx, ir_ref str);
-ir_ref ir_const_label(ir_ctx *ctx, ir_ref str);
+ir_ref ir_const_func_addr(ir_ctx *ctx, uintptr_t c, ir_str proto);
+ir_ref ir_const_func(ir_ctx *ctx, ir_str str, ir_str proto);
+ir_ref ir_const_sym(ir_ctx *ctx, ir_str str);
+ir_ref ir_const_str(ir_ctx *ctx, ir_str str);
+ir_ref ir_const_label(ir_ctx *ctx, ir_str str);

 ir_ref ir_unique_const_addr(ir_ctx *ctx, uintptr_t c);

+ir_ref ir_long_const(ir_ctx *ctx, ir_type type, size_t size);
+void *ir_long_const_ptr(ir_ctx *ctx, ir_ref ref);
+ir_ref ir_long_const_commit(ir_ctx *ctx, ir_ref ref);
+
+ir_ref ir_const_vector(ir_ctx *ctx, ir_type type);
+
 void ir_print_const(const ir_ctx *ctx, const ir_insn *insn, FILE *f, bool quoted);

-ir_ref ir_str(ir_ctx *ctx, const char *s);
-ir_ref ir_strl(ir_ctx *ctx, const char *s, size_t len);
-const char *ir_get_str(const ir_ctx *ctx, ir_ref idx);
-const char *ir_get_strl(const ir_ctx *ctx, ir_ref idx, size_t *len);
+ir_str ir_string(ir_ctx *ctx, const char *s);
+ir_str ir_stringl(ir_ctx *ctx, const char *s, size_t len);
+const char *ir_get_str(const ir_ctx *ctx, ir_str idx);
+const char *ir_get_strl(const ir_ctx *ctx, ir_str idx, size_t *len);

 #define IR_MAX_PROTO_PARAMS 255

@@ -759,15 +849,15 @@ typedef struct _ir_proto_t {
 	uint8_t param_types[5];
 } ir_proto_t;

-ir_ref ir_proto_0(ir_ctx *ctx, uint8_t flags, ir_type ret_type);
-ir_ref ir_proto_1(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1);
-ir_ref ir_proto_2(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2);
-ir_ref ir_proto_3(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3);
-ir_ref ir_proto_4(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
+ir_str ir_proto_0(ir_ctx *ctx, uint8_t flags, ir_type ret_type);
+ir_str ir_proto_1(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1);
+ir_str ir_proto_2(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2);
+ir_str ir_proto_3(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3);
+ir_str ir_proto_4(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
                                                                 ir_type t4);
-ir_ref ir_proto_5(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
+ir_str ir_proto_5(ir_ctx *ctx, uint8_t flags, ir_type ret_type, ir_type t1, ir_type t2, ir_type t3,
                                                                 ir_type t4, ir_type t5);
-ir_ref ir_proto(ir_ctx *ctx, uint8_t flags, ir_type ret_type, uint32_t params_counts, uint8_t *param_types);
+ir_str ir_proto(ir_ctx *ctx, uint8_t flags, ir_type ret_type, uint32_t params_counts, uint8_t *param_types);

 ir_ref ir_emit(ir_ctx *ctx, uint32_t opt, ir_ref op1, ir_ref op2, ir_ref op3);

@@ -827,7 +917,9 @@ ir_ref ir_fold2(ir_ctx *ctx, uint32_t opt, ir_ref op1, ir_ref op2);
 ir_ref ir_fold3(ir_ctx *ctx, uint32_t opt, ir_ref op1, ir_ref op2, ir_ref op3);

 ir_ref ir_param(ir_ctx *ctx, ir_type type, ir_ref region, const char *name, int pos);
+ir_ref ir_param_ex(ir_ctx *ctx, ir_type type, ir_ref region, ir_str name, int pos);
 ir_ref ir_var(ir_ctx *ctx, ir_type type, ir_ref region, const char *name);
+ir_ref ir_var_ex(ir_ctx *ctx, ir_type type, ir_ref region, ir_str name);

 /* IR Binding */
 ir_ref ir_bind(ir_ctx *ctx, ir_ref var, ir_ref def);
@@ -868,6 +960,7 @@ int ir_compute_live_ranges(ir_ctx *ctx);
 int ir_coalesce(ir_ctx *ctx);
 int ir_compute_dessa_moves(ir_ctx *ctx);
 int ir_reg_alloc(ir_ctx *ctx);
+int ir_reg_alloc_simple(ir_ctx *ctx);

 int ir_regs_number(void);
 bool ir_reg_is_int(int32_t reg);
@@ -882,6 +975,11 @@ bool ir_needs_thunk(const ir_code_buffer *code_buffer, void *addr);
 void *ir_emit_thunk(ir_code_buffer *code_buffer, void *addr, size_t *size_ptr);
 void ir_fix_thunk(void *thunk_entry, void *addr);

+#if defined(_MSC_VER) && defined(IR_TARGET_X86)
+/* MSVC doesn't enforce 16-byte stack alignment */
+int ir_call_with_aligned_stack(int (*func)(int, const char**), int argc, const char **argv);
+#endif
+
 /* Target address resolution (implementation in ir_emit.c) */
 void *ir_resolve_sym_name(const char *name);

@@ -932,10 +1030,12 @@ struct _ir_loader {
 	bool (*sym_data_end)      (ir_loader *loader, uint32_t flags);
 	bool (*func_init)         (ir_loader *loader, ir_ctx *ctx, const char *name);
 	bool (*func_process)      (ir_loader *loader, ir_ctx *ctx, const char *name);
-	void*(*resolve_sym_name)  (ir_loader *loader, const char *name, uint32_t flags);
+	void*(*resolve_sym_name)  (ir_loader *loader, ir_ctx *ctx, ir_str name, uint32_t flags);
 	bool (*has_sym)           (ir_loader *loader, const char *name);
 	bool (*add_sym)           (ir_loader *loader, const char *name, void *addr);
 	bool (*add_label)         (ir_loader *loader, const char *name, void *addr);
+	const char * (*get_str)   (ir_loader *loader, ir_str idx);
+	const char * (*get_strl)  (ir_loader *loader, ir_str idx, size_t *len);
 };

 void ir_loader_init(void);
@@ -957,6 +1057,7 @@ int ir_load_llvm_asm(ir_loader *loader, const char *filename);
 void ir_print_func_proto(const ir_ctx *ctx, const char *name, bool prefix, FILE *f);
 void ir_print_proto(const ir_ctx *ctx, ir_ref proto, FILE *f);
 void ir_print_proto_ex(uint8_t flags, ir_type ret_type, uint32_t params_count, const uint8_t *param_types, FILE *f);
+void ir_print_type_cname(ir_type type, FILE *f);
 void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f);

 /* IR debug dump API (implementation in ir_dump.c) */
@@ -980,6 +1081,7 @@ void ir_emit_llvm_sym_decl(const char *name, uint32_t flags, FILE *f);

 /* IR verification API (implementation in ir_check.c) */
 bool ir_check(const ir_ctx *ctx);
+bool ir_check_prototype(const ir_ctx *ctx, uint32_t flags, uint8_t ret_type, uint32_t params_count, uint8_t *param_types);
 void ir_consistency_check(void);

 /* Code patching (implementation in ir_patch.c) */
@@ -995,7 +1097,8 @@ int ir_patch(const void *code, size_t size, uint32_t jmp_table_size, const void
 # define IR_X86_AVX      (1<<5)
 # define IR_X86_AVX2     (1<<6)
 # define IR_X86_BMI1     (1<<7)
-# define IR_X86_CLDEMOTE (1<<8)
+# define IR_X86_BMI2     (1<<8)
+# define IR_X86_CLDEMOTE (1<<9)
 #endif

 uint32_t ir_cpuinfo(void);
@@ -1011,14 +1114,15 @@ IR_ALWAYS_INLINE void *ir_jit_compile(ir_ctx *ctx, int opt_level, size_t *size)
 			// IR_ASSERT(0 && "IR_OPT_FOLDING is incompatible with -O0");
 			return NULL;
 		}
-		ctx->flags &= ~(IR_OPT_CFG | IR_OPT_CODEGEN);
+		ctx->flags &= ~(IR_OPT_CFG | IR_OPT_CODEGEN | IR_OPT_TAILCALL);

 		ir_build_def_use_lists(ctx);

 		if (!ir_build_cfg(ctx)
 		 || !ir_match(ctx)
 		 || !ir_assign_virtual_registers(ctx)
-		 || !ir_compute_dessa_moves(ctx)) {
+		 || !ir_compute_dessa_moves(ctx)
+		 || !ir_reg_alloc_simple(ctx)) {
 			return NULL;
 		}

@@ -1028,7 +1132,7 @@ IR_ALWAYS_INLINE void *ir_jit_compile(ir_ctx *ctx, int opt_level, size_t *size)
 			// IR_ASSERT(0 && "IR_OPT_FOLDING must be set in ir_init() for -O1 and -O2");
 			return NULL;
 		}
-		ctx->flags |= IR_OPT_CFG | IR_OPT_CODEGEN;
+		ctx->flags |= IR_OPT_CFG | IR_OPT_CODEGEN | IR_OPT_TAILCALL;

 		ir_build_def_use_lists(ctx);

diff --git a/ext/opcache/jit/ir/ir_aarch64.dasc b/ext/opcache/jit/ir/ir_aarch64.dasc
index fc4bb84f1e0..539108dec9a 100644
--- a/ext/opcache/jit/ir/ir_aarch64.dasc
+++ b/ext/opcache/jit/ir/ir_aarch64.dasc
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Aarch64 native code generator based on DynAsm)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -38,7 +38,7 @@ IR_ALWAYS_INLINE ir_mem IR_MEM(ir_reg base, int32_t offset, ir_reg index, int32_
 	IR_ASSERT(base == IR_REG_NONE || (base >= IR_REG_GP_FIRST && base <= IR_REG_GP_LAST));
 	IR_ASSERT(index == IR_REG_NONE || (index >= IR_REG_GP_FIRST && index <= IR_REG_GP_LAST));
 	IR_ASSERT(index == IR_REG_NONE || offset == 0);
-	IR_ASSERT(shift == 0); // TODO: ???
+	IR_ASSERT(shift == 0 || (shift >= 1 && shift <= 3 && index != IR_REG_NONE));
 #ifdef IR_DEBUG
 	mem.v =
 #else
@@ -232,10 +232,14 @@ typedef struct _ir_aarch64_sysv_va_list {
 	#name64,
 #define IR_GP_REG_NAME32(code, name64, name32) \
 	#name32,
-#define IR_FP_REG_NAME(code, name64, name32, name16, name8) \
+#define IR_GP_REG_NAME_VEC(code, name64, name32) \
+	NULL,
+#define IR_FP_REG_NAME(code, name64, name32, name16, name8, name_vec) \
 	#name64,
-#define IR_FP_REG_NAME32(code, name64, name32, name16, name8) \
+#define IR_FP_REG_NAME32(code, name64, name32, name16, name8, name_vec) \
 	#name32,
+#define IR_FP_REG_NAME_VEC(code, name64, name32, name16, name8, name_vec) \
+	#name_vec,

 static const char *_ir_reg_name[] = {
 	IR_GP_REGS(IR_GP_REG_NAME)
@@ -249,6 +253,11 @@ static const char *_ir_reg_name32[IR_REG_NUM] = {
 	IR_FP_REGS(IR_FP_REG_NAME32)
 };

+static const char *_ir_reg_name_vec[IR_REG_NUM] = {
+	IR_GP_REGS(IR_GP_REG_NAME_VEC)
+	IR_FP_REGS(IR_FP_REG_NAME_VEC)
+};
+
 const char *ir_reg_name(int8_t reg, ir_type type)
 {
 	if (reg >= IR_REG_NUM) {
@@ -259,13 +268,23 @@ const char *ir_reg_name(int8_t reg, ir_type type)
 	if (type == IR_VOID) {
 		type = (reg < IR_REG_FP_FIRST) ? IR_ADDR : IR_DOUBLE;
 	}
-	if (ir_type_size[type] == 8) {
+	if (IR_IS_TYPE_VECTOR(type)) {
+		return _ir_reg_name_vec[reg];
+	} else if (ir_type_size[type] == 8) {
 		return _ir_reg_name[reg];
 	} else {
 		return _ir_reg_name32[reg];
 	}
 }

+void ir_dump_reg(const ir_ctx *ctx, int8_t reg, ir_ref ref, bool store, FILE *f)
+{
+	if (reg != IR_REG_NONE) {
+		fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[ref].type),
+			(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? (store ? ":store" : ":load") : "");
+	}
+}
+
 /* Calling Conventions */
 #define IR_REG_SCRATCH_AARCH64         IR_REG_SET_1

@@ -287,12 +306,18 @@ const ir_call_conv_dsc ir_call_conv_aarch64_sysv = {
 	0,           /* shadow_store_size       */
 	8,           /* int_param_regs_count    */
 	8,           /* fp_param_regs_count     */
-	IR_REG_X0 ,  /* int_ret_reg             */
+	8,           /* vecto_param_regs_count  */
+	IR_REG_X0,   /* int_ret_reg             */
+	IR_REG_X1,   /* int_ret2_reg (up to X7) */
 	IR_REG_V0,   /* fp_ret_reg              */
+	IR_REG_V1,   /* fp_ret2_reg (up to V7)  */
+	IR_REG_V0,   /* vector_ret_reg          */
+	IR_REG_V1,   /* vector_ret2_reg         */
 	IR_REG_NONE, /* fp_varargs_reg          */
 	IR_REG_SCRATCH_AARCH64,
 	(const int8_t[8]){IR_REG_X0, IR_REG_X1, IR_REG_X2, IR_REG_X3, IR_REG_X4, IR_REG_X5, IR_REG_X6, IR_REG_X7},
 	(const int8_t[8]){IR_REG_V0, IR_REG_V1, IR_REG_V2, IR_REG_V3, IR_REG_V4, IR_REG_V5, IR_REG_V6, IR_REG_V7},
+	(const int8_t[8]){IR_REG_V0, IR_REG_V1, IR_REG_V2, IR_REG_V3, IR_REG_V4, IR_REG_V5, IR_REG_V6, IR_REG_V7},
 	IR_REGSET_INTERVAL(IR_REG_X19, IR_REG_X30) | IR_REGSET_INTERVAL(IR_REG_V8, IR_REG_V15),

 };
@@ -305,12 +330,18 @@ const ir_call_conv_dsc ir_call_conv_aarch64_darwin = {
 	0,           /* shadow_store_size       */
 	8,           /* int_param_regs_count    */
 	8,           /* fp_param_regs_count     */
+	8,           /* vector_param_regs_count */
 	IR_REG_X0 ,  /* int_ret_reg             */
+	IR_REG_X1,   /* int_ret2_reg (up to X7) */
 	IR_REG_V0,   /* fp_ret_reg              */
+	IR_REG_V1,   /* fp_ret2_reg (up to V7)  */
+	IR_REG_V0,   /* vector_ret_reg          */
+	IR_REG_V1,   /* vector_ret2_reg         */
 	IR_REG_NONE, /* fp_varargs_reg          */
 	IR_REG_SCRATCH_AARCH64,
 	(const int8_t[8]){IR_REG_X0, IR_REG_X1, IR_REG_X2, IR_REG_X3, IR_REG_X4, IR_REG_X5, IR_REG_X6, IR_REG_X7},
 	(const int8_t[8]){IR_REG_V0, IR_REG_V1, IR_REG_V2, IR_REG_V3, IR_REG_V4, IR_REG_V5, IR_REG_V6, IR_REG_V7},
+	(const int8_t[8]){IR_REG_V0, IR_REG_V1, IR_REG_V2, IR_REG_V3, IR_REG_V4, IR_REG_V5, IR_REG_V6, IR_REG_V7},
 	IR_REGSET_INTERVAL(IR_REG_X19, IR_REG_X30) | IR_REGSET_INTERVAL(IR_REG_V8, IR_REG_V15),

 };
@@ -323,8 +354,13 @@ const ir_call_conv_dsc ir_call_conv_aarch64_preserve_none = {
 	0,           /* shadow_store_size       */
 	23,          /* int_param_regs_count    */
 	8,           /* fp_param_regs_count     */
+	8,           /* vector_param_regs_count */
 	IR_REG_X0 ,  /* int_ret_reg             */
+	IR_REG_X1,   /* int_ret2_reg (up to X7) */
 	IR_REG_V0,   /* fp_ret_reg              */
+	IR_REG_V1,   /* fp_ret2_reg (up to V7)  */
+	IR_REG_V0,   /* vector_ret_reg          */
+	IR_REG_V1,   /* vector_ret2_reg         */
 	IR_REG_NONE, /* fp_varargs_reg          */
 	IR_REG_ALL,
 	(const int8_t[23]){IR_REG_X20, IR_REG_X21, IR_REG_X22, IR_REG_X23, IR_REG_X24, IR_REG_X25, IR_REG_X26, IR_REG_X27,
@@ -332,6 +368,7 @@ const ir_call_conv_dsc ir_call_conv_aarch64_preserve_none = {
 	                   IR_REG_X0, IR_REG_X1, IR_REG_X2, IR_REG_X3, IR_REG_X4, IR_REG_X5, IR_REG_X6, IR_REG_X7,
 	                   IR_REG_X10, IR_REG_X11, IR_REG_X12, IR_REG_X13, IR_REG_X14, IR_REG_X9},
 	(const int8_t[8]){IR_REG_V0, IR_REG_V1, IR_REG_V2, IR_REG_V3, IR_REG_V4, IR_REG_V5, IR_REG_V6, IR_REG_V7},
+	(const int8_t[8]){IR_REG_V0, IR_REG_V1, IR_REG_V2, IR_REG_V3, IR_REG_V4, IR_REG_V5, IR_REG_V6, IR_REG_V7},
 	IR_REGSET_EMPTY,

 };
@@ -375,6 +412,30 @@ const ir_call_conv_dsc ir_call_conv_aarch64_preserve_none = {
 	_(RETURN_INT)          \
 	_(RETURN_FP)           \
 	_(IGOTO_DUP)           \
+	_(TLS_LOAD)            \
+	_(TLS_STORE)           \
+
+#if IR_SIMD
+# define IR_RULES_SIMD(_)  \
+	_(VECTOR_OP)           \
+	_(VECTOR_BINOP)        \
+	_(VECTOR_EXT)          \
+	_(VECTOR_TRUNC)        \
+	_(VECTOR_FP2FP)        \
+	_(VECTOR_FP2INT)       \
+	_(VECTOR_INT2FP)       \
+	_(SHUFFLE_DUP)         \
+	_(SHUFFLE_REV2)        \
+	_(SHUFFLE_EXT)         \
+	_(SHUFFLE_TRN)         \
+	_(SHUFFLE_ZIP)         \
+	_(SHUFFLE_UZP)         \
+	_(SHUFFLE_1EXT)        \
+	_(SHUFFLE_1TRN)        \
+	_(SHUFFLE_1ZIP)        \
+	_(SHUFFLE_1UZP)        \
+
+#endif

 #define IR_RULE_ENUM(name) IR_ ## name,

@@ -383,6 +444,9 @@ const ir_call_conv_dsc ir_call_conv_aarch64_preserve_none = {
 enum _ir_rule {
 	IR_FIRST_RULE = IR_LAST_OP,
 	IR_RULES(IR_RULE_ENUM)
+#if IR_SIMD
+	IR_RULES_SIMD(IR_RULE_ENUM)
+#endif
 	IR_LAST_RULE
 };

@@ -390,6 +454,9 @@ enum _ir_rule {
 const char *ir_rule_name[IR_LAST_OP] = {
 	NULL,
 	IR_RULES(IR_RULE_NAME)
+#if IR_SIMD
+	IR_RULES_SIMD(IR_RULE_NAME)
+#endif
 	NULL
 };

@@ -403,6 +470,7 @@ int ir_get_target_constraints(ir_ctx *ctx, ir_ref ref, ir_target_constraints *co
 	const ir_proto_t *proto;
 	const ir_call_conv_dsc *cc;
 	ir_ref next;
+	ir_type type;

 	constraints->def_reg = IR_REG_NONE;
 	constraints->hints_count = 0;
@@ -647,6 +715,7 @@ int ir_get_target_constraints(ir_ctx *ctx, ir_ref ref, ir_target_constraints *co
 			n++;
 			break;
 		case IR_ARGVAL:
+			flags = IR_OP1_SHOULD_BE_IN_REG;
 			/* memcpy() clobbers all scratch registers */
 			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_SCRATCH_AARCH64, IR_DEF_SUB_REF - IR_SUB_REFS_COUNT, IR_USE_SUB_REF);
 			n = 1;
@@ -668,6 +737,11 @@ int ir_get_target_constraints(ir_ctx *ctx, ir_ref ref, ir_target_constraints *co
 			break;
 		case IR_TAILCALL:
 			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op2)
+			 && ctx->ir_base[insn->op2].op == IR_FUNC
+			 && ctx->func_name == ctx->ir_base[insn->op2].val.name) {
+				ctx->flags2 |= IR_RECURSIVE_TAILCALL;
+			}
 			if (insn->inputs_count > 2) {
 				proto = ir_call_proto(ctx, insn);
 				cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
@@ -681,19 +755,6 @@ get_arg_hints:
 			}
 			flags = IR_USE_SHOULD_BE_IN_REG | IR_OP2_MUST_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
 			break;
-		case IR_IGOTO:
-			insn = &ctx->ir_base[ref];
-			if (ctx->ir_base[insn->op1].op == IR_MERGE || ctx->ir_base[insn->op1].op == IR_LOOP_BEGIN) {
-				ir_insn *merge = &ctx->ir_base[insn->op1];
-				ir_ref *p, n = merge->inputs_count;
-
-				for (p = merge->ops + 1; n > 0; p++, n--) {
-					ir_ref input = *p;
-					IR_ASSERT(ctx->ir_base[input].op == IR_END || ctx->ir_base[input].op == IR_LOOP_END);
-					ctx->rules[input] = IR_IGOTO_DUP;
-				}
-			}
-			return insn->op;
 		case IR_COND:
 			insn = &ctx->ir_base[ref];
 			n = 0;
@@ -793,6 +854,152 @@ get_arg_hints:
 				n++;
 			}
 			break;
+		case IR_TLS_LOAD:
+			insn = &ctx->ir_base[ref];
+			flags = IR_USE_MUST_BE_IN_REG;
+			if (!IR_IS_TYPE_INT(insn->type)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(3, IR_ADDR, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			break;
+		case IR_TLS_STORE:
+			insn = &ctx->ir_base[ref];
+			flags = IR_OP3_MUST_BE_IN_REG;
+			if (IR_IS_CONST_REF(insn->op3)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(3, ctx->ir_base[insn->op3].type, IR_LOAD_SUB_REF, IR_USE_SUB_REF);
+				n++;
+			}
+			constraints->tmp_regs[n] = IR_TMP_REG(0, IR_ADDR, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n++;
+			break;
+#if IR_SIMD
+		case IR_SPLAT:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			break;
+		case IR_EXTRACT:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			if (!IR_IS_CONST_REF(insn->op2)) {
+				/* extract through stack */
+				ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+			}
+			break;
+		case IR_REPLACE:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_MUST_BE_IN_REG | IR_OP3_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op3)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(3, ctx->ir_base[insn->op3].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			if (!IR_IS_CONST_REF(insn->op2)) {
+				/* replace through stack */
+				ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+			}
+			break;
+		case IR_SHUFFLE_DUP:
+		case IR_SHUFFLE_REV2:
+		case IR_SHUFFLE_EXT:
+		case IR_SHUFFLE_TRN:
+		case IR_SHUFFLE_ZIP:
+		case IR_SHUFFLE_UZP:
+		case IR_SHUFFLE_1EXT:
+		case IR_SHUFFLE_1TRN:
+		case IR_SHUFFLE_1ZIP:
+		case IR_SHUFFLE_1UZP:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op2].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			break;
+		case IR_SHUFFLE:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG | IR_OP3_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op2].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			if (IR_IS_CONST_REF(insn->op3)) {
+				/* shuffle through sequence of moves */
+				flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+			} else if (insn->op1 == insn->op2
+					&& ir_type_size[IR_VECTOR_BASE_TYPE(insn->type)] == 1
+					&& ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op3].type)] == 1) {
+				/* shuffle through TBL */
+			} else {
+				/* shuffle through stack memory */
+				constraints->tmp_regs[n] = IR_TMP_REG(4, IR_U64, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			break;
+		case IR_VECTOR_OP:
+		case IR_VECTOR_EXT:
+		case IR_VECTOR_TRUNC:
+		case IR_VECTOR_FP2FP:
+		case IR_VECTOR_FP2INT:
+		case IR_VECTOR_INT2FP:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			break;
+		case IR_VECTOR_BINOP:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+				type = ctx->ir_base[insn->op1].type;
+			} else {
+				type = insn->type;
+			}
+
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op2 != insn->op1) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			if (insn->op == IR_SHR || insn->op == IR_SAR) {
+				constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			} else if (insn->op == IR_SHL && IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			} else if (insn->op == IR_MUL
+					&& (IR_VECTOR_BASE_TYPE(type) == IR_I64 || IR_VECTOR_BASE_TYPE(type) == IR_U64)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(3, IR_I64, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				constraints->tmp_regs[n] = IR_TMP_REG(4, IR_I64, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			} else if (insn->op == IR_MOD || (insn->op == IR_DIV && IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(type)))) {
+				constraints->tmp_regs[n] = IR_TMP_REG(3, IR_I64, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				constraints->tmp_regs[n] = IR_TMP_REG(4, IR_I64, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			break;
+#endif
 	}
 	constraints->tmps_count = n;

@@ -831,6 +1038,11 @@ static void ir_match_fuse_addr(ir_ctx *ctx, ir_ref addr_ref, ir_type type)
 	}
 }

+static uint32_t ir_match_builtin_call(ir_ctx *ctx, const ir_insn *func)
+{
+	return 0;
+}
+
 static bool all_usages_are_fusable(ir_ctx *ctx, ir_ref ref)
 {
 	ir_insn *insn = &ctx->ir_base[ref];
@@ -858,6 +1070,192 @@ static bool all_usages_are_fusable(ir_ctx *ctx, ir_ref ref)
 	return 0;
 }

+bool ir_may_fuse_tls_addr(ir_ctx *ctx, ir_ref ref)
+{
+	ir_use_list *use_list = &ctx->use_lists[ref];
+	ir_ref n, *p;
+
+	if (use_list->count == 2 || (ctx->rules[ref] & IR_FUSED)) {
+		return 1;
+	}
+	n = use_list->count;
+	for (p = ctx->use_edges + use_list->refs; n > 0; p++, n--) {
+		ir_ref use = *p;
+		ir_op op = ctx->ir_base[use].op;
+		if (op == IR_LOAD || op == IR_LOAD_v) {
+			/* pass */
+		} else if (op == IR_STORE || op == IR_STORE_v) {
+			if (ctx->ir_base[use].op3 == ref) {
+				return 0;
+			}
+		} else if (ctx->ir_base[use].op1 == ref && (ir_op_flags[op] & (IR_OP_FLAG_CONTROL|IR_OP_FLAG_MEM))) {
+			/* ignore control link */
+		} else {
+			return 0;
+		}
+	}
+	return 1;
+}
+
+#if IR_SIMD
+# define IR_SHUFFLE_MASK(i) (p[(i)*s])
+
+#define MAY_BE_DUP    (1<<0)   // 0000 1111 2222 3333 4444 5555 6666 7777
+#define MAY_BE_REV2   (1<<1)   // 1032
+
+#define MAY_BE_EXT    (1<<2)   // 1234 2345 3456 5670 6701 7012
+#define MAY_BE_TRN    (1<<3)   // 0426 1537 4062 5173
+#define MAY_BE_ZIP    (1<<4)   // 0415 2367 4051 6273
+#define MAY_BE_UZP    (1<<5)   // 0246 1357 4602 5713
+
+#define MAY_BE_1EXT   (1<<6)   // 1230 2301 3012 5674 6745 7456
+#define MAY_BE_1TRN   (1<<7)   // 0022 1133 4466 5577
+#define MAY_BE_1ZIP   (1<<8)   // 0011 2233 4455 6677
+#define MAY_BE_1UZP   (1<<9)   // 0202 1313 4646 5757
+
+static uint32_t ir_match_shuffle(ir_ctx *ctx, const ir_insn *insn)
+{
+	if (IR_IS_CONST_REF(insn->op3)) {
+		ir_insn *op3_insn = &ctx->ir_base[insn->op3];
+		IR_ASSERT(IR_IS_TYPE_VECTOR(insn->type) && IR_IS_TYPE_VECTOR(op3_insn->type));
+		IR_ASSERT(insn->type == ctx->ir_base[insn->op1].type && insn->type == ctx->ir_base[insn->op2].type);
+		if (IR_VECTOR_SIZE(insn->type) == 16) {
+			uint32_t n = IR_VECTOR_LENGTH(op3_insn->type);
+			uint32_t element_size = ir_type_size[IR_VECTOR_BASE_TYPE(insn->type)];
+
+			if (n * element_size == 16) {
+				int8_t *p = ir_long_const_ptr(ctx, insn->op3);
+				uint32_t s = ir_type_size[IR_VECTOR_BASE_TYPE(op3_insn->type)];
+				uint32_t mod = IR_VECTOR_LENGTH(insn->type);
+				uint32_t mask, prev, next, i;
+
+				if (insn->op1 != insn->op2) {
+					mod += IR_VECTOR_LENGTH(insn->type);
+				}
+				prev = IR_SHUFFLE_MASK(0) % mod;
+				if (prev == 0 || prev == n) {
+					mask = MAY_BE_DUP | MAY_BE_EXT | MAY_BE_TRN | MAY_BE_ZIP | MAY_BE_UZP;
+				} else if (prev == 1 || prev == n + 1) {
+					mask = MAY_BE_DUP | MAY_BE_EXT | MAY_BE_TRN | MAY_BE_UZP;
+					if (element_size != 8) {
+						mask |= MAY_BE_REV2;
+					}
+				} else if (prev == n / 2 || prev == n + n / 2) {
+					mask = MAY_BE_DUP | MAY_BE_EXT | MAY_BE_ZIP;
+				} else {
+					mask = MAY_BE_DUP | MAY_BE_EXT;
+				}
+				if (insn->op1 != insn->op2) {
+					mask = mask | ((mask & (MAY_BE_EXT|MAY_BE_TRN|MAY_BE_ZIP|MAY_BE_UZP)) << 4);
+				}
+				for (i = 1; i < n; i++) {
+					next = IR_SHUFFLE_MASK(i) % mod;
+					if (mask & MAY_BE_DUP) {
+						if (next != prev) {
+							mask &= ~MAY_BE_DUP;
+							if (!mask) break;
+						}
+					}
+					if (mask & MAY_BE_REV2) {
+						if (i % 2 == 1) {
+							if (next != i - 1 + ((prev < n) ? 0 : n)) {
+								mask &= ~MAY_BE_REV2;
+								if (!mask) break;
+							}
+						} else {
+							if (next != i + 1 + ((prev < n) ? 0 : n)) {
+								mask &= ~MAY_BE_REV2;
+								if (!mask) break;
+							}
+						}
+					}
+					if (mask & MAY_BE_EXT) {
+						if (next != (prev + 1) % mod) {
+							mask &= ~MAY_BE_EXT;
+							if (!mask) break;
+						}
+					}
+					if (mask & MAY_BE_1EXT) {
+						if (next != ((prev + 1) % n) + ((prev < n) ? 0 : n)) {
+							mask &= ~MAY_BE_1EXT;
+							if (!mask) break;
+						}
+					}
+					if (mask & MAY_BE_TRN) {
+						if (i % 2 == 1) {
+							if (next != (prev + n) % mod) {
+								mask &= ~MAY_BE_TRN;
+								if (!mask) break;
+							}
+						} else {
+							if (next != (prev + n + 2) % mod) {
+								mask &= ~MAY_BE_TRN;
+								if (!mask) break;
+							}
+						}
+					}
+					if (mask & MAY_BE_1TRN) {
+						if (i % 2 == 1) {
+							if (next != (prev + n) % n + ((prev < n) ? 0 : n)) {
+								mask &= ~MAY_BE_1TRN;
+								if (!mask) break;
+							}
+						} else {
+							if (next != (prev + n + 2) % n + ((prev < n) ? 0 : n)) {
+								mask &= ~MAY_BE_1TRN;
+								if (!mask) break;
+							}
+						}
+					}
+					if (mask & MAY_BE_ZIP) {
+						if (i % 2 == 1) {
+							if (next != (prev + n) % mod) {
+								mask &= ~MAY_BE_ZIP;
+								if (!mask) break;
+							}
+						} else {
+							if (next != (prev + n) % mod + 1) {
+								mask &= ~MAY_BE_ZIP;
+								if (!mask) break;
+							}
+						}
+					}
+					if (mask & MAY_BE_1ZIP) {
+						if (i % 2 == 1) {
+							if (next != (prev + n) % n + ((prev < n) ? 0 : n)) {
+								mask &= ~MAY_BE_1ZIP;
+								if (!mask) break;
+							}
+						} else {
+							if (next != (prev + n) % n + 1 + ((prev < n) ? 0 : n)) {
+								mask &= ~MAY_BE_1ZIP;
+								if (!mask) break;
+							}
+						}
+					}
+					if (mask & MAY_BE_UZP) {
+						if (next != (prev + 2) % mod) {
+							mask &= ~MAY_BE_UZP;
+							if (!mask) break;
+						}
+					}
+					if (mask & MAY_BE_1UZP) {
+						if (next != (prev + 2) % n + ((prev < n) ? 0 : n)) {
+							mask &= ~MAY_BE_1UZP;
+							if (!mask) break;
+						}
+					}
+					prev = next;
+				}
+				if (mask) {
+					return IR_SHUFFLE_DUP + ir_ntz(mask);
+				}
+			}
+		}
+	}
+	return IR_SHUFFLE;
+}
+#endif

 static uint32_t ir_match_insn(ir_ctx *ctx, ir_ref ref)
 {
@@ -877,36 +1275,52 @@ static uint32_t ir_match_insn(ir_ctx *ctx, ir_ref ref)
 		case IR_UGT:
 			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
 				return IR_CMP_INT;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op1].type)) {
+				return IR_VECTOR_BINOP;
+#endif
 			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(ctx->ir_base[insn->op1].type));
 				return IR_CMP_FP;
 			}
 			break;
 		case IR_ORDERED:
 		case IR_UNORDERED:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op1].type)) {
+				return IR_VECTOR_BINOP;
+			}
+#endif
 			return IR_CMP_FP;
 		case IR_ADD:
 		case IR_SUB:
 			if (IR_IS_TYPE_INT(insn->type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+				if (IR_IS_CONST_REF(insn->op2)) {
 					op2_insn = &ctx->ir_base[insn->op2];
 					if (IR_IS_SYM_CONST(op2_insn->op)) {
 						/* pass */
 					} else if (IR_IS_CONST_REF(insn->op1)) {
 						// const
 					} else if (op2_insn->val.i64 == 0) {
-						// return IR_COPY_INT;
+						return IR_COPY_INT | IR_MAY_REUSE;
 					}
 				}
 binop_int:
 				return IR_BINOP_INT;
 			} else {
 binop_fp:
+#if IR_SIMD
+				if (IR_IS_TYPE_VECTOR(insn->type)) {
+					return IR_VECTOR_BINOP;
+				}
+#endif
+				IR_ASSERT(IR_IS_TYPE_FP(insn->type));
 				return IR_BINOP_FP;
 			}
 			break;
 		case IR_MUL:
 			if (IR_IS_TYPE_INT(insn->type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+				if (IR_IS_CONST_REF(insn->op2)) {
 					op2_insn = &ctx->ir_base[insn->op2];
 					if (IR_IS_SYM_CONST(op2_insn->op)) {
 						/* pass */
@@ -915,7 +1329,7 @@ binop_fp:
 					} else if (op2_insn->val.u64 == 0) {
 						// 0
 					} else if (op2_insn->val.u64 == 1) {
-						// return IR_COPY_INT;
+						return IR_COPY_INT | IR_MAY_REUSE;
 					} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
 						return IR_MUL_PWR2;
 					}
@@ -934,14 +1348,14 @@ binop_fp:
 			goto binop_int;
 		case IR_DIV:
 			if (IR_IS_TYPE_INT(insn->type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+				if (IR_IS_CONST_REF(insn->op2)) {
 					op2_insn = &ctx->ir_base[insn->op2];
 					if (IR_IS_SYM_CONST(op2_insn->op)) {
 						/* pass */
 					} else if (IR_IS_CONST_REF(insn->op1)) {
 						// const
 					} else if (op2_insn->val.u64 == 1) {
-						// return IR_COPY_INT;
+						return IR_COPY_INT | IR_MAY_REUSE;
 					} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
 						if (IR_IS_TYPE_UNSIGNED(insn->type)) {
 							return IR_DIV_PWR2;
@@ -956,12 +1370,19 @@ binop_fp:
 			}
 			break;
 		case IR_MOD:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
 				op2_insn = &ctx->ir_base[insn->op2];
 				if (IR_IS_SYM_CONST(op2_insn->op)) {
 					/* pass */
 				} else if (IR_IS_CONST_REF(insn->op1)) {
 					// const
+				} else if (op2_insn->val.u64 == 1) {
+					// 0
 				} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
 					if (IR_IS_TYPE_UNSIGNED(insn->type)) {
 						return IR_MOD_PWR2;
@@ -971,8 +1392,14 @@ binop_fp:
 				}
 			}
 			return IR_BINOP_INT;
-		case IR_BSWAP:
 		case IR_NOT:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_OP;
+			}
+			IR_FALLTHROUGH;
+#endif
+		case IR_BSWAP:
 		case IR_CTLZ:
 		case IR_CTTZ:
 			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
@@ -981,25 +1408,40 @@ binop_fp:
 		case IR_ABS:
 			if (IR_IS_TYPE_INT(insn->type)) {
 				return IR_OP_INT;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_OP;
+#endif
 			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(insn->type));
 				return IR_OP_FP;
 			}
 		case IR_OR:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
 				op2_insn = &ctx->ir_base[insn->op2];
 				if (IR_IS_SYM_CONST(op2_insn->op)) {
 					/* pass */
 				} else if (IR_IS_CONST_REF(insn->op1)) {
 					// const
 				} else if (op2_insn->val.i64 == 0) {
-					// return IR_COPY_INT;
+					return IR_COPY_INT | IR_MAY_REUSE;
 				} else if (op2_insn->val.i64 == -1) {
 					// -1
 				}
 			}
 			goto binop_int;
 		case IR_AND:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
 				op2_insn = &ctx->ir_base[insn->op2];
 				if (IR_IS_SYM_CONST(op2_insn->op)) {
 					/* pass */
@@ -1008,12 +1450,17 @@ binop_fp:
 				} else if (op2_insn->val.i64 == 0) {
 					// 0
 				} else if (op2_insn->val.i64 == -1) {
-					// return IR_COPY_INT;
+					return IR_COPY_INT | IR_MAY_REUSE;
 				}
 			}
 			goto binop_int;
 		case IR_XOR:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
 				op2_insn = &ctx->ir_base[insn->op2];
 				if (IR_IS_SYM_CONST(op2_insn->op)) {
 					/* pass */
@@ -1023,23 +1470,26 @@ binop_fp:
 			}
 			goto binop_int;
 		case IR_SHL:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP;
+			}
+#endif
 			if (IR_IS_CONST_REF(insn->op2)) {
-				if (ctx->flags & IR_OPT_CODEGEN) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_SYM_CONST(op2_insn->op)) {
-						/* pass */
-					} else if (IR_IS_CONST_REF(insn->op1)) {
-						// const
-					} else if (op2_insn->val.u64 == 0) {
-						// return IR_COPY_INT;
-					} else if (ir_type_size[insn->type] >= 4) {
-						if (op2_insn->val.u64 == 1) {
-							// lea [op1*2]
-						} else if (op2_insn->val.u64 == 2) {
-							// lea [op1*4]
-						} else if (op2_insn->val.u64 == 3) {
-							// lea [op1*8]
-						}
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.u64 == 0) {
+					return IR_COPY_INT | IR_MAY_REUSE;
+				} else if (ir_type_size[insn->type] >= 4) {
+					if (op2_insn->val.u64 == 1) {
+						// lea [op1*2]
+					} else if (op2_insn->val.u64 == 2) {
+						// lea [op1*4]
+					} else if (op2_insn->val.u64 == 3) {
+						// lea [op1*8]
 					}
 				}
 				return IR_SHIFT_CONST;
@@ -1047,18 +1497,22 @@ binop_fp:
 			return IR_SHIFT;
 		case IR_SHR:
 		case IR_SAR:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP;
+			}
+			IR_FALLTHROUGH;
+#endif
 		case IR_ROL:
 		case IR_ROR:
 			if (IR_IS_CONST_REF(insn->op2)) {
-				if (ctx->flags & IR_OPT_CODEGEN) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_SYM_CONST(op2_insn->op)) {
-						/* pass */
-					} else if (IR_IS_CONST_REF(insn->op1)) {
-						// const
-					} else if (op2_insn->val.u64 == 0) {
-						// return IR_COPY_INT;
-					}
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.u64 == 0) {
+					return IR_COPY_INT | IR_MAY_REUSE;
 				}
 				return IR_SHIFT_CONST;
 			}
@@ -1094,6 +1548,19 @@ binop_fp:
 			}
 			ctx->flags2 |= IR_HAS_CALLS;
 			return IR_CALL;
+		case IR_IGOTO:
+			insn = &ctx->ir_base[ref];
+			if (ctx->ir_base[insn->op1].op == IR_MERGE || ctx->ir_base[insn->op1].op == IR_LOOP_BEGIN) {
+				ir_insn *merge = &ctx->ir_base[insn->op1];
+				ir_ref *p, n = merge->inputs_count;
+
+				for (p = merge->ops + 1; n > 0; p++, n--) {
+					ir_ref input = *p;
+					IR_ASSERT(ctx->ir_base[input].op == IR_END || ctx->ir_base[input].op == IR_LOOP_END);
+					ctx->rules[input] = IR_IGOTO_DUP;
+				}
+			}
+			return insn->op;
 		case IR_VAR:
 			return IR_STATIC_ALLOCA;
 		case IR_PARAM:
@@ -1113,6 +1580,10 @@ binop_fp:
 			return IR_ALLOCA;
 		case IR_LOAD:
 		case IR_LOAD_v:
+			if (ctx->ir_base[insn->op2].op == IR_TLS_ADDR && ir_may_fuse_tls_addr(ctx, insn->op2)) {
+				ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_TLS_ADDR;
+				return IR_TLS_LOAD;
+			}
 			ir_match_fuse_addr(ctx, insn->op2, insn->type);
 			if (IR_IS_TYPE_INT(insn->type)) {
 				return IR_LOAD_INT;
@@ -1122,6 +1593,10 @@ binop_fp:
 			break;
 		case IR_STORE:
 		case IR_STORE_v:
+			if (ctx->ir_base[insn->op2].op == IR_TLS_ADDR && ir_may_fuse_tls_addr(ctx, insn->op2)) {
+				ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_TLS_ADDR;
+				return IR_TLS_STORE;
+			}
 			ir_match_fuse_addr(ctx, insn->op2, ctx->ir_base[insn->op3].type);
 			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
 				return IR_STORE_INT;
@@ -1136,7 +1611,7 @@ binop_fp:
 			return IR_RLOAD;
 		case IR_RSTORE:
 			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && ir_in_same_block(ctx, insn->op2) && ctx->use_lists[insn->op2].count == 1) {
+				if (ir_in_same_block(ctx, insn->op2) && ctx->use_lists[insn->op2].count == 1) {
 					ir_insn *op_insn = &ctx->ir_base[insn->op2];

 					if (!ctx->rules[insn->op2]) {
@@ -1219,16 +1694,46 @@ binop_fp:
 				}
 			}
 			return insn->op;
-		case IR_VA_START:
-			ctx->flags2 |= IR_HAS_VA_START;
-			if ((ctx->ir_base[insn->op2].op == IR_ALLOCA) || (ctx->ir_base[insn->op2].op == IR_VADDR)) {
-				ir_use_list *use_list = &ctx->use_lists[insn->op2];
-				ir_ref *p, n = use_list->count;
-				for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
-					ir_insn *use_insn = &ctx->ir_base[*p];
-					if (use_insn->op == IR_VA_START || use_insn->op == IR_VA_END) {
-					} else if (use_insn->op == IR_VA_COPY) {
-						if (use_insn->op3 == insn->op2) {
+#if IR_SIMD
+		case IR_SEXT:
+		case IR_ZEXT:
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_EXT;
+			}
+			return insn->op;
+		case IR_TRUNC:
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_TRUNC;
+			}
+			return insn->op;
+		case IR_FP2FP:
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_FP2FP;
+			}
+			return insn->op;
+		case IR_FP2INT:
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_FP2INT;
+			}
+			return insn->op;
+		case IR_INT2FP:
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_INT2FP;
+			}
+			return insn->op;
+		case IR_SHUFFLE:
+			return ir_match_shuffle(ctx, insn);
+#endif
+		case IR_VA_START:
+			ctx->flags2 |= IR_HAS_VA_START;
+			if ((ctx->ir_base[insn->op2].op == IR_ALLOCA) || (ctx->ir_base[insn->op2].op == IR_VADDR)) {
+				ir_use_list *use_list = &ctx->use_lists[insn->op2];
+				ir_ref *p, n = use_list->count;
+				for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
+					ir_insn *use_insn = &ctx->ir_base[*p];
+					if (use_insn->op == IR_VA_START || use_insn->op == IR_VA_END) {
+					} else if (use_insn->op == IR_VA_COPY) {
+						if (use_insn->op3 == insn->op2) {
 							ctx->flags2 |= IR_HAS_VA_COPY;
 						}
 					} else if (use_insn->op == IR_VA_ARG) {
@@ -1427,6 +1932,7 @@ static void ir_emit_load_mem_int(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem m
 	ir_reg base_reg = IR_MEM_BASE(mem);
 	ir_reg index_reg = IR_MEM_INDEX(mem);
 	int32_t offset = IR_MEM_OFFSET(mem);
+	int32_t shift = IR_MEM_SHIFT(mem);

 	if (index_reg == IR_REG_NONE) {
 		if (aarch64_may_encode_addr_offset(offset, ir_type_size[type])) {
@@ -1468,19 +1974,40 @@ static void ir_emit_load_mem_int(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem m
 		default:
 			IR_ASSERT(0);
 		case 8:
-			|	ldr Rx(reg), [Rx(base_reg), Rx(index_reg)]
+			if (shift == 0) {
+				|	ldr Rx(reg), [Rx(base_reg), Rx(index_reg)]
+			} else {
+				IR_ASSERT(shift == 3);
+				|	ldr Rx(reg), [Rx(base_reg), Rx(index_reg), lsl #3]
+			}
 			break;
 		case 4:
-			|	ldr Rw(reg), [Rx(base_reg), Rx(index_reg)]
+			if (shift == 0) {
+				|	ldr Rw(reg), [Rx(base_reg), Rx(index_reg)]
+			} else {
+				IR_ASSERT(shift == 2);
+				|	ldr Rw(reg), [Rx(base_reg), Rx(index_reg), lsl #2]
+			}
 			break;
 		case 2:
 			if (IR_IS_TYPE_SIGNED(type)) {
-				|	ldrsh Rw(reg), [Rx(base_reg), Rx(index_reg)]
+				if (shift == 0) {
+					|	ldrsh Rw(reg), [Rx(base_reg), Rx(index_reg)]
+				} else {
+					IR_ASSERT(shift == 1);
+					|	ldrsh Rw(reg), [Rx(base_reg), Rx(index_reg), lsl #1]
+				}
 			} else {
-				|	ldrh Rw(reg), [Rx(base_reg), Rx(index_reg)]
+				if (shift == 0) {
+					|	ldrh Rw(reg), [Rx(base_reg), Rx(index_reg)]
+				} else {
+					IR_ASSERT(shift == 1);
+					|	ldrh Rw(reg), [Rx(base_reg), Rx(index_reg), lsl #1]
+				}
 			}
 			break;
 		case 1:
+			IR_ASSERT(shift == 0);
 			if (IR_IS_TYPE_SIGNED(type)) {
 				|	ldrsb Rw(reg), [Rx(base_reg), Rx(index_reg)]
 			} else {
@@ -1505,9 +2032,30 @@ static void ir_emit_load_imm_fp(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref sr
 		label = ir_get_const_label(ctx, src);
 		if (type == IR_DOUBLE) {
 			|	ldr Rd(reg-IR_REG_FP_FIRST), =>label
-		} else {
-			IR_ASSERT(type == IR_FLOAT);
+		} else if (type == IR_FLOAT) {
 			|	ldr Rs(reg-IR_REG_FP_FIRST), =>label
+#if IR_SIMD
+		} else if (IR_IS_TYPE_VECTOR(type)) {
+			uint32_t width = IR_VECTOR_SIZE(type);
+
+			switch (width) {
+				case 16:
+					|	ldr Rq(reg-IR_REG_FP_FIRST), =>label
+					break;
+				case 8:
+					|	ldr Rd(reg-IR_REG_FP_FIRST), =>label
+					break;
+				case 4:
+				case 2:
+				case 1:
+					|	ldr Rs(reg-IR_REG_FP_FIRST), =>label
+					break;
+				default:
+					IR_ASSERT(0 && "unsupported vector width");
+			}
+#endif
+		} else {
+			IR_ASSERT(0);
 		}
 	}
 }
@@ -1519,30 +2067,89 @@ static void ir_emit_load_mem_fp(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem me
 	ir_reg base_reg = IR_MEM_BASE(mem);
 	ir_ref index_reg = IR_MEM_INDEX(mem);
 	int32_t offset = IR_MEM_OFFSET(mem);
+	int32_t shift = IR_MEM_SHIFT(mem);

 	if (index_reg == IR_REG_NONE) {
-		if (aarch64_may_encode_addr_offset(offset, ir_type_size[type])) {
+		if (aarch64_may_encode_addr_offset(offset, ir_get_type_size(type))) {
 			if (type == IR_DOUBLE) {
 				|	ldr Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
-			} else {
-				IR_ASSERT(type == IR_FLOAT);
+			} else if (type == IR_FLOAT) {
 				|	ldr Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(type)) {
+				uint32_t width = IR_VECTOR_SIZE(type);
+
+				if (width == 16) {
+					//??? |	ldr Rq(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+					IR_ASSERT(reg >= IR_REG_FP_FIRST && reg <= IR_REG_FP_LAST);
+					IR_ASSERT(base_reg >= IR_REG_GP_FIRST && base_reg <= IR_REG_GP_LAST);
+					IR_ASSERT(offset >= 0 && offset <= 0xfff0 && (offset & 0xf) == 0);
+					uint32_t code = 0x3dc00000 | (reg-IR_REG_FP_FIRST) | (base_reg << 5) | ((offset >> 4) << 10);
+					|	.long code
+				} else if (width == 8) {
+					|	ldr Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+				} else if (width == 4) {
+					|	ldr Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+				} else if (width == 2) {
+					|	ldrh Rw(IR_REG_INT_TMP), [Rx(base_reg), #offset]
+					|	fmov Rs(reg-IR_REG_FP_FIRST), Rw(IR_REG_INT_TMP)
+				} else if (width == 1) {
+					|	ldrb Rw(IR_REG_INT_TMP), [Rx(base_reg), #offset]
+					|	fmov Rs(reg-IR_REG_FP_FIRST), Rw(IR_REG_INT_TMP)
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+#endif
+			} else {
+				IR_ASSERT(0);
 			}
+			return;
 		} else {
 			index_reg = IR_REG_INT_TMP; /* reserved temporary register */

 			ir_emit_load_imm_int(ctx, IR_ADDR, index_reg, offset);
 		}
-		return;
 	} else {
 		IR_ASSERT(offset == 0);
 	}

 	if (type == IR_DOUBLE) {
-		|	ldr Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		if (shift == 0) {
+			|	ldr Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else {
+			IR_ASSERT(shift == 3);
+			|	ldr Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg), lsl #3]
+		}
+	} else if (type == IR_FLOAT) {
+		if (shift == 0) {
+			|	ldr Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else {
+			IR_ASSERT(shift == 2);
+			|	ldr Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg), lsl #2]
+		}
+#if IR_SIMD
+	} else if (IR_IS_TYPE_VECTOR(type)) {
+		uint32_t width = IR_VECTOR_SIZE(type);
+
+		IR_ASSERT(shift == 0);
+		if (width == 16) {
+			|	ldr Rq(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 8) {
+			|	ldr Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 4) {
+			|	ldr Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 2) {
+			|	ldrh Rw(IR_REG_INT_TMP), [Rx(base_reg), Rx(index_reg)]
+			|	fmov Rs(reg-IR_REG_FP_FIRST), Rw(IR_REG_INT_TMP)
+		} else if (width == 1) {
+			|	ldrb Rw(IR_REG_INT_TMP), [Rx(base_reg), Rx(index_reg)]
+			|	fmov Rs(reg-IR_REG_FP_FIRST), Rw(IR_REG_INT_TMP)
+		} else {
+			IR_ASSERT(0 && "unsupported vector width");
+		}
+#endif
 	} else {
-		IR_ASSERT(type == IR_FLOAT);
-		|	ldr Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		IR_ASSERT(0);
 	}
 }

@@ -1654,6 +2261,7 @@ static void ir_emit_store_mem_int(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg
 	ir_reg base_reg = IR_MEM_BASE(mem);
 	ir_reg index_reg = IR_MEM_INDEX(mem);
 	int32_t offset = IR_MEM_OFFSET(mem);
+	int32_t shift = IR_MEM_SHIFT(mem);

 	if (index_reg == IR_REG_NONE) {
 		if (aarch64_may_encode_addr_offset(offset, ir_type_size[type])) {
@@ -1687,15 +2295,31 @@ static void ir_emit_store_mem_int(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg
 		default:
 			IR_ASSERT(0);
 		case 8:
-			|	str Rx(reg), [Rx(base_reg), Rx(index_reg)]
+			if (shift == 0) {
+				|	str Rx(reg), [Rx(base_reg), Rx(index_reg)]
+			} else {
+				IR_ASSERT(shift == 3);
+				|	str Rx(reg), [Rx(base_reg), Rx(index_reg), lsl #3]
+			}
 			break;
 		case 4:
-			|	str Rw(reg), [Rx(base_reg), Rx(index_reg)]
+			if (shift == 0) {
+				|	str Rw(reg), [Rx(base_reg), Rx(index_reg)]
+			} else {
+				IR_ASSERT(shift == 2);
+				|	str Rw(reg), [Rx(base_reg), Rx(index_reg), lsl #2]
+			}
 			break;
 		case 2:
-			|	strh Rw(reg), [Rx(base_reg), Rx(index_reg)]
+			if (shift == 0) {
+				|	strh Rw(reg), [Rx(base_reg), Rx(index_reg)]
+			} else {
+				IR_ASSERT(shift == 1);
+				|	strh Rw(reg), [Rx(base_reg), Rx(index_reg), lsl #1]
+			}
 			break;
 		case 1:
+			IR_ASSERT(shift == 0);
 			|	strb Rw(reg), [Rx(base_reg), Rx(index_reg)]
 			break;
 	}
@@ -1708,30 +2332,91 @@ static void ir_emit_store_mem_fp(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg r
 	ir_reg base_reg = IR_MEM_BASE(mem);
 	ir_reg index_reg = IR_MEM_INDEX(mem);
 	int32_t offset = IR_MEM_OFFSET(mem);
+	int32_t shift = IR_MEM_SHIFT(mem);

 	if (index_reg == IR_REG_NONE) {
-		if (aarch64_may_encode_addr_offset(offset, ir_type_size[type])) {
+		if (aarch64_may_encode_addr_offset(offset, ir_get_type_size(type))) {
 			if (type == IR_DOUBLE) {
 				|	str Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
-			} else {
-				IR_ASSERT(type == IR_FLOAT);
+			} else if (type == IR_FLOAT) {
 				|	str Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(type)) {
+				uint32_t width = IR_VECTOR_SIZE(type);
+
+				if (width == 16) {
+					//??? |	str Rq(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+					IR_ASSERT(reg >= IR_REG_FP_FIRST && reg <= IR_REG_FP_LAST);
+					IR_ASSERT(base_reg >= IR_REG_GP_FIRST && base_reg <= IR_REG_GP_LAST);
+					IR_ASSERT(offset >= 0 && offset <= 0xfff0 && (offset & 0xf) == 0);
+					uint32_t code = 0x3d800000 | (reg-IR_REG_FP_FIRST) | (base_reg << 5) | ((offset >> 4) << 10);
+					|	.long code
+				} else if (width == 8) {
+					|	str Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+				} else if (width == 4) {
+					|	str Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), #offset]
+				} else if (width == 2) {
+					|	fmov Rw(IR_REG_INT_TMP), Rs(reg-IR_REG_FP_FIRST)
+					|	strh Rw(IR_REG_INT_TMP), [Rx(base_reg), #offset]
+				} else if (width == 1) {
+					|	fmov Rw(IR_REG_INT_TMP), Rs(reg-IR_REG_FP_FIRST)
+					|	strb Rw(IR_REG_INT_TMP), [Rx(base_reg), #offset]
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+#endif
+			} else {
+				IR_ASSERT(0);
 			}
+			return;
 		} else {
 			index_reg = IR_REG_INT_TMP; /* reserved temporary register */

 			ir_emit_load_imm_int(ctx, IR_ADDR, index_reg, offset);
 		}
-		return;
 	} else {
 		IR_ASSERT(offset == 0);
 	}

 	if (type == IR_DOUBLE) {
-		|	str Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		if (shift == 0) {
+			|	str Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else {
+			IR_ASSERT(shift == 3);
+			|	str Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg), lsl #3]
+		}
+	} else if (type == IR_FLOAT) {
+		if (shift == 0) {
+			|	str Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else {
+			IR_ASSERT(shift == 2);
+			|	str Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg), lsl #2]
+		}
+#if IR_SIMD
+	} else if (IR_IS_TYPE_VECTOR(type)) {
+		uint32_t width = IR_VECTOR_SIZE(type);
+
+		IR_ASSERT(shift == 0);
+		if (width == 16) {
+			|	str Rq(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 8) {
+			|	str Rd(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 4) {
+			|	str Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 2) {
+			IR_ASSERT(index_reg != IR_REG_INT_TMP);
+			|	fmov Rw(IR_REG_INT_TMP), Rs(reg-IR_REG_FP_FIRST)
+			|	strh Rw(IR_REG_INT_TMP), [Rx(base_reg), Rx(index_reg)]
+		} else if (width == 1) {
+			IR_ASSERT(index_reg != IR_REG_INT_TMP);
+			|	fmov Rw(IR_REG_INT_TMP), Rs(reg-IR_REG_FP_FIRST)
+			|	strb Rw(IR_REG_INT_TMP), [Rx(base_reg), Rx(index_reg)]
+		} else {
+			IR_ASSERT(0 && "unsupported vector width");
+		}
+#endif
 	} else {
-		IR_ASSERT(type == IR_FLOAT);
-		|	str Rs(reg-IR_REG_FP_FIRST), [Rx(base_reg), Rx(index_reg)]
+		IR_ASSERT(0);
 	}
 }

@@ -1779,15 +2464,40 @@ static void ir_emit_mov_ext(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
 		|	mov Rw(dst), Rw(src)
 	}
 }
+
+#if IR_SIMD
+static uint32_t neon_MOV(ir_reg dst, ir_reg src)
+{
+	/* mov Rv(dst-IR_REG_FP_FIRST).16b, Rv(src-IR_REG_FP_FIRST).16b */
+	return 0x0ea01c00 | (dst-IR_REG_FP_FIRST) | ((src-IR_REG_FP_FIRST) << 5) |
+		((src-IR_REG_FP_FIRST) << 16) | (1<<30);
+}
+#endif
+
 static void ir_emit_fp_mov(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;

-	if (ir_type_size[type] == 8) {
+	if (type == IR_DOUBLE) {
 		|	fmov Rd(dst-IR_REG_FP_FIRST), Rd(src-IR_REG_FP_FIRST)
-	} else {
+	} else if (type == IR_FLOAT) {
 		|	fmov Rs(dst-IR_REG_FP_FIRST), Rs(src-IR_REG_FP_FIRST)
+#if IR_SIMD
+	} else if (IR_IS_TYPE_VECTOR(type)) {
+		uint32_t width = IR_VECTOR_SIZE(type);
+		if (width == 16) {
+			uint32_t code = neon_MOV(dst, src);
+			|	.long code
+		} else if (width == 8) {
+			|	fmov Rd(dst-IR_REG_FP_FIRST), Rd(src-IR_REG_FP_FIRST)
+		} else {
+			IR_ASSERT(width <= 4);
+			|	fmov Rs(dst-IR_REG_FP_FIRST), Rs(src-IR_REG_FP_FIRST)
+		}
+#endif
+	} else {
+		IR_ASSERT(0);
 	}
 }

@@ -2260,21 +2970,26 @@ static void ir_emit_min_max_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg op2_reg = ctx->regs[def][2];

-	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE && op2_reg != IR_REG_NONE);
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);

 	if (IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
 		ir_emit_load(ctx, type, op1_reg, op1);
 	}
+
+	if (op1 == op2) {
+		if (def_reg != op1_reg) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		}
+		goto done;
+	}
+
+	IR_ASSERT(op2_reg != IR_REG_NONE);
 	if (IR_REG_SPILLED(op2_reg)) {
 		op2_reg = IR_REG_NUM(op2_reg);
 		ir_emit_load(ctx, type, op2_reg, op2);
 	}

-	if (op1 == op2) {
-		return;
-	}
-
 	if (ir_type_size[type] == 8) {
 		|	cmp Rx(op1_reg), Rx(op2_reg)
 		if (insn->op == IR_MIN) {
@@ -2309,6 +3024,7 @@ static void ir_emit_min_max_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 		}
 	}

+done:
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
 		ir_emit_store(ctx, type, def, def_reg);
 	}
@@ -2701,6 +3417,7 @@ static void ir_emit_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg tmp_reg;
+	uint32_t type_bits = ir_type_size[type] * 8;

 	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
 	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
@@ -2710,6 +3427,25 @@ static void ir_emit_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 		op1_reg = IR_REG_NUM(op1_reg);
 		ir_emit_load(ctx, type, op1_reg, op1);
 	}
+	if (insn->op == IR_ROL || insn->op == IR_ROR) {
+		shift %= type_bits;
+	} else {
+		shift %= IR_MAX(type_bits, 32);
+	}
+	if (shift == 0) {
+		if (def_reg != op1_reg) {
+			|	ASM_REG_REG_OP mov, type, def_reg, op1_reg
+		}
+		goto done;
+	}
+	if (shift >= type_bits) {
+		if (insn->op == IR_SAR) {
+			shift = type_bits - 1;
+		} else {
+			|	mov Rw(def_reg), wzr
+			goto done;
+		}
+	}
 	switch (insn->op) {
 		default:
 			IR_ASSERT(0);
@@ -2771,6 +3507,8 @@ static void ir_emit_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 			}
 			break;
 	}
+
+done:
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
 		ir_emit_store(ctx, type, def, def_reg);
 	}
@@ -2780,7 +3518,7 @@ static void ir_emit_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_ref op1 = insn->op1;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
@@ -2789,19 +3527,19 @@ static void ir_emit_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)

 	if (IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+		ir_emit_load(ctx, src_type, op1_reg, op1);
 	}
 	if (insn->op == IR_NOT) {
-		if (insn->type == IR_BOOL) {
-			|	ASM_REG_IMM_OP cmp, type, op1_reg, 0
+		if (src_type == IR_BOOL) {
+			|	ASM_REG_IMM_OP cmp, src_type, op1_reg, 0
 			|	cset Rw(def_reg), eq
 		} else {
-			|	ASM_REG_REG_OP mvn, insn->type, def_reg, op1_reg
+			|	ASM_REG_REG_OP mvn, src_type, def_reg, op1_reg
 		}
 	} else if (insn->op == IR_NEG) {
-		|	ASM_REG_REG_OP neg, insn->type, def_reg, op1_reg
+		|	ASM_REG_REG_OP neg, src_type, def_reg, op1_reg
 	} else if (insn->op == IR_ABS) {
-		if (ir_type_size[type] == 8) {
+		if (ir_type_size[src_type] == 8) {
 			|	cmp Rx(op1_reg), #0
 			|	cneg Rx(def_reg), Rx(op1_reg), lt
 		} else {
@@ -2809,26 +3547,26 @@ static void ir_emit_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 			|	cneg Rw(def_reg), Rw(op1_reg), lt
 		}
 	} else if (insn->op == IR_CTLZ) {
-		if (ir_type_size[type] == 1) {
+		if (ir_type_size[src_type] == 1) {
 			|	and	Rw(def_reg), Rw(op1_reg), #0xff
 			|	clz Rw(def_reg), Rw(def_reg)
 			|	sub Rw(def_reg), Rw(def_reg), #24
-		} else if (ir_type_size[type] == 2) {
+		} else if (ir_type_size[src_type] == 2) {
 			|	and	Rw(def_reg), Rw(op1_reg), #0xffff
 			|	clz Rw(def_reg), Rw(def_reg)
 			|	sub Rw(def_reg), Rw(def_reg), #16
 		} else {
-			|	ASM_REG_REG_OP clz, type, def_reg, op1_reg
+			|	ASM_REG_REG_OP clz, src_type, def_reg, op1_reg
 		}
 	} else if (insn->op == IR_CTTZ) {
-		|	ASM_REG_REG_OP rbit, insn->type, def_reg, op1_reg
-		|	ASM_REG_REG_OP clz, insn->type, def_reg, def_reg
+		|	ASM_REG_REG_OP rbit, src_type, def_reg, op1_reg
+		|	ASM_REG_REG_OP clz, src_type, def_reg, def_reg
 	} else {
 		IR_ASSERT(insn->op == IR_BSWAP);
-		|	ASM_REG_REG_OP rev, insn->type, def_reg, op1_reg
+		|	ASM_REG_REG_OP rev, src_type, def_reg, op1_reg
 	}
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

@@ -2836,7 +3574,7 @@ static void ir_emit_ctpop(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_ref op1 = insn->op1;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
@@ -2848,9 +3586,9 @@ static void ir_emit_ctpop(ir_ctx *ctx, ir_ref def, ir_insn *insn)

 	if (IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+		ir_emit_load(ctx, src_type, op1_reg, op1);
 	}
-	switch (ir_type_size[insn->type]) {
+	switch (ir_type_size[src_type]) {
 		default:
 			IR_ASSERT(0);
 		case 1:
@@ -2881,7 +3619,7 @@ static void ir_emit_ctpop(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 			break;
 	}
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

@@ -2932,11 +3670,15 @@ static void ir_emit_binop_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg op2_reg = ctx->regs[def][2];

-	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE && op2_reg != IR_REG_NONE);
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
 	if (IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
 		ir_emit_load(ctx, type, op1_reg, op1);
 	}
+	if (op2_reg == IR_REG_NONE && op1 == op2) {
+		op2_reg = op1_reg;
+	}
+	IR_ASSERT(op2_reg != IR_REG_NONE);
 	if (IR_REG_SPILLED(op2_reg)) {
 		op2_reg = IR_REG_NUM(op2_reg);
 		if (op1 != op2) {
@@ -2997,6 +3739,10 @@ static void ir_emit_cmp_int_common(ir_ctx *ctx, ir_type type, ir_reg op1_reg, ir
 	dasm_State **Dst = &data->dasm_state;

 	IR_ASSERT(op1_reg != IR_REG_NONE);
+	if (op2_reg == IR_REG_NONE && op1 == op2) {
+		op2_reg = op1_reg;
+	}
+
 	if (ir_type_size[type] < 4) {
 		ir_emit_fix_type(ctx, type, op1_reg);
 	}
@@ -3139,6 +3885,9 @@ static ir_op ir_emit_cmp_fp_common(ir_ctx *ctx, ir_ref root, ir_ref cmp_ref, ir_
 		op1_reg = ctx->regs[cmp_ref][1];
 		op2_reg = ctx->regs[cmp_ref][2];
 	}
+	if (op2_reg == IR_REG_NONE && op1 == op2) {
+		op2_reg = op1_reg;
+	}

 	IR_ASSERT(op1_reg != IR_REG_NONE && op2_reg != IR_REG_NONE);
 	if (IR_REG_SPILLED(op1_reg)) {
@@ -3469,8 +4218,7 @@ static void ir_emit_if_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, u
 		op2_reg = IR_REG_NUM(op2_reg);
 		ir_emit_load(ctx, type, op2_reg, insn->op2);
 	}
-	|	ASM_REG_IMM_OP cmp, type, op2_reg, 0
-	ir_emit_jcc(ctx, b, def, insn, next_block, IR_NE, 1);
+	ir_emit_jz(ctx, b, next_block, IR_NE, type, op2_reg);
 }

 static void ir_emit_cond(ir_ctx *ctx, ir_ref def, ir_insn *insn)
@@ -3621,7 +4369,19 @@ static void ir_emit_sext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 			|	sxtw Rx(def_reg), Rw(op1_reg)
 		}
 	} else if (IR_IS_CONST_REF(insn->op1)) {
-		IR_ASSERT(0);
+		int64_t val;
+
+		if (ir_type_size[src_type] == 1) {
+			val = ctx->ir_base[insn->op1].val.i8;
+		} else if (ir_type_size[src_type] == 2) {
+			val = ctx->ir_base[insn->op1].val.i16;
+		} else if (ir_type_size[src_type] == 4) {
+			val = ctx->ir_base[insn->op1].val.i32;
+		} else {
+			IR_ASSERT(ir_type_size[src_type] == 8);
+			val = ctx->ir_base[insn->op1].val.i64;
+		}
+		ir_emit_load_imm_int(ctx, dst_type, def_reg, val);
 	} else {
 		ir_reg fp;
 		int32_t offset = ir_ref_spill_slot_offset(ctx, insn->op1, &fp);
@@ -3707,7 +4467,19 @@ static void ir_emit_zext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 			|	mov Rw(def_reg), Rw(op1_reg)
 		}
 	} else if (IR_IS_CONST_REF(insn->op1)) {
-		IR_ASSERT(0);
+		uint64_t val;
+
+		if (ir_type_size[src_type] == 1) {
+			val = ctx->ir_base[insn->op1].val.u8;
+		} else if (ir_type_size[src_type] == 2) {
+			val = ctx->ir_base[insn->op1].val.u16;
+		} else if (ir_type_size[src_type] == 4) {
+			val = ctx->ir_base[insn->op1].val.u32;
+		} else {
+			IR_ASSERT(ir_type_size[src_type] == 8);
+			val = ctx->ir_base[insn->op1].val.u64;
+		}
+		ir_emit_load_imm_int(ctx, dst_type, def_reg, val);
 	} else {
 		ir_reg fp;
 		int32_t offset = ir_ref_spill_slot_offset(ctx, insn->op1, &fp);
@@ -3781,8 +4553,9 @@ static void ir_emit_bitcast(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 	dasm_State **Dst = &data->dasm_state;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t size = ir_get_type_size(src_type);

-	IR_ASSERT(ir_type_size[dst_type] == ir_type_size[src_type]);
+	IR_ASSERT(ir_get_type_size(dst_type) == size);
 	IR_ASSERT(def_reg != IR_REG_NONE);
 	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
@@ -3800,7 +4573,8 @@ static void ir_emit_bitcast(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 		} else {
 			ir_emit_load(ctx, dst_type, def_reg, insn->op1);
 		}
-	} else if (IR_IS_TYPE_FP(src_type) && IR_IS_TYPE_FP(dst_type)) {
+	} else if ((IR_IS_TYPE_FP(src_type) || IR_IS_TYPE_VECTOR(src_type))
+			&& (IR_IS_TYPE_FP(dst_type) || IR_IS_TYPE_VECTOR(dst_type))) {
 		if (op1_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op1_reg)) {
 				op1_reg = IR_REG_NUM(op1_reg);
@@ -3812,75 +4586,86 @@ static void ir_emit_bitcast(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 		} else {
 			ir_emit_load(ctx, dst_type, def_reg, insn->op1);
 		}
-	} else if (IR_IS_TYPE_FP(src_type)) {
+	} else if (IR_IS_TYPE_FP(src_type) || IR_IS_TYPE_VECTOR(src_type)) {
 		IR_ASSERT(IR_IS_TYPE_INT(dst_type));
 		if (op1_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op1_reg)) {
 				op1_reg = IR_REG_NUM(op1_reg);
 				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
 			}
-			if (src_type == IR_DOUBLE) {
+			if (size == 8) {
 				|	fmov Rx(def_reg), Rd(op1_reg-IR_REG_FP_FIRST)
 			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
+				IR_ASSERT(size <= 4);
 				|	fmov Rw(def_reg), Rs(op1_reg-IR_REG_FP_FIRST)
 			}
 		} else if (IR_IS_CONST_REF(insn->op1)) {
-			IR_ASSERT(0); //???
+			if (src_type == IR_DOUBLE) {
+				ir_emit_load_imm_int(ctx, dst_type, def_reg, ctx->ir_base[insn->op1].val.i64);
+			} else {
+				IR_ASSERT(src_type == IR_FLOAT);
+				ir_emit_load_imm_int(ctx, dst_type, def_reg, ctx->ir_base[insn->op1].val.i32);
+			}
 		} else {
 			ir_reg fp;
 			int32_t offset = ir_ref_spill_slot_offset(ctx, insn->op1, &fp);

-			if (aarch64_may_encode_addr_offset(offset, ir_type_size[src_type])) {
-				if (src_type == IR_DOUBLE) {
+			if (aarch64_may_encode_addr_offset(offset, ir_get_type_size(src_type))) {
+				if (size == 8) {
 					|	ldr Rx(def_reg), [Rx(fp), #offset]
 				} else {
-					IR_ASSERT(src_type == IR_FLOAT);
+					IR_ASSERT(size <= 4);
 					|	ldr Rw(def_reg), [Rx(fp), #offset]
 				}
 			} else {
 				ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_INT_TMP, offset);
-				if (src_type == IR_DOUBLE) {
+				if (size == 8) {
 					|	ldr Rx(def_reg), [Rx(fp), Rx(IR_REG_INT_TMP)]
 				} else {
-					IR_ASSERT(src_type == IR_FLOAT);
+					IR_ASSERT(size <= 4);
 					|	ldr Rw(def_reg), [Rx(fp), Rx(IR_REG_INT_TMP)]
 				}
 			}
 		}
-	} else if (IR_IS_TYPE_FP(dst_type)) {
+	} else if (IR_IS_TYPE_FP(dst_type) || IR_IS_TYPE_VECTOR(dst_type)) {
 		IR_ASSERT(IR_IS_TYPE_INT(src_type));
 		if (op1_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op1_reg)) {
 				op1_reg = IR_REG_NUM(op1_reg);
 				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
 			}
-			if (dst_type == IR_DOUBLE) {
+			if (size == 8) {
 				|	fmov Rd(def_reg-IR_REG_FP_FIRST), Rx(op1_reg)
 			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
+				IR_ASSERT(size <= 4);
 				|	fmov Rs(def_reg-IR_REG_FP_FIRST), Rw(op1_reg)
 			}
 		} else if (IR_IS_CONST_REF(insn->op1)) {
-			IR_ASSERT(0); //???
+			ir_emit_load_imm_int(ctx, src_type, IR_REG_INT_TMP, ctx->ir_base[insn->op1].val.i64);
+			if (dst_type == IR_DOUBLE) {
+				|	fmov Rd(def_reg-IR_REG_FP_FIRST), Rx(IR_REG_INT_TMP)
+			} else {
+				IR_ASSERT(dst_type == IR_FLOAT);
+				|	fmov Rs(def_reg-IR_REG_FP_FIRST), Rw(IR_REG_INT_TMP)
+			}
 		} else {
 			ir_reg fp;
 			int32_t offset = ir_ref_spill_slot_offset(ctx, insn->op1, &fp);

-			if (aarch64_may_encode_addr_offset(offset, ir_type_size[src_type])) {
-				if (dst_type == IR_DOUBLE) {
-					|	ldr Rd(def_reg), [Rx(fp), #offset]
+			if (aarch64_may_encode_addr_offset(offset, ir_get_type_size(src_type))) {
+				if (size == 8) {
+					|	ldr Rd(def_reg-IR_REG_FP_FIRST), [Rx(fp), #offset]
 				} else {
-					IR_ASSERT(dst_type == IR_FLOAT);
-					|	ldr Rs(def_reg), [Rx(fp), #offset]
+					IR_ASSERT(size <= 4);
+					|	ldr Rs(def_reg-IR_REG_FP_FIRST), [Rx(fp), #offset]
 				}
 			 } else {
 				ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_INT_TMP, offset);
-				if (dst_type == IR_DOUBLE) {
-					|	ldr Rd(def_reg), [Rx(fp), Rx(IR_REG_INT_TMP)]
+				if (size == 8) {
+					|	ldr Rd(def_reg-IR_REG_FP_FIRST), [Rx(fp), Rx(IR_REG_INT_TMP)]
 				} else {
-					IR_ASSERT(dst_type == IR_FLOAT);
-					|	ldr Rs(def_reg), [Rx(fp), Rx(IR_REG_INT_TMP)]
+					IR_ASSERT(size <= 4);
+					|	ldr Rs(def_reg-IR_REG_FP_FIRST), [Rx(fp), Rx(IR_REG_INT_TMP)]
 				}
 			 }
 		}
@@ -5114,7 +5899,7 @@ static int32_t ir_call_used_stack(ir_ctx *ctx, ir_insn *insn, const ir_call_conv
 			}
 			int_param++;
 		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
+			IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
 			if (fp_param >= cc->fp_param_regs_count) {
 				used_stack += IR_MAX(sizeof(void*), ir_type_size[type]);
 			}
@@ -5127,7 +5912,7 @@ static int32_t ir_call_used_stack(ir_ctx *ctx, ir_insn *insn, const ir_call_conv
 	return used_stack + copy_stack;
 }

-static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_call_conv_dsc *cc, ir_reg tmp_reg)
+static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_call_conv_dsc *cc, ir_op op, ir_reg tmp_reg)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
@@ -5154,7 +5939,7 @@ static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const i
 		tmp_reg = IR_REG_IP0;
 	}

-	if (insn->op == IR_CALL && (ctx->flags & IR_PREALLOCATED_STACK)) {
+	if (op == IR_CALL && (ctx->flags & IR_PREALLOCATED_STACK)) {
 		// TODO: support for preallocated stack
 		used_stack = 0;
 	} else {
@@ -5166,7 +5951,7 @@ static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const i
 		} else {
 			ctx->call_stack_size += used_stack;
 			if (used_stack) {
-				if (insn->op == IR_TAILCALL && !(ctx->flags & IR_USE_FRAME_POINTER)) {
+				if (op == IR_TAILCALL && !(ctx->flags & IR_USE_FRAME_POINTER)) {
 					ctx->flags |= IR_USE_FRAME_POINTER;
 					|	stp x29, x30, [sp, # (-(ctx->stack_frame_size+16))]!
 					|	mov x29, sp
@@ -5255,7 +6040,7 @@ static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const i
 				continue;
 			}
 		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
+			IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
 			if (fp_param < cc->fp_param_regs_count) {
 				dst_reg = cc->fp_param_regs[fp_param];
 			} else {
@@ -5342,7 +6127,7 @@ static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const i
 				}
 				int_param++;
 			} else {
-				IR_ASSERT(IR_IS_TYPE_FP(type));
+				IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
 				if (fp_param < cc->fp_param_regs_count) {
 					dst_reg = cc->fp_param_regs[fp_param];
 				} else {
@@ -5379,7 +6164,7 @@ static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const i
 					if (src_reg == IR_REG_NONE) {
 						IR_ASSERT(tmp_fp_reg != IR_REG_NONE);
 						ir_emit_load(ctx, type, tmp_fp_reg, arg);
-						ir_emit_store_mem_fp(ctx, IR_DOUBLE, IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset), tmp_fp_reg);
+						ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset), tmp_fp_reg);
 					} else if (IR_REG_SPILLED(src_reg)) {
 						src_reg = IR_REG_NUM(src_reg);
 						ir_emit_load(ctx, type, src_reg, arg);
@@ -5438,7 +6223,7 @@ static void ir_emit_call_ex(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_cal
 				ir_emit_store(ctx, insn->type, def, cc->int_ret_reg);
 			}
 		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+			IR_ASSERT(IR_IS_TYPE_FP(insn->type) || IR_IS_TYPE_VECTOR(insn->type));
 			def_reg = IR_REG_NUM(ctx->regs[def][0]);
 			if (def_reg != IR_REG_NONE) {
 				if (def_reg != cc->fp_ret_reg) {
@@ -5458,7 +6243,7 @@ static void ir_emit_call(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	const ir_proto_t *proto = ir_call_proto(ctx, insn);
 	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
-	int32_t used_stack = ir_emit_arguments(ctx, def, insn, cc, ctx->regs[def][1]);
+	int32_t used_stack = ir_emit_arguments(ctx, def, insn, cc, IR_CALL, ctx->regs[def][1]);
 	ir_emit_call_ex(ctx, def, insn, cc, used_stack);
 }

@@ -5468,7 +6253,7 @@ static void ir_emit_tailcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 	dasm_State **Dst = &data->dasm_state;
 	const ir_proto_t *proto = ir_call_proto(ctx, insn);
 	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
-	int32_t used_stack = ir_emit_arguments(ctx, def, insn, cc, ctx->regs[def][1]);
+	int32_t used_stack = ir_emit_arguments(ctx, def, insn, cc, IR_TAILCALL, ctx->regs[def][1]);

 	if (used_stack != 0) {
 		ir_emit_call_ex(ctx, def, insn, cc, used_stack);
@@ -5498,6 +6283,27 @@ static void ir_emit_tailcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 		}
 	}

+	if (IR_IS_CONST_REF(insn->op2)
+	 && (ctx->flags2 & IR_RECURSIVE_TAILCALL)
+	 && ctx->ir_base[insn->op2].op == IR_FUNC
+	 && ctx->func_name == ctx->ir_base[insn->op2].val.name) {
+		if (ctx->flags2 & IR_HAS_ALLOCA) {
+			IR_ASSERT(ctx->flags & IR_USE_FRAME_POINTER);
+			if (!ctx->call_stack_size) {
+				|	mov sp, x29
+			} else if (aarch64_may_encode_imm12(ctx->call_stack_size)) {
+				|	sub sp, x29, #(ctx->call_stack_size)
+			} else {
+				ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_INT_TMP, ctx->call_stack_size);
+				|	sub sp, x29, Rx(IR_REG_INT_TMP)
+			}
+		}
+
+		|	b =>0
+
+		return;
+	}
+
 	ir_emit_epilogue(ctx);

 	if (IR_IS_CONST_REF(insn->op2)) {
@@ -5830,115 +6636,2132 @@ static void ir_emit_guard_overflow(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 	}
 }

-static void ir_emit_tls(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_tls_base_addr(ir_ctx *ctx, ir_reg reg, int mod)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
 	uint32_t code;
-	ir_reg reg = IR_REG_NUM(ctx->regs[def][0]);
-
-	if (ctx->use_lists[def].count == 1) {
-		/* dead load */
-		return;
-	}

 ||#ifdef __APPLE__
 ||	code = 0xd53bd060 | reg; // TODO: hard-coded: mrs reg, tpidrro_el0
 |	.long code
 |	and Rx(reg), Rx(reg), #0xfffffffffffffff8
-|//???	MEM_ACCESS_64_WITH_UOFFSET_64 ldr, Rx(reg), Rx(reg), #insn->op2, TMP1
-|//???	MEM_ACCESS_64_WITH_UOFFSET_64 ldr, Rx(reg), Rx(reg), #insn->op3, TMP1
+|//???	MEM_ACCESS_64_WITH_UOFFSET_64 ldr, Rx(reg), Rx(reg), #mod, TMP1
 ||#else
 ||	code = 0xd53bd040 | reg; // TODO: hard-coded: mrs reg, tpidr_el0
 |	.long code
-||# ifdef __FreeBSD__
-||	if (insn->op3 == IR_NULL) {
-|		ldr Rx(reg), [Rx(reg), #insn->op2]
-||	} else {
+||# ifndef __MUSL__
+||	if (mod >= 0) {
+||		IR_ASSERT(aarch64_may_encode_addr_offset(mod, sizeof(void*)));
 |		ldr Rx(reg), [Rx(reg), #0]
-|		ldr Rx(reg), [Rx(reg), #insn->op2]
-|		ldr Rx(reg), [Rx(reg), #insn->op3]
+|		ldr Rx(reg), [Rx(reg), #mod]
 ||	}
-||# elif defined(__MUSL__)
-||	if (insn->op3 == IR_NULL) {
-|		ldr Rx(reg), [Rx(reg), #insn->op2]
-||	} else {
+||# else
+||	if (mod >= 0) {
+||		IR_ASSERT(aarch64_may_encode_addr_offset(mod, sizeof(void*)));
 |		ldr Rx(reg), [Rx(reg), #-8]
-|		ldr Rx(reg), [Rx(reg), #insn->op2]
-|		ldr Rx(reg), [Rx(reg), #insn->op3]
+|		ldr Rx(reg), [Rx(reg), #mod]
 ||	}
-||# else
-||//???	IR_ASSERT(insn->op2 <= LDR_STR_PIMM64);
-|	ldr Rx(reg), [Rx(reg), #insn->op2]
 ||# endif
 ||#endif
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, IR_ADDR, def, reg);
-	}
 }

-static void ir_emit_exitcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_tls_addr(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
-	const ir_call_conv_dsc *cc = &ir_call_conv_default;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-
-	IR_ASSERT(def_reg != IR_REG_NONE);
-
-	|	stp d30, d31, [sp, #-16]!
-	|	stp d28, d29, [sp, #-16]!
-	|	stp d26, d27, [sp, #-16]!
-	|	stp d24, d25, [sp, #-16]!
-	|	stp d22, d23, [sp, #-16]!
-	|	stp d20, d21, [sp, #-16]!
-	|	stp d18, d19, [sp, #-16]!
-	|	stp d16, d17, [sp, #-16]!
-	|	stp d14, d15, [sp, #-16]!
-	|	stp d12, d13, [sp, #-16]!
-	|	stp d10, d11, [sp, #-16]!
-	|	stp d8, d9, [sp, #-16]!
-	|	stp d6, d7, [sp, #-16]!
-	|	stp d4, d5, [sp, #-16]!
-	|	stp d2, d3, [sp, #-16]!
-	|	stp d0, d1, [sp, #-16]!
-
-	|	str x30, [sp, #-16]!
-	|	stp x28, x29, [sp, #-16]!
-	|	stp x26, x27, [sp, #-16]!
-	|	stp x24, x25, [sp, #-16]!
-	|	stp x22, x23, [sp, #-16]!
-	|	stp x20, x21, [sp, #-16]!
-	|	stp x18, x19, [sp, #-16]!
-	|	stp x16, x17, [sp, #-16]!
-	|	stp x14, x15, [sp, #-16]!
-	|	stp x12, x13, [sp, #-16]!
-	|	stp x10, x11, [sp, #-16]!
-	|	stp x8, x9, [sp, #-16]!
-	|	stp x6, x7, [sp, #-16]!
-	|	stp x4, x5, [sp, #-16]!
-	|	stp x2, x3, [sp, #-16]!
-	|	stp x0, x1, [sp, #-16]!
+	ir_reg reg = IR_REG_NUM(ctx->regs[def][0]);

-	|	mov Rx(cc->int_param_regs[1]), sp
-	|	add Rx(cc->int_param_regs[0]), Rx(cc->int_param_regs[1]), #(32*8+32*8)
-	|	str Rx(cc->int_param_regs[0]), [sp, #(31*8)]
-	|	mov Rx(cc->int_param_regs[0]), Rx(IR_REG_INT_TMP)
+	IR_ASSERT(reg != IR_REG_NONE);

-	if (IR_IS_CONST_REF(insn->op2)) {
-		void *addr = ir_call_addr(ctx, insn, &ctx->ir_base[insn->op2]);
+	ir_emit_tls_base_addr(ctx, reg, insn->op2);

-		if (aarch64_may_use_b(ctx->code_buffer, addr)) {
-			|	bl &addr
+	if (insn->op3) {
+		if (aarch64_may_encode_imm12(insn->op3)) {
+			|	add Rx(reg), Rx(reg), #insn->op3
+		} else if (aarch64_may_encode_imm12(-insn->op3)) {
+			|	sub Rx(reg), Rx(reg), #(-insn->op3)
 		} else {
-			ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_INT_TMP, (intptr_t)addr);
-			|	blr Rx(IR_REG_INT_TMP)
+			ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_INT_TMP, insn->op3);
+			|	add Rx(reg), Rx(reg), Rx(IR_REG_INT_TMP)
 		}
-	} else {
-		IR_ASSERT(0);
 	}

-	|	add sp, sp, #(32*8+32*8)
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, IR_ADDR, def, reg);
+	}
+}
+
+static void ir_emit_tls_load(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg reg = IR_IS_TYPE_INT(insn->type) ? def_reg : ctx->regs[def][3];
+	ir_insn *addr_insn = &ctx->ir_base[insn->op2];
+	ir_mem mem;
+
+	if (ctx->use_lists[def].count == 1) {
+		/* dead load */
+		return;
+	}
+
+	IR_ASSERT(def_reg != IR_REG_NONE && reg != IR_REG_NONE);
+	IR_ASSERT(addr_insn->op == IR_TLS_ADDR);
+	ir_emit_tls_base_addr(ctx, reg, addr_insn->op2);
+	mem = IR_MEM_BO(reg, addr_insn->op3);
+	ir_emit_load_mem(ctx, insn->type, def_reg, mem);
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_tls_store(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+{
+	ir_type type = ctx->ir_base[insn->op3].type;
+	ir_reg op3_reg = ctx->regs[ref][3];
+	ir_insn *addr_insn = &ctx->ir_base[insn->op2];
+	ir_reg reg = ctx->regs[ref][0];
+	ir_mem mem;
+
+	IR_ASSERT(addr_insn->op == IR_TLS_ADDR);
+
+	IR_ASSERT(op3_reg != IR_REG_NONE && reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, insn->op3);
+	}
+
+	ir_emit_tls_base_addr(ctx, reg, addr_insn->op2);
+	mem = IR_MEM_BO(reg, addr_insn->op3);
+	ir_emit_store_mem(ctx, type, mem, op3_reg);
+}
+
+#if IR_SIMD
+static uint32_t neon_INS_int(uint32_t element_size, ir_reg dst_reg, ir_reg src_reg, uint32_t lane)
+{
+	IR_ASSERT(src_reg >= IR_REG_GP_FIRST && src_reg <= IR_REG_GP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	if (element_size == 8) {
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).d[lane], Rx(src_reg) */
+		return 0x4e001c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(((lane << 4) | 0x8) << 16);
+	} else if (element_size == 4) {
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).s[lane], Rw(src_reg) */
+		return 0x4e001c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(((lane << 3) | 0x4) << 16);
+	} else if (element_size == 2) {
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).h[lane], Rw(src_reg) */
+		return 0x4e001c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(((lane << 2) | 0x2) << 16);
+	} else {
+		IR_ASSERT(element_size == 1);
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).b[lane], Rw(src_reg) */
+		return 0x4e001c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(((lane << 1) | 0x1) << 16);
+	}
+}
+
+static uint32_t neon_INS_lane(uint32_t element_size, ir_reg dst_reg, ir_reg src_reg, uint32_t dst_lane, uint32_t src_lane)
+{
+	IR_ASSERT(src_reg >= IR_REG_FP_FIRST && src_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	if (element_size == 8) {
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).d[dst_lane], Rv(src_reg-IR_REG_FP_FIRST).d[src_lane] */
+		return 0x6e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((dst_lane << 4) | 0x8) << 16) | ((src_lane << 3) << 11);
+	} else if (element_size == 4) {
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).s[dst_lane], Rv(src_reg-IR_REG_FP_FIRST).s[src_lane] */
+		return 0x6e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((dst_lane << 3) | 0x4) << 16) | ((src_lane << 2) << 11);
+	} else if (element_size == 2) {
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).h[dst_lane], Rv(src_reg-IR_REG_FP_FIRST).h[src_lane] */
+		return 0x6e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((dst_lane << 2) | 0x2) << 16) | ((src_lane << 1) << 11);
+	} else {
+		IR_ASSERT(element_size == 1);
+		/* ins Rv(dst_reg-IR_REG_FP_FIRST).b[dst_lane], Rv(src_reg-IR_REG_FP_FIRST).b[src_lane] */
+		return 0x6e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((dst_lane << 1) | 0x1) << 16) | (src_lane << 11);
+	}
+}
+
+static uint32_t neon_DUP_int(uint32_t element_size, ir_reg dst_reg, ir_reg src_reg)
+{
+	IR_ASSERT(src_reg >= IR_REG_GP_FIRST && src_reg <= IR_REG_GP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	if (element_size == 8) {
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).2d, Rx(src_reg) */
+		return 0x0e000c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(1<<30) | (0x8 << 16);
+	} else if (element_size == 4) {
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).4s, Rw(src_reg) */
+		return 0x0e000c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(1<<30) | (0x4 << 16);
+	} else if (element_size == 2) {
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).8h, Rw(src_reg) */
+		return 0x0e000c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(1<<30) | (0x2 << 16);
+	} else {
+		IR_ASSERT(element_size == 1);
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).16b, Rw(src_reg) */
+		return 0x0e000c00 | (dst_reg-IR_REG_FP_FIRST) | (src_reg << 5) |
+			(1<<30) | (0x1 << 16);
+	}
+}
+
+static uint32_t neon_DUP_lane(uint32_t element_size, ir_reg dst_reg, ir_reg src_reg, uint32_t lane)
+{
+	IR_ASSERT(src_reg >= IR_REG_FP_FIRST && src_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	if (element_size == 8) {
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).2d, Rv(src_reg-IR_REG_FP_FIRST).d[lane] */
+		return 0x0e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (((lane << 4) | 0x8) << 16);
+	} else if (element_size == 4) {
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).4s, Rv(src_reg-IR_REG_FP_FIRST).s[lane] */
+		return 0x0e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (((lane << 3) | 0x4) << 16);
+	} else if (element_size == 2) {
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).8h, Rv(src_reg-IR_REG_FP_FIRST).h[lane] */
+		return 0x0e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (((lane << 2) | 0x2) << 16);
+	} else {
+		IR_ASSERT(element_size == 1);
+		/* dup Rv(dst_reg-IR_REG_FP_FIRST).8h, Rv(src_reg-IR_REG_FP_FIRST).h[lane] */
+		return 0x0e000400 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (((lane << 1) | 0x1) << 16);
+	}
+}
+
+static uint32_t neon_REV(uint32_t element_size, ir_reg dst_reg, ir_reg src_reg, uint32_t n)
+{
+	IR_ASSERT(src_reg >= IR_REG_FP_FIRST && src_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	if (n == 8) {
+		IR_ASSERT(element_size == 4);
+		/* rev64 Rv(dst_reg-IR_REG_FP_FIRST).4s, Rv(src_reg-IR_REG_FP_FIRST).4s */
+		return 0x0e200800 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (0x2 << 22);
+	} else if (n == 4) {
+		IR_ASSERT(element_size == 2);
+		/* rev32 Rv(dst_reg-IR_REG_FP_FIRST).8h, Rv(src_reg-IR_REG_FP_FIRST).8h */
+		return 0x2e200800 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (0x1 << 22);
+	} else {
+		IR_ASSERT(n == 2 && element_size == 1);
+		/* rev16 Rv(dst_reg-IR_REG_FP_FIRST).16b, Rv(src_reg-IR_REG_FP_FIRST).16b */
+		return 0x0e001800 | (dst_reg-IR_REG_FP_FIRST) | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (0x0 << 22);
+	}
+}
+
+static uint32_t neon_EXT(uint32_t element_size, ir_reg dst_reg, ir_reg src1_reg, ir_reg src2_reg, uint32_t first)
+{
+	IR_ASSERT(first < 16);
+	IR_ASSERT(src1_reg >= IR_REG_FP_FIRST && src2_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(src2_reg >= IR_REG_FP_FIRST && src2_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	/* ext Rv(dst_reg-IR_REG_FP_FIRST).16b, Rv(src1_reg-IR_REG_FP_FIRST).16b, Rv(src1_reg-IR_REG_FP_FIRST).16b, first */
+	return 0x2e000000 | (dst_reg-IR_REG_FP_FIRST) | ((src1_reg-IR_REG_FP_FIRST) << 5) |
+		((src2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (first << 11);
+}
+
+static uint32_t neon_TRN(uint32_t element_size, ir_reg dst_reg, ir_reg src1_reg, ir_reg src2_reg, uint32_t n)
+{
+	IR_ASSERT(src1_reg >= IR_REG_FP_FIRST && src1_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(src2_reg >= IR_REG_FP_FIRST && src2_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(n == 0 || n == 1);
+	/* trn? Rv(dst_reg-IR_REG_FP_FIRST).?, Rv(src1_reg-IR_REG_FP_FIRST).?, Rv(src2_reg-IR_REG_FP_FIRST).? */
+	return 0x0e002800 | (dst_reg-IR_REG_FP_FIRST) | ((src1_reg-IR_REG_FP_FIRST) << 5) |
+		((src2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (ir_ntz(element_size) << 22) | (n << 14);
+}
+
+static uint32_t neon_ZIP(uint32_t element_size, ir_reg dst_reg, ir_reg src1_reg, ir_reg src2_reg, uint32_t n)
+{
+	IR_ASSERT(src1_reg >= IR_REG_FP_FIRST && src1_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(src2_reg >= IR_REG_FP_FIRST && src2_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(n == 0 || n == 1);
+	/* trn? Rv(dst_reg-IR_REG_FP_FIRST).?, Rv(src1_reg-IR_REG_FP_FIRST).?, Rv(src2_reg-IR_REG_FP_FIRST).? */
+	return 0x0e003800 | (dst_reg-IR_REG_FP_FIRST) | ((src1_reg-IR_REG_FP_FIRST) << 5) |
+		((src2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (ir_ntz(element_size) << 22) | (n << 14);
+}
+
+static uint32_t neon_UZP(uint32_t element_size, ir_reg dst_reg, ir_reg src1_reg, ir_reg src2_reg, uint32_t n)
+{
+	IR_ASSERT(src1_reg >= IR_REG_FP_FIRST && src1_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(src2_reg >= IR_REG_FP_FIRST && src2_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_FP_FIRST && dst_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(n == 0 || n == 1);
+	/* trn? Rv(dst_reg-IR_REG_FP_FIRST).?, Rv(src1_reg-IR_REG_FP_FIRST).?, Rv(src2_reg-IR_REG_FP_FIRST).? */
+	return 0x0e001800 | (dst_reg-IR_REG_FP_FIRST) | ((src1_reg-IR_REG_FP_FIRST) << 5) |
+		((src2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (ir_ntz(element_size) << 22) | (n << 14);
+}
+
+static uint32_t ir_vector_extract_int(ir_type element_type, ir_reg dst_reg, ir_reg src_reg, uint32_t lane)
+{
+	IR_ASSERT(src_reg >= IR_REG_FP_FIRST && src_reg <= IR_REG_FP_LAST);
+	IR_ASSERT(dst_reg >= IR_REG_GP_FIRST && dst_reg <= IR_REG_GP_LAST);
+	if (element_type == IR_I8) {
+		/* smov Rw(dst_reg), Rv(src_reg-IR_REG_FP_FIRST).b[lane] */
+		return 0x0e002c00 | dst_reg | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((lane << 1) | 0x1) << 16);
+	} else if (element_type == IR_U8) {
+		/* umov Rw(dst_reg), Rv(src_reg-IR_REG_FP_FIRST).b[lane] */
+		return 0x0e003c00 | dst_reg | ((src_reg-IR_REG_FP_FIRST) << 5) |
+					(((lane << 1) | 0x1) << 16);
+	} else if (element_type == IR_I16) {
+		/* smov Rw(dst_reg), Rv(src_reg-IR_REG_FP_FIRST).h[lane] */
+		return 0x0e002c00 | dst_reg | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((lane << 2) | 0x2) << 16);
+	} else if (element_type == IR_U16) {
+		/* umov Rw(dst_reg), Rv(src_reg-IR_REG_FP_FIRST).h[lane] */
+		return 0x0e003c00 | dst_reg | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((lane << 2) | 0x2) << 16);
+	} else if (element_type == IR_I32 || element_type == IR_U32) {
+		/* umov Rw(dst_reg), Rv(src_reg-IR_REG_FP_FIRST).s[lane] */
+		return 0x0e003c00 | dst_reg | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(((lane << 3) | 0x4) << 16);
+	} else if (element_type == IR_I64 || element_type == IR_U64) {
+		/* umov Rx(dst_reg), Rv(src_reg-IR_REG_FP_FIRST).d[lane] */
+		return 0x0e003c00 | dst_reg | ((src_reg-IR_REG_FP_FIRST) << 5) |
+			(1<<30) | (((lane << 4) | 0x8) << 16);
+	} else {
+		IR_ASSERT(0);
+		return 0;
+	}
+
+}
+
+static void ir_emit_vector_extract(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[insn->op1].type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(insn->type == element_type ||
+		(IR_IS_TYPE_INT(insn->type) &&
+		 IR_IS_TYPE_INT(element_type) &&
+		 ir_type_size[insn->type] == ir_type_size[element_type]));
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+
+	if (IR_IS_CONST_REF(insn->op2)) {
+		uint32_t lane = ctx->ir_base[insn->op2].val.u32;
+		uint32_t code = 0;
+
+		IR_ASSERT(width <= 16 && lane < IR_VECTOR_LENGTH(type));
+		if (IR_IS_TYPE_INT(element_type)) {
+			code = ir_vector_extract_int(element_type, def_reg, op1_reg, lane);
+			|	.long code
+		} else {
+			if (lane != 0 || def_reg != op1_reg) {
+				code = neon_DUP_lane(ir_type_size[element_type], def_reg, op1_reg, lane);
+				|	.long code
+			}
+		}
+	} else {
+		ir_reg op2_reg = ctx->regs[def][2];
+
+		IR_ASSERT(op2_reg != IR_REG_NONE);
+
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, element_type, op2_reg, insn->op2);
+		}
+
+		IR_ASSERT(width <= 16);
+
+		/* extract through stack memory */
+		ir_mem mem = IR_MEM(IR_REG_STACK_POINTER, 0, IR_REG_NONE, 0);
+		ir_mem mem2 = IR_MEM(IR_REG_STACK_POINTER, 0, op2_reg, ir_ntz(ir_type_size[element_type]));
+
+		|	sub sp, sp, #width
+		ir_emit_store_mem_fp(ctx, type, mem, op1_reg);
+		ir_emit_load_mem(ctx, element_type, def_reg, mem2);
+		|	add sp, sp, #width
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_replace(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type) && type == ctx->ir_base[insn->op1].type);
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(element_type == ctx->ir_base[insn->op3].type ||
+		(IR_IS_TYPE_INT(element_type) &&
+		 IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type) &&
+		 ir_type_size[element_type] == ir_type_size[ctx->ir_base[insn->op3].type]));
+	IR_ASSERT(def_reg != IR_REG_NONE && op3_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, insn->op1);
+		}
+		if (op1_reg != def_reg) {
+			ir_emit_fp_mov(ctx, insn->type, def_reg, op1_reg);
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		ir_emit_load_imm_fp(ctx, insn->type, def_reg, insn->op1);
+	} else {
+		ir_mem mem = ir_ref_spill_slot(ctx, insn->op1);
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, mem);
+	}
+
+	if (IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, element_type, op3_reg, insn->op3);
+	}
+
+	if (IR_IS_CONST_REF(insn->op2)) {
+		uint32_t lane = ctx->ir_base[insn->op2].val.u32;
+		uint32_t code = 0;
+
+		IR_ASSERT(width <= 16 && lane < IR_VECTOR_LENGTH(type));
+		if (IR_IS_TYPE_INT(element_type)) {
+			code = neon_INS_int(ir_type_size[element_type], def_reg, op3_reg, lane);
+			|	.long code
+		} else {
+			if (lane != 0 || def_reg != op3_reg) {
+				code = neon_INS_lane(ir_type_size[element_type], def_reg, op3_reg, lane, 0);
+				|	.long code
+			}
+		}
+	} else {
+		ir_reg op2_reg = ctx->regs[def][2];
+
+		IR_ASSERT(op2_reg != IR_REG_NONE);
+
+		if (IR_REG_SPILLED(op2_reg)) {
+			op3_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, element_type, op2_reg, insn->op2);
+		}
+
+		IR_ASSERT(width <= 16);
+
+		/* modify through stack memory */
+		ir_mem mem = IR_MEM(IR_REG_STACK_POINTER, 0, IR_REG_NONE, 0);
+		ir_mem mem2 = IR_MEM(IR_REG_STACK_POINTER, 0, op2_reg, ir_ntz(ir_type_size[element_type]));
+
+		|	sub sp, sp, #width
+		ir_emit_store_mem_fp(ctx, insn->type, mem, op1_reg);
+		if (IR_IS_TYPE_INT(element_type)) {
+			ir_emit_store_mem_int(ctx, element_type, mem2, op3_reg);
+		} else {
+			ir_emit_store_mem_fp(ctx, element_type, mem2, op3_reg);
+		}
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, mem);
+		|	add sp, sp, #width
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_splat(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	uint32_t code = 0;
+
+	(void)width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(element_type == ctx->ir_base[insn->op1].type ||
+		(IR_IS_TYPE_INT(element_type) &&
+		 IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type) &&
+		 ir_type_size[element_type] == ir_type_size[ctx->ir_base[insn->op1].type]));
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+	IR_ASSERT(width <= 16);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, element_type, op1_reg, insn->op1);
+	}
+
+	if (IR_IS_TYPE_INT(element_type)) {
+		code = neon_DUP_int(ir_type_size[element_type], def_reg, op1_reg);
+	} else {
+		code = neon_DUP_lane(ir_type_size[element_type], def_reg, op1_reg, 0);
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_shuffle(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type mask_type = ctx->ir_base[insn->op3].type;
+	ir_type mask_element_type = IR_VECTOR_BASE_TYPE(mask_type);
+	uint32_t mask_element_size = ir_type_size[mask_element_type];
+	uint32_t mask_len = IR_VECTOR_LENGTH(mask_type);
+	ir_type element_type;
+	uint32_t width, element_size;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_type src1_type = ctx->ir_base[insn->op1].type;
+	ir_type src2_type = ctx->ir_base[insn->op2].type;
+	uint32_t code;
+	uint32_t mod, src1_len;
+
+	(void)width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type) && IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op3].type));
+
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	element_size = ir_type_size[element_type];
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(width <= 16);
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src1_type) && element_type == IR_VECTOR_BASE_TYPE(src1_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src2_type) && element_type == IR_VECTOR_BASE_TYPE(src2_type));
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+
+	src1_len = mod = IR_VECTOR_LENGTH(src1_type);
+	if (insn->op1 == insn->op2) {
+		op2_reg = op1_reg;
+	} else {
+		mod += IR_VECTOR_LENGTH(src2_type);
+		IR_ASSERT(op2_reg != IR_REG_NONE);
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
+		}
+	}
+
+	if (IR_IS_CONST_REF(insn->op3)) {
+		/* shuffle through sequence of moves */
+		int8_t *p = ir_long_const_ptr(ctx, insn->op3);
+		uint32_t s = mask_element_size;
+
+		if (rule != IR_SHUFFLE) {
+			uint32_t n = IR_VECTOR_LENGTH(insn->type);
+			uint32_t first = IR_SHUFFLE_MASK(0);
+
+			switch (rule) {
+				case IR_SHUFFLE_DUP:
+					if (first < n) {
+						code = neon_DUP_lane(element_size, def_reg, op1_reg, first);
+					} else {
+						code = neon_DUP_lane(element_size, def_reg, op2_reg, first % n);
+					}
+					break;
+				case IR_SHUFFLE_REV2:
+					if (first < n) {
+						code = neon_REV(element_size, def_reg, op1_reg, element_size * 2);
+					} else {
+						code = neon_REV(element_size, def_reg, op2_reg, element_size * 2);
+					}
+					break;
+				case IR_SHUFFLE_EXT:
+					if (first < n) {
+						if (first == 0) {
+							code = neon_MOV(def_reg, op1_reg);
+						} else {
+							code = neon_EXT(element_size, def_reg, op1_reg, op2_reg, first * element_size);
+						}
+					} else {
+						if (first == n) {
+							code = neon_MOV(def_reg, op2_reg);
+						} else {
+							code = neon_EXT(element_size, def_reg, op2_reg, op1_reg, (first % n) * element_size);
+						}
+					}
+					break;
+				case IR_SHUFFLE_TRN:
+					if (first < n) {
+						code = neon_TRN(element_size, def_reg, op1_reg, op2_reg, first != 0);
+					} else {
+						code = neon_TRN(element_size, def_reg, op2_reg, op1_reg, (first % n) != 0);
+					}
+					break;
+				case IR_SHUFFLE_ZIP:
+					if (first < n) {
+						code = neon_ZIP(element_size, def_reg, op1_reg, op2_reg, first != 0);
+					} else {
+						code = neon_ZIP(element_size, def_reg, op2_reg, op1_reg, (first % n) != 0);
+					}
+					break;
+				case IR_SHUFFLE_UZP:
+					if (first < n) {
+						code = neon_UZP(element_size, def_reg, op1_reg, op2_reg, first != 0);
+					} else {
+						code = neon_UZP(element_size, def_reg, op2_reg, op1_reg, (first % n) != 0);
+					}
+					break;
+				case IR_SHUFFLE_1EXT:
+					if (first < n) {
+						if (first == 0) {
+							code = neon_MOV(def_reg, op1_reg);
+						} else {
+							code = neon_EXT(element_size, def_reg, op1_reg, op1_reg, first * element_size);
+						}
+					} else {
+						if (first == n) {
+							code = neon_MOV(def_reg, op2_reg);
+						} else {
+							code = neon_EXT(element_size, def_reg, op2_reg, op2_reg, (first % n) * element_size);
+						}
+					}
+					break;
+				case IR_SHUFFLE_1TRN:
+					if (first < n) {
+						code = neon_TRN(element_size, def_reg, op1_reg, op1_reg, first != 0);
+					} else {
+						code = neon_TRN(element_size, def_reg, op2_reg, op2_reg, (first % n) != 0);
+					}
+					break;
+				case IR_SHUFFLE_1ZIP:
+					if (first < n) {
+						code = neon_ZIP(element_size, def_reg, op1_reg, op1_reg, first != 0);
+					} else {
+						code = neon_ZIP(element_size, def_reg, op2_reg, op2_reg, (first % n) != 0);
+					}
+					break;
+				case IR_SHUFFLE_1UZP:
+					if (first < n) {
+						code = neon_UZP(element_size, def_reg, op1_reg, op1_reg, first != 0);
+					} else {
+						code = neon_UZP(element_size, def_reg, op2_reg, op2_reg, (first % n) != 0);
+					}
+					break;
+				default:
+					IR_ASSERT(0);
+					return;
+			}
+			|	.long code
+		} else {
+			uint32_t i, j;
+
+			IR_ASSERT(op1_reg != def_reg);
+			for (i = 0; i < mask_len; i++) {
+				j = IR_SHUFFLE_MASK(i) % mod;
+				if (j < src1_len) {
+					code = neon_INS_lane(element_size, def_reg, op1_reg, i, j);
+				} else {
+					code = neon_INS_lane(element_size, def_reg, op2_reg, i, j - src1_len);
+				}
+				|	.long code
+			}
+		}
+	} else if (element_size == 1 && mask_element_size == 1 && insn->op1 == insn->op2) {
+		code = 0x4e000000 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			((op3_reg-IR_REG_FP_FIRST) << 16);
+		|	.long code
+	} else if (element_size == 1 && mask_element_size == 1 && op2_reg == op1_reg + 1 && IR_VECTOR_SIZE(src1_type) == 16) {
+		code = 0x4e001000 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			((op3_reg-IR_REG_FP_FIRST) << 16);
+		|	.long code
+	} else {
+		/* shuffle through stack memory */
+		uint32_t stack_space;
+		ir_mem src;
+		ir_reg tmp_reg = ctx->tmp_regs[def];
+		uint32_t i;
+
+		IR_ASSERT(op3_reg != IR_REG_NONE);
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, insn->op3);
+		}
+
+		stack_space = IR_VECTOR_SIZE(src1_type);
+		if (insn->op1 != insn->op2) {
+			stack_space += IR_VECTOR_SIZE(src2_type);
+		}
+
+		|	sub sp, sp, #stack_space
+
+		ir_emit_store_mem_fp(ctx, src1_type, IR_MEM(IR_REG_STACK_POINTER, 0, IR_REG_NONE, 0), op1_reg);
+		if (insn->op1 != insn->op2) {
+			uint32_t offset = IR_VECTOR_SIZE(src1_type);
+			ir_emit_store_mem_fp(ctx, src2_type, IR_MEM(IR_REG_STACK_POINTER, offset, IR_REG_NONE, 0), op2_reg);
+		}
+
+		if (element_type == IR_DOUBLE) {
+			element_type = IR_U64;
+		} else if (element_type == IR_FLOAT) {
+			element_type = IR_U32;
+		}
+		for (i = 0; i < mask_len; i++) {
+			code = ir_vector_extract_int(mask_element_type, tmp_reg, op3_reg, i);
+			|	.long code
+
+			IR_ASSERT(mod != 0 && ((mod - 1) & mod) == 0);
+			|	and Rx(tmp_reg), Rx(tmp_reg), #(mod-1)
+
+			src = IR_MEM(IR_REG_STACK_POINTER, 0, tmp_reg, ir_ntz(element_size));
+			ir_emit_load_mem_int(ctx, element_type, tmp_reg, src);
+			code = neon_INS_int(element_size, def_reg, tmp_reg, i);
+			|	.long code
+		}
+
+		|	add sp, sp, #stack_space
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_op(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t code = 0;
+
+	(void)width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+	IR_ASSERT(width <= 16);
+
+	IR_ASSERT(type == ctx->ir_base[insn->op1].type);
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+
+	switch (insn->op) {
+		default:
+		case IR_ABS:
+			IR_ASSERT(0 && "NIY unary op");
+			break;
+		case IR_NEG:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* neg Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b */
+					code = 0x2e20b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* neg Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).8h */
+					code = 0x2e20b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* neg Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e20b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* neg Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e20b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fneg Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2ea0f800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fneg Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2ea0f800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_NOT:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			/* not Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b */
+			code = 0x2e205800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) | (1<<30);
+			break;
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_binop(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	uint32_t code = 0;
+
+	(void)width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+		type = ctx->ir_base[op1].type;
+		IR_ASSERT(type == ctx->ir_base[op2].type);
+	} else if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+		IR_ASSERT(type == ctx->ir_base[op1].type);
+		IR_ASSERT(insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR);
+	} else {
+		IR_ASSERT(type == ctx->ir_base[op1].type && type == ctx->ir_base[op2].type);
+	}
+
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg == IR_REG_NONE && op1 == op2) {
+		op2_reg = op1_reg;
+	}
+	IR_ASSERT(op2_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, ctx->ir_base[op2].type, op2_reg, op2);
+	}
+
+	IR_ASSERT(width <= 16);
+
+	switch (insn->op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_ADD:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* add Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x0e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* add Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).86, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x0e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* add Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* add Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x0e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fadd Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x0e20d400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fadd Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0e20d400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_SUB:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* sub Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x2e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* sub Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x2e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* sub Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* sub Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e208400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fsub Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x0ea0d400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fsub Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0ea0d400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_MUL:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* mul Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x0e209c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* mul Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x0e209c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* mul Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0e209c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					ir_reg tmp2_reg;
+
+					IR_ASSERT(ctx->tmp_regs);
+					tmp2_reg = ctx->tmp_regs[def];
+					IR_ASSERT(tmp_reg != IR_REG_NONE && tmp2_reg != IR_REG_NONE);
+					/* mov Rx(tmp_reg), Rv(op1_reg).d[1] */
+					code = 0x0e003c00 | tmp_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (((1 << 4) | 0x8) << 16);
+					|	.long code
+					/* mov Rx(tmp2_reg), Rv(op2_reg).d[1] */
+					code = 0x0e003c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (((1 << 4) | 0x8) << 16);
+					|	.long code
+					|	mul Rx(tmp_reg), Rx(tmp_reg), Rx(tmp2_reg)
+					/* ins Rv(def_reg).d[1], Rx(tmp_reg) */
+					code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+						(((1 << 4) | 0x8) << 16);
+					|	.long code
+					/* mov Rx(tmp_reg), Rv(op1_reg).d[0] */
+					code = 0x0e003c00 | tmp_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (((0 << 4) | 0x8) << 16);
+					|	.long code
+					/* mov Rx(tmp2_reg), Rv(op2_reg).d[0] */
+					code = 0x0e003c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+						(1<<30) | (((0 << 4) | 0x8) << 16);
+					|	.long code
+					|	mul Rx(tmp_reg), Rx(tmp_reg), Rx(tmp2_reg)
+					/* ins Rv(def_reg).d[0], Rx(tmp_reg) */
+					code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+						(((0 << 4) | 0x8) << 16);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fmul Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e20dc00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fmul Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e20dc00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_DIV:
+		case IR_MOD:
+			if (IR_IS_TYPE_INT(element_type)) {
+				ir_reg tmp1_reg = (insn->op == IR_DIV) ? tmp_reg : IR_REG_INT_TMP;
+				ir_reg tmp2_reg;
+				uint32_t n = IR_VECTOR_LENGTH(type);
+
+				IR_ASSERT(ctx->tmp_regs);
+				tmp2_reg = ctx->tmp_regs[def];
+				IR_ASSERT(tmp_reg != IR_REG_NONE && tmp2_reg != IR_REG_NONE);
+				if (element_type == IR_I8) {
+					IR_ASSERT(n <= 16);
+					while (1) {
+						n--;
+						/* smov Rw(tmp1_reg), Rv(op1_reg).b[n] */
+						code = 0x0e002c00 | tmp1_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 1) | 0x1) << 16);
+						|	.long code
+						/* smov Rw(tmp2_reg), Rv(op2_reg).b[n] */
+						code = 0x0e002c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 1) | 0x1) << 16);
+						|	.long code
+						|	sdiv Rw(tmp_reg), Rw(tmp1_reg), Rw(tmp2_reg)
+						if (insn->op == IR_MOD) {
+							|	msub Rw(tmp_reg), Rw(tmp_reg), Rw(tmp2_reg), Rw(tmp1_reg)
+						}
+						/* ins Rv(def_reg).b[n], Rw(tmp_reg) */
+						code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+							(((n << 1) | 0x1) << 16);
+						if (n == 0) break;
+						|	.long code
+					}
+				} else if (element_type == IR_U8) {
+					IR_ASSERT(n <= 16);
+					while (1) {
+						n--;
+						/* mov Rw(tmp1_reg), Rv(op1_reg).b[n] */
+						code = 0x0e003c00 | tmp1_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 1) | 0x1) << 16);
+						|	.long code
+						/* mov Rw(tmp2_reg), Rv(op2_reg).b[n] */
+						code = 0x0e003c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 1) | 0x1) << 16);
+						|	.long code
+						|	udiv Rw(tmp_reg), Rw(tmp1_reg), Rw(tmp2_reg)
+						if (insn->op == IR_MOD) {
+							|	msub Rw(tmp_reg), Rw(tmp_reg), Rw(tmp2_reg), Rw(tmp1_reg)
+						}
+						/* ins Rv(def_reg).b[n], Rw(tmp_reg) */
+						code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+							(((n << 1) | 0x1) << 16);
+						if (n == 0) break;
+						|	.long code
+					}
+				} else if (element_type == IR_I16) {
+					IR_ASSERT(n <= 8);
+					while (1) {
+						n--;
+						/* smov Rw(tmp1_reg), Rv(op1_reg).h[n] */
+						code = 0x0e002c00 | tmp1_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 2) | 0x2) << 16);
+						|	.long code
+						/* smov Rw(tmp2_reg), Rv(op2_reg).h[n] */
+						code = 0x0e002c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 2) | 0x2) << 16);
+						|	.long code
+						|	sdiv Rw(tmp_reg), Rw(tmp1_reg), Rw(tmp2_reg)
+						if (insn->op == IR_MOD) {
+							|	msub Rw(tmp_reg), Rw(tmp_reg), Rw(tmp2_reg), Rw(tmp1_reg)
+						}
+						/* ins Rv(def_reg).h[n], Rw(tmp_reg) */
+						code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+							(((n << 2) | 0x2) << 16);
+						if (n == 0) break;
+						|	.long code
+					}
+				} else if (element_type == IR_U16) {
+					IR_ASSERT(n <= 8);
+					while (1) {
+						n--;
+						/* mov Rw(tmp1_reg), Rv(op1_reg).h[n] */
+						code = 0x0e003c00 | tmp1_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 2) | 0x2) << 16);
+						|	.long code
+						/* mov Rw(tmp2_reg), Rv(op2_reg).h[n] */
+						code = 0x0e003c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 2) | 0x2) << 16);
+						|	.long code
+						|	udiv Rw(tmp_reg), Rw(tmp1_reg), Rw(tmp2_reg)
+						if (insn->op == IR_MOD) {
+							|	msub Rw(tmp_reg), Rw(tmp_reg), Rw(tmp2_reg), Rw(tmp1_reg)
+						}
+						/* ins Rv(def_reg).h[n], Rw(tmp_reg) */
+						code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+							(((n << 2) | 0x2) << 16);
+						if (n == 0) break;
+						|	.long code
+					}
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					IR_ASSERT(n <= 4);
+					while (1) {
+						n--;
+						/* mov Rw(tmp1_reg), Rv(op1_reg).s[n] */
+						code = 0x0e003c00 | tmp1_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 3) | 0x4) << 16);
+						|	.long code
+						/* mov Rw(tmp2_reg), Rv(op2_reg).s[n] */
+						code = 0x0e003c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+							(0<<30) | (((n << 3) | 0x4) << 16);
+						|	.long code
+						if (IR_IS_TYPE_SIGNED(element_type)) {
+							|	sdiv Rw(tmp_reg), Rw(tmp1_reg), Rw(tmp2_reg)
+						} else {
+							|	udiv Rw(tmp_reg), Rw(tmp1_reg), Rw(tmp2_reg)
+						}
+						if (insn->op == IR_MOD) {
+							|	msub Rw(tmp_reg), Rw(tmp_reg), Rw(tmp2_reg), Rw(tmp1_reg)
+						}
+						/* ins Rv(def_reg).s[n], Rw(tmp_reg) */
+						code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+							(((n << 3) | 0x4) << 16);
+						if (n == 0) break;
+						|	.long code
+					}
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					IR_ASSERT(n <= 2);
+					while (1) {
+						n--;
+						/* mov Rx(tmp1_reg), Rv(op1_reg).d[n] */
+						code = 0x0e003c00 | tmp1_reg | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+							(1<<30) | (((n << 4) | 0x8) << 16);
+						|	.long code
+						/* mov Rx(tmp2_reg), Rv(op2_reg).d[n] */
+						code = 0x0e003c00 | tmp2_reg | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+							(1<<30) | (((n << 4) | 0x8) << 16);
+						|	.long code
+						if (IR_IS_TYPE_SIGNED(element_type)) {
+							|	sdiv Rx(tmp_reg), Rx(tmp1_reg), Rx(tmp2_reg)
+						} else {
+							|	udiv Rx(tmp_reg), Rx(tmp1_reg), Rx(tmp2_reg)
+						}
+						if (insn->op == IR_MOD) {
+							|	msub Rx(tmp_reg), Rx(tmp_reg), Rx(tmp2_reg), Rx(tmp1_reg)
+						}
+						/* ins Rv(def_reg).d[n], Rx(tmp_reg) */
+						code = 0x4e001c00 | (def_reg-IR_REG_FP_FIRST) | (tmp_reg << 5) |
+							(((n << 4) | 0x8) << 16);
+						if (n == 0) break;
+						|	.long code
+					}
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				IR_ASSERT(insn->op == IR_DIV);
+				if (element_type == IR_DOUBLE) {
+					/* fdiv Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e20fc00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fdiv Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e20fc00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_AND:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			/* and Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+			code = 0x0e201c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30);
+			break;
+		case IR_OR:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			/* orr Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+			code = 0x0ea01c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30);
+			break;
+		case IR_XOR:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			/* eor Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+			code = 0x2e201c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30);
+			break;
+		case IR_EQ:
+		case IR_NE:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* cmeq Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x2e208c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* cmeq Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).86, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x2e208c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* cmeq Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e208c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* cmeq Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e208c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fcmeq Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x0e20e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fcmeq Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0e20e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			if (insn->op == IR_NE) {
+				|	.long code
+				/* not Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(def_reg-IR_REG_FP_FIRST).16b */
+				code = 0x2e205800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) | (1<<30);
+			}
+			break;
+		case IR_LT:
+			SWAP_REGS(op1_reg, op2_reg);
+			IR_FALLTHROUGH;
+		case IR_GT:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* cmgt Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x0e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* cmgt Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).86, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x0e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* cmgt Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* cmgt Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x0e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fcmgt Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2ea0e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fcmgt Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x02ea0e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_LE:
+			SWAP_REGS(op1_reg, op2_reg);
+			IR_FALLTHROUGH;
+		case IR_GE:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* cmge Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x0e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* cmge Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).86, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x0e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* cmge Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x0e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* cmge Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x0e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fcmge Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e20e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fcmge Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e20e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_ULT:
+			SWAP_REGS(op1_reg, op2_reg);
+			IR_FALLTHROUGH;
+		case IR_UGT:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* cmhi Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x2e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* cmhi Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).86, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x2e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* cmhi Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* cmhi Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e203400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fcmgt Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2ea0e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fcmgt Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x02ea0e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_ULE:
+			SWAP_REGS(op1_reg, op2_reg);
+			IR_FALLTHROUGH;
+		case IR_UGE:
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (element_type == IR_I8 || element_type == IR_U8) {
+					/* cmhs Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+					code = 0x2e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				} else if (element_type == IR_I16 || element_type == IR_U16) {
+					/* cmhs Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).86, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+					code = 0x2e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					/* cmhs Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					/* cmhs Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e203c00 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				if (element_type == IR_DOUBLE) {
+					/* fcmge Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+					code = 0x2e20e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					/* fcmge Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+					code = 0x2e20e400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+						((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+				}
+			}
+			break;
+		case IR_SHL:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			if (element_type == IR_I8 || element_type == IR_U8) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					IR_ASSERT(tmp_reg >= IR_REG_FP_FIRST && tmp_reg <= IR_REG_FP_LAST);
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).16b, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x1 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+			} else if (element_type == IR_I16 || element_type == IR_U16) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).8h, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x2 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+			} else if (element_type == IR_I32 || element_type == IR_U32) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).4s, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x4 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).2d, Rx(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x8 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((op2_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+			} else {
+				IR_ASSERT(0);
+			}
+			break;
+		case IR_SHR:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			IR_ASSERT(tmp_reg >= IR_REG_FP_FIRST && tmp_reg <= IR_REG_FP_LAST);
+			if (element_type == IR_I8 || element_type == IR_U8) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).16b, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x1 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (0<<22);
+				|	.long code
+				/* ushl Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+				code = 0x2e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+			} else if (element_type == IR_I16 || element_type == IR_U16) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).8h, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x2 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (1<<22);
+				|	.long code
+				/* ushl Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+				code = 0x2e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+			} else if (element_type == IR_I32 || element_type == IR_U32) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).4s, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x4 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (2<<22);
+				|	.long code
+				/* ushl Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+				code = 0x2e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).2d, Rx(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x8 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (3<<22);
+				|	.long code
+				/* ushl Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+				code = 0x2e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+			} else {
+				IR_ASSERT(0);
+			}
+			break;
+		case IR_SAR:
+			IR_ASSERT(IR_IS_TYPE_INT(element_type));
+			IR_ASSERT(tmp_reg >= IR_REG_FP_FIRST && tmp_reg <= IR_REG_FP_LAST);
+			if (element_type == IR_I8 || element_type == IR_U8) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).16b, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x1 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (0<<22);
+				|	.long code
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).16b, Rv(op1_reg-IR_REG_FP_FIRST).16b, Rv(op2_reg-IR_REG_FP_FIRST).16b */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (0<<22);
+			} else if (element_type == IR_I16 || element_type == IR_U16) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).8h, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x2 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (1<<22);
+				|	.long code
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).8h, Rv(op1_reg-IR_REG_FP_FIRST).8h, Rv(op2_reg-IR_REG_FP_FIRST).8h */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (1<<22);
+			} else if (element_type == IR_I32 || element_type == IR_U32) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).4s, Rw(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x4 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (2<<22);
+				|	.long code
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s, Rv(op2_reg-IR_REG_FP_FIRST).4s */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (2<<22);
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+					/* dup Rv(def_reg-IR_REG_FP_FIRST).2d, Rx(op1_reg) */
+					code = 0x0e000c00 | (tmp_reg-IR_REG_FP_FIRST) | (op2_reg << 5) |
+						(1<<30) | (0x8 << 16);
+					|	.long code
+					op2_reg = tmp_reg;
+				}
+				/* neg Rv(tmp_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+				code = 0x2e20b800 | (tmp_reg-IR_REG_FP_FIRST) | ((op2_reg-IR_REG_FP_FIRST) << 5) |
+					(1<<30) | (3<<22);
+				|	.long code
+				/* sshl Rv(def_reg-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d, Rv(op2_reg-IR_REG_FP_FIRST).2d */
+				code = 0x0e204400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					((tmp_reg-IR_REG_FP_FIRST) << 16) | (1<<30) | (3<<22);
+			} else {
+				IR_ASSERT(0);
+			}
+			break;
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_ext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_type dst_element_type, src_element_type;
+	uint32_t src_width, dst_width, src_size, dst_size;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t code = 0;
+
+	(void)src_width;
+	(void)dst_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type) && IR_IS_TYPE_VECTOR(dst_type));
+	src_element_type = IR_VECTOR_BASE_TYPE(src_type);
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_element_type = IR_VECTOR_BASE_TYPE(dst_type);
+	dst_width = IR_VECTOR_SIZE(dst_type);
+
+	IR_ASSERT(dst_width <= 16);
+	IR_ASSERT(IR_IS_TYPE_INT(src_element_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_element_type));
+	src_size = ir_type_size[src_element_type];
+	dst_size = ir_type_size[dst_element_type];
+	IR_ASSERT(src_size < dst_size);
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	}
+	if (src_size == 1) {
+		if (insn->op == IR_ZEXT) {
+			/* uxtl Rv(def_ref-IR_REG_FP_FIRST).8h, Rv(op1_ref-IR_REG_FP_FIRST).8b */
+			code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (1<<19);
+			if (dst_size != 2) {
+				|	.long code
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(def_ref-IR_REG_FP_FIRST).4h */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				if (dst_size != 4) {
+					|	.long code
+					IR_ASSERT(dst_size == 8);
+					/* uxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_ref-IR_REG_FP_FIRST).2s */
+					code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+						(0<<30) | (4<<19);
+				}
+			}
+		} else {
+			/* sxtl Rv(def_ref-IR_REG_FP_FIRST).8h, Rv(op1_ref-IR_REG_FP_FIRST).8b */
+			code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (1<<19);
+			if (dst_size != 2) {
+				|	.long code
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(def_ref-IR_REG_FP_FIRST).4h */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				if (dst_size != 4) {
+					|	.long code
+					IR_ASSERT(dst_size == 8);
+					/* sxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_ref-IR_REG_FP_FIRST).2s */
+					code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+						(0<<30) | (4<<19);
+				}
+			}
+		}
+	} else if (src_size == 2) {
+		if (insn->op == IR_ZEXT) {
+			/* uxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_ref-IR_REG_FP_FIRST).4h */
+			code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (2<<19);
+			if (dst_size != 4) {
+				|	.long code
+				IR_ASSERT(dst_size == 8);
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_ref-IR_REG_FP_FIRST).2s */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (4<<19);
+			}
+		} else {
+			/* sxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_ref-IR_REG_FP_FIRST).4h */
+			code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (2<<19);
+			if (dst_size != 4) {
+				|	.long code
+				IR_ASSERT(dst_size == 8);
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_ref-IR_REG_FP_FIRST).2s */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (4<<19);
+			}
+		}
+	} else {
+		IR_ASSERT(src_size == 4);
+		IR_ASSERT(dst_size == 8);
+		if (insn->op == IR_ZEXT) {
+			/* uxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_ref-IR_REG_FP_FIRST).2s */
+			code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (4<<19);
+		} else {
+			/* sxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_ref-IR_REG_FP_FIRST).2s */
+			code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (4<<19);
+		}
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_trunc(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_type dst_element_type, src_element_type;
+	uint32_t src_width, src_size, dst_size;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t code = 0;
+
+	(void)src_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type) && IR_IS_TYPE_VECTOR(dst_type));
+	src_element_type = IR_VECTOR_BASE_TYPE(src_type);
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_element_type = IR_VECTOR_BASE_TYPE(dst_type);
+
+	IR_ASSERT(src_width <= 16);
+	IR_ASSERT(IR_IS_TYPE_INT(src_element_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_element_type));
+	src_size = ir_type_size[src_element_type];
+	dst_size = ir_type_size[dst_element_type];
+	IR_ASSERT(src_size > dst_size);
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	}
+	if (src_size == 8) {
+		/* xtn Rv(def_ref-IR_REG_FP_FIRST).2s, Rv(op1_ref-IR_REG_FP_FIRST).2d */
+		code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (2<<22);
+		if (dst_size != 4) {
+			|	.long code
+			/* xtn Rv(def_ref-IR_REG_FP_FIRST).4h, Rv(def_ref-IR_REG_FP_FIRST).4s */
+			code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (1<<22);
+			if (dst_size != 2) {
+				|	.long code
+				IR_ASSERT(dst_size == 1);
+				/* xtn Rv(def_ref-IR_REG_FP_FIRST).8b, Rv(def_ref-IR_REG_FP_FIRST).8h */
+				code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (0<<22);
+			}
+		}
+	} else if (src_size == 4) {
+		/* xtn Rv(def_ref-IR_REG_FP_FIRST).4h, Rv(op1_ref-IR_REG_FP_FIRST).4s */
+		code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (1<<22);
+		if (dst_size != 2) {
+			|	.long code
+			IR_ASSERT(dst_size == 1);
+			/* xtn Rv(def_ref-IR_REG_FP_FIRST).8b, Rv(def_ref-IR_REG_FP_FIRST).8h */
+			code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (0<<22);
+		}
+	} else if (src_size == 2) {
+		IR_ASSERT(dst_size == 1);
+		/* xtn Rv(def_ref-IR_REG_FP_FIRST).8b, Rv(op1_ref-IR_REG_FP_FIRST).8h */
+		code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (0<<22);
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_fp2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t src_width;
+	uint32_t code = 0;
+
+	(void)src_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+	src_width = IR_VECTOR_SIZE(src_type);
+	src_type = IR_VECTOR_BASE_TYPE(src_type);
+	dst_type = IR_VECTOR_BASE_TYPE(dst_type);
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	}
+	if (src_type == dst_type) {
+		if (op1_reg != def_reg) {
+			ir_emit_fp_mov(ctx, dst_type, def_reg, op1_reg);
+		}
+	} else if (src_type == IR_DOUBLE) {
+		IR_ASSERT(dst_type == IR_FLOAT);
+		IR_ASSERT(src_width <= 16);
+		/* fcvtn Rv(def_ref-IR_REG_FP_FIRST).2s, Rv(op1_ref-IR_REG_FP_FIRST).2d */
+		code = 0x0e216800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (1<<22);
+	} else {
+		IR_ASSERT(src_type == IR_FLOAT);
+		IR_ASSERT(dst_type == IR_DOUBLE);
+		IR_ASSERT(src_width <= 8);
+		/* fcvtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_ref-IR_REG_FP_FIRST).2s */
+		code = 0x0e217800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (1<<22);
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_fp2int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t src_width, dst_width, dst_size;
+	uint32_t code = 0;
+
+	(void)src_width;
+	(void)dst_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_width = IR_VECTOR_SIZE(dst_type);
+	src_type = IR_VECTOR_BASE_TYPE(src_type);
+	dst_type = IR_VECTOR_BASE_TYPE(dst_type);
+	dst_size = ir_type_size[dst_type];
+
+	IR_ASSERT(src_width <= 16 && dst_width <= 16);
+	IR_ASSERT(IR_IS_TYPE_FP(src_type) && IR_IS_TYPE_INT(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (src_type == IR_DOUBLE) {
+		if (IR_IS_TYPE_SIGNED(dst_type)) {
+			/* fcvtzs Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_ref-IR_REG_FP_FIRST).2d */
+			code = 0x0ea1b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		} else {
+			/* fcvtzu Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_ref-IR_REG_FP_FIRST).2d */
+			code = 0x2ea1b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		}
+		if (dst_size != 8) {
+			|	.long code
+			/* xtn Rv(def_ref-IR_REG_FP_FIRST).2s, Rv(def_ref-IR_REG_FP_FIRST).2d */
+			code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (2<<22);
+			if (dst_size != 4) {
+				|	.long code
+				/* xtn Rv(def_ref-IR_REG_FP_FIRST).4h, Rv(def_ref-IR_REG_FP_FIRST).4s */
+				code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (1<<22);
+				if (dst_size != 2) {
+					|	.long code
+					IR_ASSERT(dst_size == 1);
+					/* xtn Rv(def_ref-IR_REG_FP_FIRST).8b, Rv(def_ref-IR_REG_FP_FIRST).8h */
+					code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+						(0<<30) | (0<<22);
+				}
+			}
+		}
+	} else if (dst_size == 8) {
+		IR_ASSERT(src_type == IR_FLOAT);
+		/* fcvtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_ref-IR_REG_FP_FIRST).2s */
+		code = 0x0e217800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (1<<22);
+		|	.long code
+		if (IR_IS_TYPE_SIGNED(dst_type)) {
+			/* fcvtzs Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_ref-IR_REG_FP_FIRST).2d */
+			code = 0x0ea1b800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		} else {
+			/* fcvtzu Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_ref-IR_REG_FP_FIRST).2d */
+			code = 0x2ea1b800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		}
+	} else {
+		IR_ASSERT(src_type == IR_FLOAT);
+		if (IR_IS_TYPE_SIGNED(dst_type)) {
+			/* fcvtzs Rv(def_ref).4s, Rv(op1_ref-IR_REG_FP_FIRST).4s */
+			code = 0x0ea1b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (0<<22);
+		} else {
+			/* fcvtzu Rv(def_ref).4s, Rv(op1_ref-IR_REG_FP_FIRST).4s */
+			code = 0x2ea1b800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (0<<22);
+		}
+		if (dst_size != 4) {
+			|	.long code
+			/* xtn Rv(def_ref-IR_REG_FP_FIRST).4h, Rv(def_ref-IR_REG_FP_FIRST).4s */
+			code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+				(0<<30) | (1<<22);
+			if (dst_size != 2) {
+				|	.long code
+				IR_ASSERT(dst_size == 1);
+				/* xtn Rv(def_ref-IR_REG_FP_FIRST).8b, Rv(def_ref-IR_REG_FP_FIRST).8h */
+				code = 0x0e212800 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (0<<22);
+			}
+		}
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_int2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t src_width, dst_width, src_size;
+	uint32_t code = 0;
+
+	(void)src_width;
+	(void)dst_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_width = IR_VECTOR_SIZE(dst_type);
+	src_type = IR_VECTOR_BASE_TYPE(src_type);
+	dst_type = IR_VECTOR_BASE_TYPE(dst_type);
+	src_size = ir_type_size[src_type];
+
+	IR_ASSERT(src_width <= 16 && dst_width <= 16);
+	IR_ASSERT(IR_IS_TYPE_INT(src_type) && IR_IS_TYPE_FP(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (dst_type == IR_DOUBLE) {
+		if (IR_IS_TYPE_SIGNED(src_type)) {
+			if (src_size == 1) {
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).8h, Rv(op1_ref-IR_REG_FP_FIRST).8b */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (1<<19);
+				|	.long code
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(def_ref-IR_REG_FP_FIRST).4h */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_reg-IR_REG_FP_FIRST).2s */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else if (src_size == 2) {
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_ref-IR_REG_FP_FIRST).4h */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_reg-IR_REG_FP_FIRST).2s */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else if (src_size == 4) {
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2s */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else {
+				IR_ASSERT(src_size == 8);
+			}
+			/* scvtf Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d */
+			code = 0x0e21d800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		} else {
+			if (src_size == 1) {
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).8h, Rv(op1_ref-IR_REG_FP_FIRST).8b */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (1<<19);
+				|	.long code
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(def_ref-IR_REG_FP_FIRST).4h */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_reg-IR_REG_FP_FIRST).2s */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else if (src_size == 2) {
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_ref-IR_REG_FP_FIRST).4h */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(def_reg-IR_REG_FP_FIRST).2s */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else if (src_size == 4) {
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2s */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else {
+				IR_ASSERT(src_size == 8);
+			}
+			/* ucvtf Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d */
+			code = 0x2e21d800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		}
+	} else if (src_size == 8) {
+		IR_ASSERT(dst_type == IR_FLOAT);
+		if (IR_IS_TYPE_SIGNED(src_type)) {
+			/* scvtf Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d */
+			code = 0x0e21d800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		} else {
+			/* ucvtf Rv(def_ref-IR_REG_FP_FIRST).2d, Rv(op1_reg-IR_REG_FP_FIRST).2d */
+			code = 0x2e21d800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (1<<22);
+		}
+		|	.long code
+		/* fcvtn Rv(def_ref-IR_REG_FP_FIRST).2s, Rv(op1_ref-IR_REG_FP_FIRST).2d */
+		code = 0x0e216800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+			(0<<30) | (1<<22);
+	} else {
+		IR_ASSERT(dst_type == IR_FLOAT);
+		if (IR_IS_TYPE_SIGNED(src_type)) {
+			if (src_size == 1) {
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).8h, Rv(op1_ref-IR_REG_FP_FIRST).8b */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (1<<19);
+				|	.long code
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(def_ref-IR_REG_FP_FIRST).4h */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else if (src_size == 2) {
+				/* sxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_ref-IR_REG_FP_FIRST).4h */
+				code = 0x0f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else {
+				IR_ASSERT(src_size == 4);
+			}
+			/* scvtf Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s */
+			code = 0x0e21d800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (0<<22);
+		} else {
+			if (src_size == 1) {
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).8h, Rv(op1_ref-IR_REG_FP_FIRST).8b */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (1<<19);
+				|	.long code
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(def_ref-IR_REG_FP_FIRST).4h */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((def_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else if (src_size == 2) {
+				/* uxtl Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_ref-IR_REG_FP_FIRST).4h */
+				code = 0x2f00a400 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+					(0<<30) | (2<<19);
+				|	.long code
+				op1_reg = def_reg;
+			} else {
+				IR_ASSERT(src_size == 4);
+			}
+			/* ucvtf Rv(def_ref-IR_REG_FP_FIRST).4s, Rv(op1_reg-IR_REG_FP_FIRST).4s */
+			code = 0x2e21d800 | (def_reg-IR_REG_FP_FIRST) | ((op1_reg-IR_REG_FP_FIRST) << 5) |
+				(1<<30) | (0<<22);
+		}
+	}
+
+	|	.long code
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
+}
+#endif
+
+static void ir_emit_exitcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	const ir_call_conv_dsc *cc = &ir_call_conv_default;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	|	stp d30, d31, [sp, #-16]!
+	|	stp d28, d29, [sp, #-16]!
+	|	stp d26, d27, [sp, #-16]!
+	|	stp d24, d25, [sp, #-16]!
+	|	stp d22, d23, [sp, #-16]!
+	|	stp d20, d21, [sp, #-16]!
+	|	stp d18, d19, [sp, #-16]!
+	|	stp d16, d17, [sp, #-16]!
+	|	stp d14, d15, [sp, #-16]!
+	|	stp d12, d13, [sp, #-16]!
+	|	stp d10, d11, [sp, #-16]!
+	|	stp d8, d9, [sp, #-16]!
+	|	stp d6, d7, [sp, #-16]!
+	|	stp d4, d5, [sp, #-16]!
+	|	stp d2, d3, [sp, #-16]!
+	|	stp d0, d1, [sp, #-16]!
+
+	|	str x30, [sp, #-16]!
+	|	stp x28, x29, [sp, #-16]!
+	|	stp x26, x27, [sp, #-16]!
+	|	stp x24, x25, [sp, #-16]!
+	|	stp x22, x23, [sp, #-16]!
+	|	stp x20, x21, [sp, #-16]!
+	|	stp x18, x19, [sp, #-16]!
+	|	stp x16, x17, [sp, #-16]!
+	|	stp x14, x15, [sp, #-16]!
+	|	stp x12, x13, [sp, #-16]!
+	|	stp x10, x11, [sp, #-16]!
+	|	stp x8, x9, [sp, #-16]!
+	|	stp x6, x7, [sp, #-16]!
+	|	stp x4, x5, [sp, #-16]!
+	|	stp x2, x3, [sp, #-16]!
+	|	stp x0, x1, [sp, #-16]!
+
+	|	mov Rx(cc->int_param_regs[1]), sp
+	|	add Rx(cc->int_param_regs[0]), Rx(cc->int_param_regs[1]), #(32*8+32*8)
+	|	str Rx(cc->int_param_regs[0]), [sp, #(31*8)]
+	|	mov Rx(cc->int_param_regs[0]), Rx(IR_REG_INT_TMP)
+
+	if (IR_IS_CONST_REF(insn->op2)) {
+		void *addr = ir_call_addr(ctx, insn, &ctx->ir_base[insn->op2]);
+
+		if (aarch64_may_use_b(ctx->code_buffer, addr)) {
+			|	bl &addr
+		} else {
+			ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_INT_TMP, (intptr_t)addr);
+			|	blr Rx(IR_REG_INT_TMP)
+		}
+	} else {
+		IR_ASSERT(0);
+	}
+
+	|	add sp, sp, #(32*8+32*8)

 	if (def_reg != cc->int_ret_reg) {
 		ir_emit_mov(ctx, insn->type, def_reg, cc->int_ret_reg);
@@ -6034,18 +8857,6 @@ static void ir_emit_load_params(ir_ctx *ctx)
 	}
 }

-static ir_reg ir_get_free_reg(ir_type type, ir_regset available)
-{
-	if (IR_IS_TYPE_INT(type)) {
-		available = IR_REGSET_INTERSECTION(available, IR_REGSET_GP);
-	} else {
-		IR_ASSERT(IR_IS_TYPE_FP(type));
-		available = IR_REGSET_INTERSECTION(available, IR_REGSET_FP);
-	}
-	IR_ASSERT(!IR_REGSET_IS_EMPTY(available));
-	return IR_REGSET_FIRST(available);
-}
-
 static int ir_fix_dessa_tmps(ir_ctx *ctx, uint8_t type, ir_ref from, ir_ref to, void *dessa_from_block)
 {
 	ir_ref ref = ctx->cfg_blocks[(intptr_t)dessa_from_block].end;
@@ -6132,202 +8943,6 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 	ctx->param_stack_size = stack_offset;
 }

-static void ir_allocate_unique_spill_slots(ir_ctx *ctx)
-{
-	uint32_t b;
-	ir_block *bb;
-	ir_insn *insn;
-	ir_ref i, n, j, *p;
-	uint32_t *rule, insn_flags;
-	ir_regset available = 0;
-	ir_target_constraints constraints;
-	uint32_t def_flags;
-	ir_reg reg;
-	ir_backend_data *data = ctx->data;
-	const ir_call_conv_dsc *cc = data->ra_data.cc;
-	ir_regset scratch = ir_scratch_regset[cc->scratch_reg - IR_REG_NUM];
-
-	ctx->regs = ir_mem_malloc(sizeof(ir_regs) * ctx->insns_count);
-	memset(ctx->regs, IR_REG_NONE, sizeof(ir_regs) * ctx->insns_count);
-
-	/* vregs + tmp + fixed + SRATCH + ALL */
-	ctx->live_intervals = ir_mem_calloc(ctx->vregs_count + 1 + IR_REG_NUM + 2, sizeof(ir_live_interval*));
-
-    if (!ctx->arena) {
-		ctx->arena = ir_arena_create(16 * 1024);
-	}
-
-	for (b = 1, bb = ctx->cfg_blocks + b; b <= ctx->cfg_blocks_count; b++, bb++) {
-		IR_ASSERT(!(bb->flags & IR_BB_UNREACHABLE));
-		for (i = bb->start, insn = ctx->ir_base + i, rule = ctx->rules + i; i <= bb->end;) {
-			switch (ctx->rules ? *rule : insn->op) {
-				case IR_START:
-				case IR_BEGIN:
-				case IR_END:
-				case IR_IF_TRUE:
-				case IR_IF_FALSE:
-				case IR_CASE_VAL:
-				case IR_CASE_RANGE:
-				case IR_CASE_DEFAULT:
-				case IR_MERGE:
-				case IR_LOOP_BEGIN:
-				case IR_LOOP_END:
-				case IR_IGOTO_DUP:
-					break;
-				default:
-					def_flags = ir_get_target_constraints(ctx, i, &constraints);
-					if (ctx->rules
-					 && *rule != IR_CMP_AND_BRANCH_INT
-					 && *rule != IR_CMP_AND_BRANCH_FP
-					 && *rule != IR_GUARD_CMP_INT
-					 && *rule != IR_GUARD_CMP_FP) {
-						available = scratch;
-					}
-					if (ctx->vregs[i]) {
-						reg = constraints.def_reg;
-						if (reg != IR_REG_NONE && IR_REGSET_IN(available, reg)) {
-							IR_REGSET_EXCL(available, reg);
-							ctx->regs[i][0] = reg | IR_REG_SPILL_STORE;
-						} else if (def_flags & IR_USE_MUST_BE_IN_REG) {
-							if ((insn->op == IR_VLOAD || insn->op == IR_VLOAD_v)
-							 && ctx->live_intervals[ctx->vregs[i]]
-							 && ctx->live_intervals[ctx->vregs[i]]->stack_spill_pos != -1
-							 && ir_is_same_mem_var(ctx, i, ctx->ir_base[insn->op2].op3)) {
-								/* pass */
-							} else if (insn->op != IR_PARAM) {
-								reg = ir_get_free_reg(insn->type, available);
-								IR_REGSET_EXCL(available, reg);
-								ctx->regs[i][0] = reg | IR_REG_SPILL_STORE;
-							}
-						}
-						if (!ctx->live_intervals[ctx->vregs[i]]) {
-							ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
-							memset(ival, 0, sizeof(ir_live_interval));
-							ctx->live_intervals[ctx->vregs[i]] = ival;
-							ival->type = insn->type;
-							ival->reg = IR_REG_NONE;
-							ival->vreg = ctx->vregs[i];
-							ival->stack_spill_pos = -1;
-							if (insn->op == IR_PARAM && reg == IR_REG_NONE) {
-								ival->flags |= IR_LIVE_INTERVAL_MEM_PARAM;
-							} else {
-								ival->stack_spill_pos = ir_allocate_spill_slot(ctx, ival->type);
-							}
-						} else if (insn->op == IR_PARAM) {
-							IR_ASSERT(0 && "unexpected PARAM");
-							return;
-						}
-					} else if (insn->op == IR_VAR) {
-						ir_use_list *use_list = &ctx->use_lists[i];
-						ir_ref n = use_list->count;
-
-						if (n > 0) {
-							int32_t stack_spill_pos = insn->op3 = ir_allocate_spill_slot(ctx, insn->type);
-							ir_ref i, *p, use;
-							ir_insn *use_insn;
-
-							for (i = 0, p = &ctx->use_edges[use_list->refs]; i < n; i++, p++) {
-								use = *p;
-								use_insn = &ctx->ir_base[use];
-								if (use_insn->op == IR_VLOAD || use_insn->op == IR_VLOAD_v) {
-									if (ctx->vregs[use]
-									 && !ctx->live_intervals[ctx->vregs[use]]) {
-										ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
-										memset(ival, 0, sizeof(ir_live_interval));
-										ctx->live_intervals[ctx->vregs[use]] = ival;
-										ival->type = insn->type;
-										ival->reg = IR_REG_NONE;
-										ival->vreg = ctx->vregs[use];
-										ival->stack_spill_pos = stack_spill_pos;
-									}
-								} else if (use_insn->op == IR_VSTORE || use_insn->op == IR_STORE_v) {
-									if (!IR_IS_CONST_REF(use_insn->op3)
-									 && ctx->vregs[use_insn->op3]
-									 && !ctx->live_intervals[ctx->vregs[use_insn->op3]]) {
-										ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
-										memset(ival, 0, sizeof(ir_live_interval));
-										ctx->live_intervals[ctx->vregs[use_insn->op3]] = ival;
-										ival->type = insn->type;
-										ival->reg = IR_REG_NONE;
-										ival->vreg = ctx->vregs[use_insn->op3];
-										ival->stack_spill_pos = stack_spill_pos;
-									}
-								}
-							}
-						}
-					}
-
-					insn_flags = ir_op_flags[insn->op];
-					n = constraints.tmps_count;
-					if (n) {
-						do {
-							n--;
-							if (constraints.tmp_regs[n].type) {
-								ir_reg reg = ir_get_free_reg(constraints.tmp_regs[n].type, available);
-								ir_ref *ops = insn->ops;
-								IR_REGSET_EXCL(available, reg);
-								if (constraints.tmp_regs[n].num > 0) {
-									if (IR_IS_CONST_REF(ops[constraints.tmp_regs[n].num])) {
-										/* rematerialization */
-										reg |= IR_REG_SPILL_LOAD;
-									} else if (ctx->ir_base[ops[constraints.tmp_regs[n].num]].op == IR_ALLOCA ||
-											ctx->ir_base[ops[constraints.tmp_regs[n].num]].op == IR_VADDR) {
-										/* local address rematerialization */
-										reg |= IR_REG_SPILL_LOAD;
-									}
-								}
-								ctx->regs[i][constraints.tmp_regs[n].num] = reg;
-							} else {
-								reg = constraints.tmp_regs[n].reg;
-								if (reg >= IR_REG_NUM) {
-									available = IR_REGSET_DIFFERENCE(available, ir_scratch_regset[reg - IR_REG_NUM]);
-								} else {
-									IR_REGSET_EXCL(available, reg);
-								}
-							}
-						} while (n);
-					}
-					n = insn->inputs_count;
-					for (j = 1, p = insn->ops + 1; j <= n; j++, p++) {
-						ir_ref input = *p;
-						if (IR_OPND_KIND(insn_flags, j) == IR_OPND_DATA && input > 0 && ctx->vregs[input]) {
-							if ((def_flags & IR_DEF_REUSES_OP1_REG) && j == 1) {
-								ir_reg reg = IR_REG_NUM(ctx->regs[i][0]);
-								ctx->regs[i][1] = reg | IR_REG_SPILL_LOAD;
-							} else {
-								uint8_t use_flags = IR_USE_FLAGS(def_flags, j);
-								ir_reg reg = (j < constraints.hints_count) ? constraints.hints[j] : IR_REG_NONE;
-
-								if (reg != IR_REG_NONE && IR_REGSET_IN(available, reg)) {
-									IR_REGSET_EXCL(available, reg);
-									ctx->regs[i][j] = reg | IR_REG_SPILL_LOAD;
-								} else if (IR_IS_FOLDABLE_OP(insn->op) && j > 1 && input == insn->op1 && ctx->regs[i][1] != IR_REG_NONE) {
-									ctx->regs[i][j] = ctx->regs[i][1];
-								} else if (use_flags & IR_USE_MUST_BE_IN_REG) {
-									reg = ir_get_free_reg(ctx->ir_base[input].type, available);
-									IR_REGSET_EXCL(available, reg);
-									ctx->regs[i][j] = reg | IR_REG_SPILL_LOAD;
-								}
-							}
-						}
-					}
-					break;
-			}
-			n = ir_insn_len(insn);
-			i += n;
-			insn += n;
-			rule += n;
-		}
-		if (bb->flags & IR_BB_DESSA_MOVES) {
-			ir_gen_dessa_moves(ctx, b, ir_fix_dessa_tmps, (void*)(intptr_t)b);
-		}
-	}
-
-	ctx->used_preserved_regs = ctx->fixed_save_regset;
-	ctx->flags |= IR_NO_STACK_COMBINE;
-	ir_fix_stack_frame(ctx);
-}
-
 static void ir_preallocate_call_stack(ir_ctx *ctx)
 {
 	int call_stack_size, peak_call_stack_size = 0;
@@ -6335,7 +8950,7 @@ static void ir_preallocate_call_stack(ir_ctx *ctx)
 	ir_insn *insn;

 	for (i = 1, insn = ctx->ir_base + 1; i < ctx->insns_count;) {
-		if (insn->op == IR_CALL) {
+		if (insn->op == IR_CALL && (ctx->rules[i] & IR_RULE_MASK) == IR_CALL) {
 			const ir_proto_t *proto = ir_call_proto(ctx, insn);
 			const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
 			int32_t copy_stack;
@@ -6446,22 +9061,11 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 	ir_ref igoto_dup_ref = IR_UNUSED;
 	uint32_t igoto_dup_block = 0;

+	memset(&data, 0, sizeof(data));
 	data.ra_data.cc = ir_get_call_conv_dsc(ctx->flags);
-	data.ra_data.unused_slot_4 = 0;
-	data.ra_data.unused_slot_2 = 0;
-	data.ra_data.unused_slot_1 = 0;
-	data.ra_data.handled = NULL;
-	data.rodata_label = 0;
-	data.jmp_table_label = 0;
-	data.resolved_label_syms = 0;
 	ctx->data = &data;

-	if (!ctx->live_intervals) {
-		ctx->stack_frame_size = 0;
-		ctx->call_stack_size = 0;
-		ctx->used_preserved_regs = 0;
-		ir_allocate_unique_spill_slots(ctx);
-	}
+	IR_ASSERT(ctx->live_intervals != NULL);

 	if (ctx->fixed_stack_frame_size != -1) {
 		if (ctx->fixed_stack_red_zone) {
@@ -6500,6 +9104,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)

 	if (!(ctx->flags & IR_SKIP_PROLOGUE)) {
 		ir_emit_prologue(ctx);
+		if (ctx->flags2 & IR_RECURSIVE_TAILCALL) {
+			|=>0:
+		}
 	}
 	if (ctx->flags & IR_FUNCTION) {
 		ir_emit_load_params(ctx);
@@ -6786,12 +9393,63 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				case IR_GUARD_OVERFLOW:
 					ir_emit_guard_overflow(ctx, i, insn);
 					break;
-				case IR_TLS:
-					ir_emit_tls(ctx, i, insn);
+				case IR_TLS_ADDR:
+					ir_emit_tls_addr(ctx, i, insn);
+					break;
+				case IR_TLS_LOAD:
+					ir_emit_tls_load(ctx, i, insn);
+					break;
+				case IR_TLS_STORE:
+					ir_emit_tls_store(ctx, i, insn);
 					break;
 				case IR_TRAP:
 					|	brk
 					break;
+#if IR_SIMD
+				case IR_EXTRACT:
+					ir_emit_vector_extract(ctx, i, insn);
+					break;
+				case IR_REPLACE:
+					ir_emit_vector_replace(ctx, i, insn);
+					break;
+				case IR_SPLAT:
+					ir_emit_vector_splat(ctx, i, insn);
+					break;
+				case IR_SHUFFLE:
+				case IR_SHUFFLE_DUP:
+				case IR_SHUFFLE_REV2:
+				case IR_SHUFFLE_EXT:
+				case IR_SHUFFLE_TRN:
+				case IR_SHUFFLE_ZIP:
+				case IR_SHUFFLE_UZP:
+				case IR_SHUFFLE_1EXT:
+				case IR_SHUFFLE_1TRN:
+				case IR_SHUFFLE_1ZIP:
+				case IR_SHUFFLE_1UZP:
+					ir_emit_vector_shuffle(ctx, i, insn, (*rule) & IR_RULE_MASK);
+					break;
+				case IR_VECTOR_OP:
+					ir_emit_vector_op(ctx, i, insn);
+					break;
+				case IR_VECTOR_BINOP:
+					ir_emit_vector_binop(ctx, i, insn);
+					break;
+				case IR_VECTOR_EXT:
+					ir_emit_vector_ext(ctx, i, insn);
+					break;
+				case IR_VECTOR_TRUNC:
+					ir_emit_vector_trunc(ctx, i, insn);
+					break;
+				case IR_VECTOR_FP2FP:
+					ir_emit_vector_fp2fp(ctx, i, insn);
+					break;
+				case IR_VECTOR_FP2INT:
+					ir_emit_vector_fp2int(ctx, i, insn);
+					break;
+				case IR_VECTOR_INT2FP:
+					ir_emit_vector_int2fp(ctx, i, insn);
+					break;
+#endif
 				default:
 					IR_ASSERT(0 && "NIY rule/instruction");
 					ir_mem_free(data.emit_constants);
@@ -6878,6 +9536,70 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				}
 			}

+		} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+			int label = ctx->cfg_blocks_count + i;
+			uint32_t size = IR_VECTOR_SIZE(insn->type);
+			uint32_t n = IR_VECTOR_LENGTH(insn->type);
+			ir_type type = IR_VECTOR_BASE_TYPE(insn->type);
+			void *p = ir_long_const_ptr(ctx, -i);
+
+			if (!data.rodata_label) {
+				data.rodata_label = ctx->cfg_blocks_count + ctx->consts_count + 2;
+
+				|.rodata
+				|=>data.rodata_label:
+			}
+			if (size >= 16) {
+				|.align 16
+			} else {
+				|.align 8
+			}
+			|=>label:
+			if (ir_type_size[type] == 8) {
+				while (n--) {
+					ir_val val;
+					val.u64 = *(uint64_t*)p;
+					|.long val.u32, val.u32_hi
+					p = (char*)p + 8;
+				}
+			} else if (ir_type_size[type] == 4) {
+				while (n--) {
+					|.long *(uint32_t*)p
+					p = (char*)p + 4;
+				}
+			} else if (ir_type_size[type] == 2) {
+				while (n) {
+					uint16_t c;
+					uint32_t w = 0;
+					uint32_t j;
+
+					for (j = 0; j < 2; j++) {
+						c = *(uint16_t*)p;
+						w |= (uint32_t)c << (16U * j);
+						p = (char*)p + 2;
+						n--;
+						if (!n) break;
+					}
+					|	.long w
+				}
+			} else if (ir_type_size[type] == 1) {
+				while (n) {
+					uint8_t c;
+					uint32_t w = 0;
+					uint32_t j;
+
+					for (j = 0; j < 4; j++) {
+						c = *(uint8_t*)p;
+						w |= (uint32_t)c << (8U * j);
+						p = (char*)p + 1;
+						n--;
+						if (!n) break;
+					}
+					|	.long w
+				}
+			} else {
+				IR_ASSERT(0);
+			}
 		} else {
 			IR_ASSERT(0);
 		}
@@ -7226,6 +9948,7 @@ void *ir_emit_thunk(ir_code_buffer *code_buffer, void *addr, size_t *size_ptr)
 	entry = code_buffer->pos;
 	entry = (void*)IR_ALIGNED_SIZE(((size_t)(entry)), 4);
 	if (size > (size_t)((char*)code_buffer->end - (char*)entry)) {
+		*size_ptr = size;
 		dasm_free(&dasm_state);
 		return NULL;
 	}
diff --git a/ext/opcache/jit/ir/ir_aarch64.h b/ext/opcache/jit/ir/ir_aarch64.h
index e0817f9b330..24e80f38f5b 100644
--- a/ext/opcache/jit/ir/ir_aarch64.h
+++ b/ext/opcache/jit/ir/ir_aarch64.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Aarch64 CPU specific definitions)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -43,43 +43,43 @@
 	_(X31,  x31,  w31) \

 # define IR_FP_REGS(_) \
-	_(V0,   d0,   s0,   h0,   b0) \
-	_(V1,   d1,   s1,   h1,   b1) \
-	_(V2,   d2,   s2,   h2,   b2) \
-	_(V3,   d3,   s3,   h3,   b3) \
-	_(V4,   d4,   s4,   h4,   b4) \
-	_(V5,   d5,   s5,   h5,   b5) \
-	_(V6,   d6,   s6,   h6,   b6) \
-	_(V7,   d7,   s7,   h7,   b7) \
-	_(V8,   d8,   s8,   h8,   b8) \
-	_(V9,   d9,   s9,   h9,   b9) \
-	_(V10,  d10,  s10,  h10,  b10) \
-	_(V11,  d11,  s11,  h11,  b11) \
-	_(V12,  d12,  s12,  h12,  b12) \
-	_(V13,  d13,  s13,  h13,  b13) \
-	_(V14,  d14,  s14,  h14,  b14) \
-	_(V15,  d15,  s15,  h15,  b15) \
-	_(V16,  d16,  s16,  h16,  b16) \
-	_(V17,  d17,  s17,  h17,  b17) \
-	_(V18,  d18,  s18,  h18,  b18) \
-	_(V19,  d19,  s19,  h19,  b19) \
-	_(V20,  d20,  s20,  h20,  b20) \
-	_(V21,  d21,  s21,  h21,  b21) \
-	_(V22,  d22,  s22,  h22,  b22) \
-	_(V23,  d23,  s23,  h23,  b23) \
-	_(V24,  d24,  s24,  h24,  b24) \
-	_(V25,  d25,  s25,  h25,  b25) \
-	_(V26,  d26,  s26,  h26,  b26) \
-	_(V27,  d27,  s27,  h27,  b27) \
-	_(V28,  d28,  s28,  h28,  b28) \
-	_(V29,  d29,  s29,  h29,  b29) \
-	_(V30,  d30,  s30,  h30,  b30) \
-	_(V31,  d31,  s31,  h31,  b31) \
+	_(V0,   d0,   s0,   h0,   b0,   v0) \
+	_(V1,   d1,   s1,   h1,   b1,   v1) \
+	_(V2,   d2,   s2,   h2,   b2,   v2) \
+	_(V3,   d3,   s3,   h3,   b3,   v3) \
+	_(V4,   d4,   s4,   h4,   b4,   v4) \
+	_(V5,   d5,   s5,   h5,   b5,   v5) \
+	_(V6,   d6,   s6,   h6,   b6,   v6) \
+	_(V7,   d7,   s7,   h7,   b7,   v7) \
+	_(V8,   d8,   s8,   h8,   b8,   v8) \
+	_(V9,   d9,   s9,   h9,   b9,   v9) \
+	_(V10,  d10,  s10,  h10,  b10,  v10) \
+	_(V11,  d11,  s11,  h11,  b11,  v11) \
+	_(V12,  d12,  s12,  h12,  b12,  v12) \
+	_(V13,  d13,  s13,  h13,  b13,  v13) \
+	_(V14,  d14,  s14,  h14,  b14,  v14) \
+	_(V15,  d15,  s15,  h15,  b15,  v15) \
+	_(V16,  d16,  s16,  h16,  b16,  v16) \
+	_(V17,  d17,  s17,  h17,  b17,  v17) \
+	_(V18,  d18,  s18,  h18,  b18,  v18) \
+	_(V19,  d19,  s19,  h19,  b19,  v19) \
+	_(V20,  d20,  s20,  h20,  b20,  v20) \
+	_(V21,  d21,  s21,  h21,  b21,  v21) \
+	_(V22,  d22,  s22,  h22,  b22,  v22) \
+	_(V23,  d23,  s23,  h23,  b23,  v23) \
+	_(V24,  d24,  s24,  h24,  b24,  v24) \
+	_(V25,  d25,  s25,  h25,  b25,  v25) \
+	_(V26,  d26,  s26,  h26,  b26,  v26) \
+	_(V27,  d27,  s27,  h27,  b27,  v27) \
+	_(V28,  d28,  s28,  h28,  b28,  v28) \
+	_(V29,  d29,  s29,  h29,  b29,  v29) \
+	_(V30,  d30,  s30,  h30,  b30,  v30) \
+	_(V31,  d31,  s31,  h31,  b31,  v31) \

 #define IR_GP_REG_ENUM(code, name64, name32) \
 	IR_REG_ ## code,

-#define IR_FP_REG_ENUM(code, name64, name32, name16, name8) \
+#define IR_FP_REG_ENUM(code, name64, name32, name16, name8, name_vec) \
 	IR_REG_ ## code,

 enum _ir_reg {
diff --git a/ext/opcache/jit/ir/ir_builder.h b/ext/opcache/jit/ir/ir_builder.h
index 9492945b136..fe90b9549f0 100644
--- a/ext/opcache/jit/ir/ir_builder.h
+++ b/ext/opcache/jit/ir/ir_builder.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (IR Construction API)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -489,6 +489,11 @@ extern "C" {
 /* Helper to add address with a constant offset */
 #define ir_ADD_OFFSET(_addr, _offset)     _ir_ADD_OFFSET(_ir_CTX, (_addr), (_offset))

+#define ir_EXTRACT(_type, _op1, _op2)     ir_fold2(_ir_CTX, IR_OPT(IR_EXTRACT, (_type)), (_op1), (_op2))
+#define ir_REPLACE(_type, _op1, _op2, _v) ir_fold3(_ir_CTX, IR_OPT(IR_REPLACE, (_type)), (_op1), (_op2), (_v))
+#define ir_SPLAT(_type, _op1)             ir_UNARY_OP(IR_SPLAT, (_type), (_op1))
+#define ir_SHUFFLE(_type, _op1, _op2, _m) ir_fold3(_ir_CTX, IR_OPT(IR_SHUFFLE, (_type)), (_op1), (_op2), (_m))
+
 /* Unfoldable variant of COPY */
 #define ir_HARD_COPY(_type, _op1)         ir_emit2(_ir_CTX, IR_OPT(IR_COPY, (_type)), (_op1), IR_COPY_HARD)
 #define ir_HARD_COPY_B(_op1)              ir_HARD_COPY(IR_BOOL, _op1)
@@ -578,7 +583,7 @@ extern "C" {
 #define ir_STORE(_addr, _val)             _ir_STORE(_ir_CTX, (_addr), (_val))
 #define ir_LOAD_v(_type, _addr)           _ir_LOAD_v(_ir_CTX, (_type), (_addr))
 #define ir_STORE_v(_addr, _val)           _ir_STORE_v(_ir_CTX, (_addr), (_val))
-#define ir_TLS(_index, _offset)           _ir_TLS(_ir_CTX, (_index), (_offset))
+#define ir_TLS_ADDR(_index, _offset)      _ir_TLS_ADDR(_ir_CTX, (_index), (_offset))
 #define ir_TRAP()                         do {_ir_CTX->control = ir_emit1(_ir_CTX, IR_TRAP, _ir_CTX->control);} while (0)

 #define ir_FRAME_ADDR()                   ir_fold0(_ir_CTX, IR_OPT(IR_FRAME_ADDR, IR_ADDR))
@@ -633,6 +638,10 @@ extern "C" {
 #define ir_MERGE_WITH_EMPTY_TRUE(_if)     do {ir_ref end = ir_END(); ir_IF_TRUE(_if); ir_MERGE_2(end, ir_END());} while (0)
 #define ir_MERGE_WITH_EMPTY_FALSE(_if)    do {ir_ref end = ir_END(); ir_IF_FALSE(_if); ir_MERGE_2(end, ir_END());} while (0)

+/* for backward compatibility only */
+#define ir_TLS(_index, _offset)           ir_LOAD_A(ir_TLS_ADDR((_offset) == IR_NULL ? -1 : (_index), \
+                                              (_offset) == IR_NULL ? (_index) : (_offset)))
+
 ir_ref _ir_DIV(ir_ctx *ctx, ir_type type, ir_ref op1, ir_ref op2);
 ir_ref _ir_MOD(ir_ctx *ctx, ir_type type, ir_ref op1, ir_ref op2);
 ir_ref _ir_ADD_OFFSET(ir_ctx *ctx, ir_ref addr, uintptr_t offset);
@@ -692,7 +701,7 @@ void   _ir_MERGE_LIST(ir_ctx *ctx, ir_ref list);
 ir_ref _ir_PHI_LIST(ir_ctx *ctx, ir_ref list);
 ir_ref _ir_LOOP_BEGIN(ir_ctx *ctx, ir_ref src1);
 ir_ref _ir_LOOP_END(ir_ctx *ctx);
-ir_ref _ir_TLS(ir_ctx *ctx, ir_ref index, ir_ref offset);
+ir_ref _ir_TLS_ADDR(ir_ctx *ctx, ir_ref index, ir_ref offset);
 void   _ir_UNREACHABLE(ir_ctx *ctx);
 ir_ref _ir_SWITCH(ir_ctx *ctx, ir_ref val);
 void   _ir_CASE_VAL(ir_ctx *ctx, ir_ref switch_ref, ir_ref val);
diff --git a/ext/opcache/jit/ir/ir_cfg.c b/ext/opcache/jit/ir/ir_cfg.c
index 80258f7515c..4e4db166d37 100644
--- a/ext/opcache/jit/ir/ir_cfg.c
+++ b/ext/opcache/jit/ir/ir_cfg.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (CFG - Control Flow Graph)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -119,7 +119,7 @@ static void ir_remove_phis_inputs(ir_ctx *ctx, ir_use_list *use_list, int new_in
 		}

 		if (p != q) {
-			use_list->count -= (p - q);
+			use_list->count -= (ir_ref)(p - q);
 			do {
 				*q = IR_UNUSED; /* clenu-op the removed tail */
 				q++;
@@ -994,7 +994,19 @@ static bool ir_dominates(const ir_block *blocks, uint32_t b1, uint32_t b2)
 #define ENTRY_TIME(b) times[(b) * 2]
 #define EXIT_TIME(b)  times[(b) * 2 + 1]

-static IR_NEVER_INLINE void ir_collect_irreducible_loops(ir_ctx *ctx, uint32_t *times, ir_worklist *work, ir_list *list)
+#define IRR_FIRST_ENTRY(b) irreducible_loops[(b) * 2]
+#define IRR_NEXT_ENTRY(b)  irreducible_loops[(b) * 2 + 1]
+
+static IR_NEVER_INLINE void ir_push_irreducible_loop_entries(ir_worklist *work, uint32_t *irreducible_loops, uint32_t b)
+{
+	b = IRR_FIRST_ENTRY(b);
+	while (b) {
+		ir_worklist_push(work, b);
+		b = IRR_NEXT_ENTRY(b);
+	}
+}
+
+static IR_NEVER_INLINE uint32_t ir_collect_irreducible_loops(ir_ctx *ctx, uint32_t loops, uint32_t *times, ir_worklist *work, ir_list *list, uint32_t *irreducible_loops)
 {
 	ir_block *blocks = ctx->cfg_blocks;
 	uint32_t *edges = ctx->cfg_edges;
@@ -1021,12 +1033,15 @@ static IR_NEVER_INLINE void ir_collect_irreducible_loops(ir_ctx *ctx, uint32_t *
 		ir_block *bb = &blocks[hdr];

 		IR_ASSERT(bb->flags & IR_BB_IRREDUCIBLE_LOOP);
-		IR_ASSERT(!bb->loop_depth);
-		if (!bb->loop_depth) {
+		IR_ASSERT(!(bb->flags & IR_BB_LOOP_HEADER));
+		if (!(bb->flags & IR_BB_LOOP_HEADER)) {
 			/* process irreducible loop */

 			bb->flags |= IR_BB_LOOP_HEADER;
-			bb->loop_depth = 1;
+			bb->next_loop = loops;
+			loops = hdr;
+			IRR_FIRST_ENTRY(hdr) = 0;
+
 			if (ctx->ir_base[bb->start].op == IR_MERGE) {
 				ctx->ir_base[bb->start].op = IR_LOOP_BEGIN;
 			}
@@ -1062,18 +1077,22 @@ static IR_NEVER_INLINE void ir_collect_irreducible_loops(ir_ctx *ctx, uint32_t *
 				for (; n > 0; p++, n--) {
 					uint32_t pred = *p;
 					if (!ir_bitset_in(work->visited, pred)) {
-						if (blocks[pred].loop_header) {
-							if (blocks[pred].loop_header == b) continue;
-							do {
-								pred = blocks[pred].loop_header;
-							} while (blocks[pred].loop_header > 0);
+						if (blocks[pred].loop_header == b) continue;
+						while (1) {
+							if (UNEXPECTED(blocks[pred].flags & IR_BB_IRREDUCIBLE_LOOP)) {
+								ir_push_irreducible_loop_entries(work, irreducible_loops, pred);
+							}
+							if (!blocks[pred].loop_header) break;
+							pred = blocks[pred].loop_header;
 						}
 						if (ENTRY_TIME(pred) > ENTRY_TIME(hdr) && EXIT_TIME(pred) < EXIT_TIME(hdr)) {
 							/* "pred" is a descendant of "hdr" */
-								ir_worklist_push(work, pred);
-						} else if (bb->predecessors_count > 1) {
+							ir_worklist_push(work, pred);
+						} else if (bb->predecessors_count > 1 && !(bb->flags & IR_BB_IRREDUCIBLE_ENTRY)) {
 							/* another entry to the irreducible loop */
-							bb->flags |= IR_BB_IRREDUCIBLE_LOOP;
+							bb->flags |= IR_BB_IRREDUCIBLE_ENTRY;
+							IRR_NEXT_ENTRY(b) = IRR_FIRST_ENTRY(hdr);
+							IRR_FIRST_ENTRY(hdr) = b;
 							if (ctx->ir_base[bb->start].op == IR_MERGE) {
 								ctx->ir_base[bb->start].op = IR_LOOP_BEGIN;
 							}
@@ -1083,6 +1102,8 @@ static IR_NEVER_INLINE void ir_collect_irreducible_loops(ir_ctx *ctx, uint32_t *
 			}
 		}
 	}
+
+	return loops;
 }

 int ir_find_loops(ir_ctx *ctx)
@@ -1092,6 +1113,8 @@ int ir_find_loops(ir_ctx *ctx)
 	ir_block *blocks = ctx->cfg_blocks;
 	uint32_t *edges = ctx->cfg_edges;
 	ir_worklist work;
+	uint32_t loops = 0; /* linked list of identified loops ordered by dom_depth */
+	uint32_t *irreducible_loops = NULL;

 	if (ctx->flags2 & IR_NO_LOOPS) {
 		return 1;
@@ -1163,7 +1186,10 @@ int ir_find_loops(ir_ctx *ctx)
 		IR_ASSERT(bb->dom_depth <= prev_dom_depth);

 		if (UNEXPECTED(bb->dom_depth < irreducible_depth)) {
-			ir_collect_irreducible_loops(ctx, times, &work, &irreducible_list);
+			if (!irreducible_loops) {
+				irreducible_loops = ir_mem_malloc(sizeof(uint32_t) * 2 * (ctx->cfg_blocks_count + 1));
+			}
+			loops = ir_collect_irreducible_loops(ctx, loops, times, &work, &irreducible_list, irreducible_loops);
 			irreducible_depth = 0;
 		}

@@ -1210,8 +1236,9 @@ int ir_find_loops(ir_ctx *ctx)
 				uint32_t hdr = b;

 				bb->flags |= IR_BB_LOOP_HEADER;
+				bb->next_loop = loops;
+				loops = b;
 				ctx->flags2 |= IR_CFG_HAS_LOOPS;
-				bb->loop_depth = 1;
 				if (ctx->ir_base[bb->start].op == IR_MERGE) {
 					ctx->ir_base[bb->start].op = IR_LOOP_BEGIN;
 				}
@@ -1230,10 +1257,16 @@ int ir_find_loops(ir_ctx *ctx)
 						for (; n > 0; p++, n--) {
 							uint32_t pred = *p;
 							if (!ir_bitset_in(work.visited, pred)) {
+								if (UNEXPECTED(blocks[pred].flags & IR_BB_IRREDUCIBLE_LOOP)) {
+									ir_push_irreducible_loop_entries(&work, irreducible_loops, pred);
+								}
 								if (blocks[pred].loop_header) {
 									if (blocks[pred].loop_header == b) continue;
 									do {
 										pred = blocks[pred].loop_header;
+										if (UNEXPECTED(blocks[pred].flags & IR_BB_IRREDUCIBLE_LOOP)) {
+											ir_push_irreducible_loop_entries(&work, irreducible_loops, pred);
+										}
 									} while (blocks[pred].loop_header > 0);
 									ir_worklist_push(&work, pred);
 								} else {
@@ -1250,39 +1283,48 @@ int ir_find_loops(ir_ctx *ctx)

 	IR_ASSERT(!irreducible_depth);
 	if (ir_list_capasity(&irreducible_list)) {
+		IR_ASSERT(irreducible_loops);
+		ir_mem_free(irreducible_loops);
 		ir_list_free(&irreducible_list);
 	}

-	if (ctx->flags2 & IR_CFG_HAS_LOOPS) {
+	if (loops) {
+		ir_block *bb;
+
+		/* Set loop_depth for loop headers */
+		b = loops;
+		do {
+			bb = &blocks[b];
+			b = bb->next_loop;
+			IR_ASSERT(bb->flags & IR_BB_LOOP_HEADER);
+			bb->loop_depth = (bb->loop_header) ? blocks[bb->loop_header].loop_depth + 1 : 1;
+		} while (b);
+
+		/* Set loop_depth for loop members */
 		n = ctx->cfg_blocks_count + 1;
-		for (j = 1; j < n; j++) {
-			b = sorted_blocks[j];
-			ir_block *bb = &blocks[b];
-			if (bb->loop_header > 0) {
-				ir_block *loop = &blocks[bb->loop_header];
-				uint32_t loop_depth = loop->loop_depth;
-
-				if (bb->flags & IR_BB_LOOP_HEADER) {
-					loop_depth++;
+		for (j = 1, bb = blocks + 1; j < n; bb++, j++) {
+			if (bb->loop_header) {
+				if (!(bb->flags & IR_BB_LOOP_HEADER)) {
+					bb->loop_depth = blocks[bb->loop_header].loop_depth;
 				}
-				bb->loop_depth = loop_depth;
-				if (bb->flags & (IR_BB_ENTRY|IR_BB_LOOP_WITH_ENTRY)) {
-					loop->flags |= IR_BB_LOOP_WITH_ENTRY;
-					if (loop_depth > 1) {
-						/* Set IR_BB_LOOP_WITH_ENTRY flag for all the enclosing loops */
-						bb = &blocks[loop->loop_header];
-						while (1) {
-							if (bb->flags & IR_BB_LOOP_WITH_ENTRY) {
-								break;
-							}
-							bb->flags |= IR_BB_LOOP_WITH_ENTRY;
-							if (bb->loop_depth == 1) {
-								break;
-							}
-							bb = &blocks[loop->loop_header];
-						}
+				if (bb->flags & IR_BB_ENTRY) {
+					if (bb->flags & IR_BB_LOOP_HEADER) {
+						bb->flags |= IR_BB_LOOP_WITH_ENTRY;
 					}
+					/* Set IR_BB_LOOP_WITH_ENTRY flag for all the enclosing loops */
+					b = bb->loop_header;
+					do {
+						ir_block *loop = &blocks[b];
+						if (loop->flags & IR_BB_LOOP_WITH_ENTRY) {
+							break;
+						}
+						loop->flags |= IR_BB_LOOP_WITH_ENTRY;
+						b = loop->loop_header;
+					} while (b);
 				}
+			} else if ((bb->flags & (IR_BB_LOOP_HEADER|IR_BB_ENTRY|IR_BB_LOOP_WITH_ENTRY)) ==
+					(IR_BB_LOOP_HEADER|IR_BB_ENTRY)) {
+				bb->flags |= IR_BB_LOOP_WITH_ENTRY;
 			}
 		}
 	}
diff --git a/ext/opcache/jit/ir/ir_check.c b/ext/opcache/jit/ir/ir_check.c
index e1be7f6544d..fd7ba4478ab 100644
--- a/ext/opcache/jit/ir/ir_check.c
+++ b/ext/opcache/jit/ir/ir_check.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (IR verification)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -173,7 +173,7 @@ bool ir_check(const ir_ctx *ctx)
 					if (IR_OPND_KIND(flags, j) != IR_OPND_DATA) {
 						fprintf(stderr, "ir_base[%d].ops[%d] reference (%d) must not be constant\n", i, j, use);
 						ok = 0;
-					} else if (use >= ctx->consts_count) {
+					} else if (-use >= ctx->consts_count) {
 						fprintf(stderr, "ir_base[%d].ops[%d] constant reference (%d) is out of range\n", i, j, use);
 						ok = 0;
 					}
@@ -181,6 +181,7 @@ bool ir_check(const ir_ctx *ctx)
 					if (use >= ctx->insns_count) {
 						fprintf(stderr, "ir_base[%d].ops[%d] insn reference (%d) is out of range\n", i, j, use);
 						ok = 0;
+						continue;
 					}
 					use_insn = &ctx->ir_base[use];
 					switch (IR_OPND_KIND(flags, j)) {
@@ -237,7 +238,8 @@ bool ir_check(const ir_ctx *ctx)
 											  || insn->op == IR_SAR
 											  || insn->op == IR_ROL
 											  || insn->op == IR_ROR)
-											 && ir_type_size[use_insn->type] < ir_type_size[insn->type]) {
+											 && IR_IS_TYPE_INT(use_insn->type)
+											 && (IR_IS_TYPE_INT(insn->type) || IR_IS_TYPE_VECTOR(insn->type))) {
 												/* second argument of SHIFT may be incompatible with result */
 												break;
 											}
@@ -351,7 +353,7 @@ bool ir_check(const ir_ctx *ctx)
 				if (type != IR_ADDR
 				 && (!IR_IS_TYPE_INT(type) || ir_type_size[type] != ir_type_size[IR_ADDR])) {
 					fprintf(stderr, "ir_base[%d].op2 must have ADDR type (%s)\n",
-						i, ir_type_name[type]);
+						i, IR_IS_TYPE_VECTOR(type) ? "VECTOR" : ir_type_name[type]);
 					ok = 0;
 				}
 				break;
@@ -488,3 +490,42 @@ bool ir_check(const ir_ctx *ctx)

 	return ok;
 }
+
+bool ir_check_prototype(const ir_ctx *ctx, uint32_t flags, uint8_t ret_type, uint32_t params_count, uint8_t *param_types)
+{
+	bool ok = 1;
+	ir_ref ref = 2;
+	uint32_t n = 0;
+
+	while (ref < ctx->insns_count && ctx->ir_base[ref].op == IR_PARAM) {
+		if (n >= params_count) {
+			fprintf(stderr, "parameter count doesn't match function signature\n");
+			ok = 0;
+			break;
+		} else if (ctx->ir_base[ref].type != param_types[n]) {
+			fprintf(stderr, "parameter %d type doesn't match function signature\n", n);
+			ok = 0;
+		}
+		ref++;
+		n++;
+	}
+
+	if (n < params_count) {
+		fprintf(stderr, "parameter count doesn't match function signature\n");
+		ok = 0;
+	}
+	if ((flags & IR_VARARG_FUNC) != (ctx->flags & IR_VARARG_FUNC)) {
+		fprintf(stderr, "IR_VARARG_FUNC flag doesn't match function signature\n");
+		ok = 0;
+	}
+	if (ret_type != ctx->ret_type) {
+		fprintf(stderr, "return type doesn't match function signature\n");
+		ok = 0;
+	}
+	if ((flags & IR_CALL_CONV_MASK) != (ctx->flags & IR_CALL_CONV_MASK)) {
+		fprintf(stderr, "calling convention doesn't match function signature\n");
+		ok = 0;
+	}
+
+	return ok;
+}
diff --git a/ext/opcache/jit/ir/ir_disasm.c b/ext/opcache/jit/ir/ir_disasm.c
index 46deee32e17..0fe8aaea6e9 100644
--- a/ext/opcache/jit/ir/ir_disasm.c
+++ b/ext/opcache/jit/ir/ir_disasm.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Disassembler based on libcapstone)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

diff --git a/ext/opcache/jit/ir/ir_dump.c b/ext/opcache/jit/ir/ir_dump.c
index 3b34294d1c7..8d15e12f842 100644
--- a/ext/opcache/jit/ir/ir_dump.c
+++ b/ext/opcache/jit/ir/ir_dump.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (debug dumps)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -16,6 +16,16 @@
 # error "Unknown IR target"
 #endif

+static void ir_dump_type_name(ir_type type, FILE *f)
+{
+	if (IR_IS_TYPE_VECTOR(type)) {
+		fprintf(f, "<%s*%d>",
+			ir_type_name[IR_VECTOR_BASE_TYPE(type)], IR_VECTOR_LENGTH(type));
+	} else {
+		fprintf(f, "%s", ir_type_name[type]);
+	}
+}
+
 void ir_dump(const ir_ctx *ctx, FILE *f)
 {
 	ir_ref i, j, n, ref, *p;
@@ -23,16 +33,23 @@ void ir_dump(const ir_ctx *ctx, FILE *f)
 	uint32_t flags;

 	for (i = 1 - ctx->consts_count, insn = ctx->ir_base + i; i < IR_UNUSED; i++, insn++) {
-		fprintf(f, "%05d %s %s(", i, ir_op_name[insn->op], ir_type_name[insn->type]);
+		fprintf(f, "%05d %s ", i, ir_op_name[insn->op]);
+		ir_dump_type_name(insn->type, f);
+		fprintf(f, "(");
 		ir_print_const(ctx, insn, f, true);
 		fprintf(f, ")\n");
+		if (insn->op == IR_LONG_CONST) {
+			i += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+			insn += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+		}
 	}

 	for (i = IR_UNUSED + 1, insn = ctx->ir_base + i; i < ctx->insns_count; i++, insn++) {
 		flags = ir_op_flags[insn->op];
 		fprintf(f, "%05d %s", i, ir_op_name[insn->op]);
 		if ((flags & IR_OP_FLAG_DATA) || ((flags & IR_OP_FLAG_MEM) && insn->type != IR_VOID)) {
-			fprintf(f, " %s", ir_type_name[insn->type]);
+			fprintf(f, " ");
+			ir_dump_type_name(insn->type, f);
 		}
 		n = ir_operands_count(ctx, insn);
 		for (j = 1, p = insn->ops + 1; j <= 3; j++, p++) {
@@ -79,10 +96,16 @@ void ir_dump_dot(const ir_ctx *ctx, const char *name, const char *comments, FILE
 	fprintf(f, "\"\n");
 	fprintf(f, "\trankdir=TB;\n");
 	for (i = 1 - ctx->consts_count, insn = ctx->ir_base + i; i < IR_UNUSED; i++, insn++) {
-		fprintf(f, "\tc%d [label=\"C%d: CONST %s(", -i, -i, ir_type_name[insn->type]);
+		fprintf(f, "\tc%d [label=\"C%d: CONST ", -i, -i);
+		ir_dump_type_name(insn->type, f);
+		fprintf(f, "(");
 		/* FIXME(tony): We still cannot handle strings with escaped double quote inside */
 		ir_print_const(ctx, insn, f, false);
 		fprintf(f, ")\",style=filled,fillcolor=yellow];\n");
+		if (insn->op == IR_LONG_CONST) {
+			i += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+			insn += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+		}
 	}

 	for (i = IR_UNUSED + 1, insn = ctx->ir_base + i; i < ctx->insns_count;) {
@@ -105,13 +128,15 @@ void ir_dump_dot(const ir_ctx *ctx, const char *name, const char *comments, FILE
 				fprintf(f, "\tn%d [label=\"%d: %s\"", i, i, ir_op_name[insn->op]);
 				fprintf(f, ",shape=diamond,style=filled,fillcolor=deepskyblue];\n");
 			} else {
+				fprintf(f, "\tn%d [label=\"%d: %s ", i, i, ir_op_name[insn->op]);
+				ir_dump_type_name(insn->type, f);
 				if (insn->op == IR_PARAM) {
-					fprintf(f, "\tn%d [label=\"%d: %s %s \\\"%s\\\"\",style=filled,fillcolor=lightblue];\n",
-						i, i, ir_op_name[insn->op], ir_type_name[insn->type], ir_get_str(ctx, insn->op2));
+					fprintf(f, " \\\"%s\\\"\",style=filled,fillcolor=lightblue];\n",
+						ir_get_str(ctx, insn->op2));
 				} else if (insn->op == IR_VAR) {
-					fprintf(f, "\tn%d [label=\"%d: %s %s \\\"%s\\\"\"];\n", i, i, ir_op_name[insn->op], ir_type_name[insn->type], ir_get_str(ctx, insn->op2));
+					fprintf(f, " \\\"%s\\\"\"];\n", ir_get_str(ctx, insn->op2));
 				} else {
-					fprintf(f, "\tn%d [label=\"%d: %s %s\",style=filled,fillcolor=deepskyblue];\n", i, i, ir_op_name[insn->op], ir_type_name[insn->type]);
+					fprintf(f, "\",style=filled,fillcolor=deepskyblue];\n");
 				}
 			}
 		}
@@ -214,16 +239,14 @@ static void ir_dump_dessa_moves(const ir_ctx *ctx, int b, ir_block *bb, FILE *f)
 				int8_t *regs = ctx->regs[use_ref];
 				int8_t reg = regs[k];
 				if (reg != IR_REG_NONE) {
-					fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[input].type),
-						(reg & (IR_REG_SPILL_LOAD|IR_REG_SPILL_SPECIAL)) ? ":load" : "");
+					ir_dump_reg(ctx, reg, input, 0, f);
 				}
 			}
 			fprintf(f, " -> d_%d {R%d}", use_ref, ctx->vregs[use_ref]);
 			if (ctx->regs) {
 				int8_t reg = ctx->regs[use_ref][0];
 				if (reg != IR_REG_NONE) {
-					fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[use_ref].type),
-						(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? ":store" : "");
+					ir_dump_reg(ctx, reg, use_ref, 1, f);
 				}
 			}
 			fprintf(f, "\n");
@@ -284,6 +307,9 @@ static void ir_dump_cfg_block(ir_ctx *ctx, FILE *f, uint32_t b, ir_block *bb)
 	if (bb->flags & IR_BB_IRREDUCIBLE_LOOP) {
 		fprintf(stderr, "\tIRREDUCIBLE_LOOP\n");
 	}
+	if (bb->flags & IR_BB_IRREDUCIBLE_ENTRY) {
+		fprintf(stderr, "\tIRREDUCIBLE_ENTRY\n");
+	}
 	if (bb->loop_header > 0) {
 		fprintf(f, "\tloop_header=BB%d\n", bb->loop_header);
 	}
@@ -401,6 +427,11 @@ void ir_dump_live_ranges(const ir_ctx *ctx, FILE *f)
 			}
 			do {
 				if (ival->reg != IR_REG_NONE) {
+#if IR_X86_I64
+					if (ival->type == IR_I64 || ival->type == IR_U64) {
+						fprintf(f, "[%%%s,%%%s]", ir_reg_name(ival->reg, IR_U32), ir_reg_name(ival->reg_hi, IR_U32));
+					} else
+#endif
 					fprintf(f, "[%%%s]", ir_reg_name(ival->reg, ival->type));
 				}
 				p = &ival->range;
@@ -435,6 +466,17 @@ void ir_dump_live_ranges(const ir_ctx *ctx, FILE *f)
 							IR_LIVE_POS_TO_REF(use_pos->pos), IR_LIVE_POS_TO_SUB_REF(use_pos->pos),
 							-use_pos->hint_ref, use_pos->op_num);
 						if (use_pos->hint >= 0) {
+#if IR_X86_I64
+							if (ival->type == IR_I64 || ival->type == IR_U64) {
+								if (use_pos->flags & IR_HINT_TWO_REGS) {
+									fprintf(f, ", hint=%%%s,%%%s",
+										ir_reg_name(IR_REG_I64_LO(use_pos->hint), IR_U32),
+										ir_reg_name(IR_REG_I64_HI(use_pos->hint), IR_U32));
+								} else {
+									fprintf(f, ", hint=%%%s", ir_reg_name(use_pos->hint, IR_U32));
+								}
+							} else
+#endif
 							fprintf(f, ", hint=%%%s", ir_reg_name(use_pos->hint, ival->type));
 						}
 						fprintf(f, ")");
@@ -451,6 +493,17 @@ void ir_dump_live_ranges(const ir_ctx *ctx, FILE *f)
 								use_pos->op_num);
 						}
 						if (use_pos->hint >= 0) {
+#if IR_X86_I64
+							if (ival->type == IR_I64 || ival->type == IR_U64) {
+								if (use_pos->flags & IR_HINT_TWO_REGS) {
+									fprintf(f, ", hint=%%%s,%%%s",
+										ir_reg_name(IR_REG_I64_LO(use_pos->hint), IR_U32),
+										ir_reg_name(IR_REG_I64_HI(use_pos->hint), IR_U32));
+								} else {
+									fprintf(f, ", hint=%%%s", ir_reg_name(use_pos->hint, IR_U32));
+							    }
+							} else
+#endif
 							fprintf(f, ", hint=%%%s", ir_reg_name(use_pos->hint, ival->type));
 						}
 						if (use_pos->hint_ref) {
@@ -505,23 +558,53 @@ void ir_dump_codegen(const ir_ctx *ctx, FILE *f)
 	bool first;

 	fprintf(f, "{\n");
-	for (i = IR_UNUSED + 1, insn = ctx->ir_base - i; i < ctx->consts_count; i++, insn--) {
-		fprintf(f, "\t%s c_%d = ", ir_type_cname[insn->type], i);
-		if (insn->op == IR_FUNC) {
-			fprintf(f, "func %s", ir_get_str(ctx, insn->val.name));
-			ir_print_proto(ctx, insn->proto, f);
-		} else if (insn->op == IR_SYM) {
-			fprintf(f, "sym(%s)", ir_get_str(ctx, insn->val.name));
-		} else if (insn->op == IR_LABEL) {
-			fprintf(f, "label(%s)", ir_get_str(ctx, insn->val.name));
-		} else if (insn->op == IR_FUNC_ADDR) {
-			fprintf(f, "func *");
-			ir_print_const(ctx, insn, f, true);
-			ir_print_proto(ctx, insn->proto, f);
-		} else {
-			ir_print_const(ctx, insn, f, true);
+	/* Separate behavior to keep tests compatibility. TODO: remove the old behavior */
+	if (ctx->flags2 & IR_HAS_LONG_CONSTANTS) {
+		for (i = 1 - ctx->consts_count, insn = ctx->ir_base + i; i < IR_UNUSED; i++, insn++) {
+			fprintf(f, "\t");
+			ir_print_type_cname(insn->type, f);
+			fprintf(f, " c_%d = ", -i);
+			if (insn->op == IR_FUNC) {
+				fprintf(f, "func %s", ir_get_str(ctx, insn->val.name));
+				ir_print_proto(ctx, insn->proto, f);
+			} else if (insn->op == IR_SYM) {
+				fprintf(f, "sym(%s)", ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_LABEL) {
+				fprintf(f, "label(%s)", ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_FUNC_ADDR) {
+				fprintf(f, "func *");
+				ir_print_const(ctx, insn, f, true);
+				ir_print_proto(ctx, insn->proto, f);
+			} else {
+				ir_print_const(ctx, insn, f, true);
+			}
+			fprintf(f, ";\n");
+			if (insn->op == IR_LONG_CONST) {
+				i += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+				insn += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+			}
+		}
+	} else {
+		for (i = IR_UNUSED + 1, insn = ctx->ir_base - i; i < ctx->consts_count; i++, insn--) {
+			fprintf(f, "\t");
+			ir_print_type_cname(insn->type, f);
+			fprintf(f, " c_%d = ", i);
+			if (insn->op == IR_FUNC) {
+				fprintf(f, "func %s", ir_get_str(ctx, insn->val.name));
+				ir_print_proto(ctx, insn->proto, f);
+			} else if (insn->op == IR_SYM) {
+				fprintf(f, "sym(%s)", ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_LABEL) {
+				fprintf(f, "label(%s)", ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_FUNC_ADDR) {
+				fprintf(f, "func *");
+				ir_print_const(ctx, insn, f, true);
+				ir_print_proto(ctx, insn->proto, f);
+			} else {
+				ir_print_const(ctx, insn, f, true);
+			}
+			fprintf(f, ";\n");
 		}
-		fprintf(f, ";\n");
 	}

 	for (_b = 1; _b <= ctx->cfg_blocks_count; _b++) {
@@ -581,15 +664,16 @@ void ir_dump_codegen(const ir_ctx *ctx, FILE *f)
 				if (!(flags & IR_OP_FLAG_MEM) || insn->type == IR_VOID) {
 					fprintf(f, "\tl_%d = ", i);
 				} else {
-					fprintf(f, "\t%s d_%d", ir_type_cname[insn->type], i);
+					fprintf(f, "\t");
+					ir_print_type_cname(insn->type, f);
+					fprintf(f, " d_%d", i);
 					if (ctx->vregs && ctx->vregs[i]) {
 						fprintf(f, " {R%d}", ctx->vregs[i]);
 					}
 					if (ctx->regs) {
 						int8_t reg = ctx->regs[i][0];
 						if (reg != IR_REG_NONE) {
-							fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), insn->type),
-								(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? ":store" : "");
+							ir_dump_reg(ctx, reg, i, 1, f);
 						}
 					}
 					fprintf(f, ", l_%d = ", i);
@@ -597,15 +681,15 @@ void ir_dump_codegen(const ir_ctx *ctx, FILE *f)
 			} else {
 				fprintf(f, "\t");
 				if (flags & IR_OP_FLAG_DATA) {
-					fprintf(f, "%s d_%d", ir_type_cname[insn->type], i);
+					ir_print_type_cname(insn->type, f);
+					fprintf(f, " d_%d", i);
 					if (ctx->vregs && ctx->vregs[i]) {
 						fprintf(f, " {R%d}", ctx->vregs[i]);
 					}
 					if (ctx->regs) {
 						int8_t reg = ctx->regs[i][0];
 						if (reg != IR_REG_NONE) {
-							fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), insn->type),
-								(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? ":store" : "");
+							ir_dump_reg(ctx, reg, i, 1, f);
 						}
 					}
 					fprintf(f, " = ");
@@ -642,8 +726,7 @@ void ir_dump_codegen(const ir_ctx *ctx, FILE *f)
 								int8_t *regs = ctx->regs[i];
 								int8_t reg = regs[j];
 								if (reg != IR_REG_NONE) {
-									fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[ref].type),
-										(reg & (IR_REG_SPILL_LOAD|IR_REG_SPILL_SPECIAL)) ? ":load" : "");
+									ir_dump_reg(ctx, reg, ref, 0, f);
 								}
 							}
 							first = 0;
@@ -727,6 +810,11 @@ void ir_dump_codegen(const ir_ctx *ctx, FILE *f)
 				if (rule & IR_SIMPLE) {
 					fprintf(f, ":SIMPLE");
 				}
+#if IR_X86_I64
+				if (rule & IR_TWO_REGS) {
+					fprintf(f, ":TWO_REGS");
+				}
+#endif
 				fprintf(f, ")");
 			}
 			fprintf(f, "\n");
diff --git a/ext/opcache/jit/ir/ir_elf.h b/ext/opcache/jit/ir/ir_elf.h
index 961789a7b4a..bf5687a58cf 100644
--- a/ext/opcache/jit/ir/ir_elf.h
+++ b/ext/opcache/jit/ir/ir_elf.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (ELF header definitions)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

diff --git a/ext/opcache/jit/ir/ir_emit.c b/ext/opcache/jit/ir/ir_emit.c
index 92b66eb0358..f27a58d388b 100644
--- a/ext/opcache/jit/ir/ir_emit.c
+++ b/ext/opcache/jit/ir/ir_emit.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Native code generator based on DynAsm)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -53,6 +53,7 @@

 #ifdef IR_DEBUG
 # define DASM_CHECKS
+# define DASM_ABORT IR_ASSERT(0)
 #endif

 typedef struct _ir_copy {
@@ -61,10 +62,12 @@ typedef struct _ir_copy {
 	ir_reg  to;
 } ir_copy;

+#define IR_U32_HI IR_U64 /* type for dessa copy of the high 32-bit value of constant */
+
 typedef struct _ir_dessa_copy {
 	ir_type type;
-	int32_t from; /* negative - constant ref, [0..IR_REG_NUM) - CPU reg, [IR_REG_NUM...) - virtual reg */
-	int32_t to;   /* [0..IR_REG_NUM) - CPU reg, [IR_REG_NUM...) - virtual reg  */
+	int32_t from; /* negative - constant ref, [0..IR_REG_NUM) - CPU reg, [IR_REG_NUM...) - memory slot */
+	int32_t to;   /* [0..IR_REG_NUM) - CPU reg, [IR_REG_NUM...) - memory slot  */
 } ir_dessa_copy;

 const ir_proto_t *ir_call_proto(const ir_ctx *ctx, const ir_insn *insn)
@@ -103,6 +106,9 @@ static ir_reg ir_get_param_reg(const ir_ctx *ctx, ir_ref ref)
 	ir_insn *insn;
 	int int_param = 0;
 	int fp_param = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param = 0;
+#endif
 	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(ctx->flags);

 	for (i = use_list->count, p = &ctx->use_edges[use_list->refs]; i > 0; p++, i--) {
@@ -115,6 +121,11 @@ static ir_reg ir_get_param_reg(const ir_ctx *ctx, ir_ref ref)
 						/* struct passed by value on stack */
 						return IR_REG_NONE;
 					} else if (int_param < cc->int_param_regs_count) {
+#if IR_X86_I64
+						if (insn->type == IR_I64 || insn->type == IR_U64) {
+							return IR_REG_NONE;
+						}
+#endif
 						return cc->int_param_regs[int_param];
 					} else {
 						return IR_REG_NONE;
@@ -127,8 +138,27 @@ static ir_reg ir_get_param_reg(const ir_ctx *ctx, ir_ref ref)
 				if (cc->shadow_param_regs) {
 					fp_param++;
 				}
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					int_param++;
+					if (cc->shadow_param_regs) {
+						fp_param++;
+					}
+				}
+#endif
+#if IR_SIMD && defined(IR_TARGET_X86)
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (use == ref) {
+					if (vector_param < cc->vector_param_regs_count) {
+						return cc->vector_param_regs[vector_param];
+					} else {
+						return IR_REG_NONE;
+					}
+				}
+				vector_param++;
+#endif
 			} else {
-				IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+				IR_ASSERT(IR_IS_TYPE_FP(insn->type) || IR_IS_TYPE_VECTOR(insn->type));
 				if (use == ref) {
 					if (fp_param < cc->fp_param_regs_count) {
 						return cc->fp_param_regs[fp_param];
@@ -152,6 +182,9 @@ static int ir_get_args_regs(const ir_ctx *ctx, const ir_insn *insn, const ir_cal
 	ir_type type;
 	int int_param = 0;
 	int fp_param = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param = 0;
+#endif
 	int count = 0;

 	n = insn->inputs_count;
@@ -161,6 +194,17 @@ static int ir_get_args_regs(const ir_ctx *ctx, const ir_insn *insn, const ir_cal
 		type = arg->type;
 		if (IR_IS_TYPE_INT(type)) {
 			if (int_param < cc->int_param_regs_count && arg->op != IR_ARGVAL) {
+#if IR_X86_I64
+				if (type == IR_I64 || type == IR_U64) {
+					regs[j] = IR_REG_NONE;
+					count = j + 1;
+					int_param += 2;
+					if (cc->shadow_param_regs) {
+						fp_param += 2;
+					}
+					continue;
+				}
+#endif
 				regs[j] = cc->int_param_regs[int_param];
 				count = j + 1;
 				int_param++;
@@ -170,8 +214,18 @@ static int ir_get_args_regs(const ir_ctx *ctx, const ir_insn *insn, const ir_cal
 			} else {
 				regs[j] = IR_REG_NONE;
 			}
+#if IR_SIMD && defined(IR_TARGET_X86)
+		} else if (IR_IS_TYPE_VECTOR(type)) {
+			if (vector_param < cc->vector_param_regs_count) {
+				regs[j] = cc->vector_param_regs[vector_param];
+				count = j + 1;
+				vector_param++;
+			} else {
+				regs[j] = IR_REG_NONE;
+			}
+#endif
 		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
+			IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
 			if (fp_param < cc->fp_param_regs_count) {
 				regs[j] = cc->fp_param_regs[fp_param];
 				count = j + 1;
@@ -241,22 +295,28 @@ void *ir_resolve_sym_name(const char *name)
 #if defined(IR_TARGET_X86) || defined(IR_TARGET_X64)
 static void* ir_sym_addr(ir_ctx *ctx, const ir_insn *addr_insn)
 {
-	const char *name = ir_get_str(ctx, addr_insn->val.name);
-	void *addr = (ctx->loader && ctx->loader->resolve_sym_name) ?
-		ctx->loader->resolve_sym_name(ctx->loader, name, IR_RESOLVE_SYM_SILENT) :
-		ir_resolve_sym_name(name);
+	void *addr;

+	if (ctx->loader && ctx->loader->resolve_sym_name) {
+		addr = ctx->loader->resolve_sym_name(ctx->loader, ctx, addr_insn->val.name, IR_RESOLVE_SYM_SILENT);
+	} else {
+		const char *name = ir_get_str(ctx, addr_insn->val.name);
+		addr = ir_resolve_sym_name(name);
+	}
 	return addr;
 }
 #endif

 static void* ir_sym_val(ir_ctx *ctx, const ir_insn *addr_insn)
 {
-	const char *name = ir_get_str(ctx, addr_insn->val.name);
-	void *addr = (ctx->loader && ctx->loader->resolve_sym_name) ?
-		ctx->loader->resolve_sym_name(ctx->loader, name, addr_insn->op == IR_FUNC ? IR_RESOLVE_SYM_ADD_THUNK : 0) :
-		ir_resolve_sym_name(name);
+	void *addr;

+	if (ctx->loader && ctx->loader->resolve_sym_name) {
+		addr = ctx->loader->resolve_sym_name(ctx->loader, ctx, addr_insn->val.name, addr_insn->op == IR_FUNC ? IR_RESOLVE_SYM_ADD_THUNK : 0);
+	} else {
+		const char *name = ir_get_str(ctx, addr_insn->val.name);
+		addr = ir_resolve_sym_name(name);
+	}
 	IR_ASSERT(addr);
 	return addr;
 }
@@ -552,7 +612,9 @@ static int ir_parallel_copy(ir_ctx *ctx, ir_copy *copies, int count, ir_reg tmp_
 	return 1;
 }

-static void ir_emit_dessa_move(ir_ctx *ctx, ir_type type, ir_ref to, ir_ref from, ir_reg tmp_reg, ir_reg tmp_fp_reg)
+static void ir_emit_dessa_move(ir_ctx *ctx, ir_mem *mem_slots,
+                               ir_type type, ir_ref to, ir_ref from,
+                               ir_reg tmp_reg, ir_reg tmp_fp_reg)
 {
 	ir_mem mem_from, mem_to;

@@ -560,6 +622,11 @@ static void ir_emit_dessa_move(ir_ctx *ctx, ir_type type, ir_ref to, ir_ref from
 	if (to < IR_REG_NUM) {
 		if (IR_IS_CONST_REF(from)) {
 			if (-from < ctx->consts_count) {
+#if IR_X86_I64
+				if (type == IR_U32_HI) {
+					ir_emit_load_i64_hi(ctx, to, from);
+				} else
+#endif
 				/* constant reference */
 				ir_emit_load(ctx, type, to, from);
 			} else {
@@ -573,14 +640,26 @@ static void ir_emit_dessa_move(ir_ctx *ctx, ir_type type, ir_ref to, ir_ref from
 				ir_emit_fp_mov(ctx, type, to, from);
 			}
 		} else {
-			mem_from = ir_vreg_spill_slot(ctx, from - IR_REG_NUM);
+			mem_from = mem_slots[from - IR_REG_NUM];
 			ir_emit_load_mem(ctx, type, to, mem_from);
 		}
 	} else {
-		mem_to = ir_vreg_spill_slot(ctx, to - IR_REG_NUM);
+		mem_to = mem_slots[to - IR_REG_NUM];
 		if (IR_IS_CONST_REF(from)) {
 			if (-from < ctx->consts_count) {
 				/* constant reference */
+#if IR_X86_I64
+				if (type == IR_U32_HI) {
+#if defined(IR_TARGET_X86) || defined(IR_TARGET_X64)
+					ir_emit_store_mem_imm(ctx, IR_U32, mem_to, ctx->ir_base[from].val.u32_hi);
+#else
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					ir_emit_load_i64_hi(ctx, tmp_reg, from);
+					ir_emit_store_mem(ctx, IR_U32, mem_to, tmp_reg);
+#endif
+					return;
+				} else
+#endif
 #if defined(IR_TARGET_X86) || defined(IR_TARGET_X64)
 				if (IR_IS_TYPE_INT(type)
 				 && !IR_IS_SYM_CONST(ctx->ir_base[from].op)
@@ -589,7 +668,7 @@ static void ir_emit_dessa_move(ir_ctx *ctx, ir_type type, ir_ref to, ir_ref from
 					return;
 				}
 #endif
-				ir_reg tmp = IR_IS_TYPE_INT(type) ?  tmp_reg : tmp_fp_reg;
+				ir_reg tmp = IR_IS_TYPE_INT(type) ? tmp_reg : tmp_fp_reg;
 				IR_ASSERT(tmp != IR_REG_NONE);
 				ir_emit_load(ctx, type, tmp, from);
 				ir_emit_store_mem(ctx, type, mem_to, tmp);
@@ -603,7 +682,7 @@ static void ir_emit_dessa_move(ir_ctx *ctx, ir_type type, ir_ref to, ir_ref from
 		} else if (from < IR_REG_NUM) {
 			ir_emit_store_mem(ctx, type, mem_to, from);
 		} else {
-			mem_from = ir_vreg_spill_slot(ctx, from - IR_REG_NUM);
+			mem_from = mem_slots[from - IR_REG_NUM];
 			IR_ASSERT(IR_MEM_VAL(mem_to) != IR_MEM_VAL(mem_from));
 			ir_reg tmp = IR_IS_TYPE_INT(type) ?  tmp_reg : tmp_fp_reg;
 			IR_ASSERT(tmp != IR_REG_NONE);
@@ -613,11 +692,14 @@ static void ir_emit_dessa_move(ir_ctx *ctx, ir_type type, ir_ref to, ir_ref from
 	}
 }

-IR_ALWAYS_INLINE void ir_dessa_resolve_cycle(ir_ctx *ctx, int32_t *pred, int32_t *loc, int8_t *types, ir_bitset todo, int32_t to, ir_reg tmp_reg, ir_reg tmp_fp_reg)
+IR_ALWAYS_INLINE void ir_dessa_resolve_cycle(ir_ctx *ctx, ir_mem *mem_slots, int32_t *pred, int32_t *loc,
+                                             int8_t *types, ir_bitset todo, int32_t root,
+                                             ir_reg tmp_reg, ir_reg tmp_fp_reg)
 {
 	ir_ref from;
 	ir_mem tmp_spill_slot;
 	ir_type type;
+	int32_t to = root;

 	IR_MEM_VAL(tmp_spill_slot) = 0;
 	IR_ASSERT(!IR_IS_CONST_REF(to));
@@ -648,7 +730,7 @@ IR_ALWAYS_INLINE void ir_dessa_resolve_cycle(ir_ctx *ctx, int32_t *pred, int32_t
 		if (to < IR_REG_NUM) {
 			ir_emit_mov(ctx, type, tmp_reg, to);
 		} else {
-			ir_emit_load_mem_int(ctx, type, tmp_reg, ir_vreg_spill_slot(ctx, to - IR_REG_NUM));
+			ir_emit_load_mem_int(ctx, type, tmp_reg, mem_slots[to - IR_REG_NUM]);
 		}
 	} else {
 #ifdef IR_HAVE_SWAP_FP
@@ -668,7 +750,7 @@ IR_ALWAYS_INLINE void ir_dessa_resolve_cycle(ir_ctx *ctx, int32_t *pred, int32_t
 		if (to < IR_REG_NUM) {
 			ir_emit_fp_mov(ctx, type, tmp_fp_reg, to);
 		} else {
-			ir_emit_load_mem_fp(ctx, type, tmp_fp_reg, ir_vreg_spill_slot(ctx, to - IR_REG_NUM));
+			ir_emit_load_mem_fp(ctx, type, tmp_fp_reg, mem_slots[to - IR_REG_NUM]);
 		}
 	}

@@ -679,38 +761,38 @@ IR_ALWAYS_INLINE void ir_dessa_resolve_cycle(ir_ctx *ctx, int32_t *pred, int32_t
 		r = loc[from];
 		type = types[to];

-		if (from == r && ir_bitset_in(todo, from)) {
-			/* Memory to memory move inside an isolated or "blocked" cycle requres an additional temporary register */
-			if (to >= IR_REG_NUM && r >= IR_REG_NUM) {
-				ir_reg tmp = IR_IS_TYPE_INT(type) ?  tmp_reg : tmp_fp_reg;
+		if (from == root) break;

-				if (!IR_MEM_VAL(tmp_spill_slot)) {
-					/* Free a register, saving it in a temporary spill slot */
-					tmp_spill_slot = IR_MEM_BO(IR_REG_STACK_POINTER, -16);
-					ir_emit_store_mem(ctx, type, tmp_spill_slot, tmp);
-				}
-				ir_emit_dessa_move(ctx, type, to, r, tmp_reg, tmp_fp_reg);
-			} else {
-				ir_emit_dessa_move(ctx, type, to, r, IR_REG_NONE, IR_REG_NONE);
+		/* Memory to memory move inside an isolated or "blocked" cycle requres an additional temporary register */
+		if (to >= IR_REG_NUM && r >= IR_REG_NUM) {
+			ir_reg tmp = IR_IS_TYPE_INT(type) ?  tmp_reg : tmp_fp_reg;
+
+			if (!IR_MEM_VAL(tmp_spill_slot)) {
+				/* Free a register, saving it in a temporary spill slot */
+				tmp_spill_slot = IR_MEM_BO(IR_REG_STACK_POINTER, -16);
+				ir_emit_store_mem(ctx, type, tmp_spill_slot, tmp);
 			}
-			ir_bitset_excl(todo, to);
-			loc[from] = to;
-			to = from;
+			ir_emit_dessa_move(ctx, mem_slots, type, to, r, tmp_reg, tmp_fp_reg);
 		} else {
-			break;
+			ir_emit_dessa_move(ctx, mem_slots, type, to, r, IR_REG_NONE, IR_REG_NONE);
 		}
+		ir_bitset_excl(todo, to);
+		loc[from] = to;
+		to = from;
 	}

 	type = types[to];
 	if (IR_MEM_VAL(tmp_spill_slot)) {
 		ir_emit_load_mem(ctx, type, IR_IS_TYPE_INT(type) ? tmp_reg : tmp_fp_reg, tmp_spill_slot);
 	}
-	ir_emit_dessa_move(ctx, type, to, loc[from], IR_REG_NONE, IR_REG_NONE);
+	ir_emit_dessa_move(ctx, mem_slots, type, to, loc[from], IR_REG_NONE, IR_REG_NONE);
 	ir_bitset_excl(todo, to);
 	loc[from] = to;
 }

-static int ir_dessa_parallel_copy(ir_ctx *ctx, ir_dessa_copy *copies, int count, ir_reg tmp_reg, ir_reg tmp_fp_reg)
+static int ir_dessa_parallel_copy(ir_ctx *ctx, ir_dessa_copy *copies, int count,
+                                  ir_mem *mem_slots, int mem_slots_count,
+                                  ir_reg tmp_reg, ir_reg tmp_fp_reg)
 {
 	int i;
 	int32_t *pred, *loc, to, from;
@@ -724,17 +806,22 @@ static int ir_dessa_parallel_copy(ir_ctx *ctx, ir_dessa_copy *copies, int count,
 		from = copies[0].from;
 		IR_ASSERT(from != to);
 		type = copies[0].type;
-		ir_emit_dessa_move(ctx, type, to, from, tmp_reg, tmp_fp_reg);
+		ir_emit_dessa_move(ctx, mem_slots, type, to, from, tmp_reg, tmp_fp_reg);
 		return 1;
 	}

-	len = IR_REG_NUM + ctx->vregs_count + 1;
-	todo = ir_bitset_malloc(len);
-	srcs = ir_bitset_malloc(len);
+	len = IR_REG_NUM + mem_slots_count + 1;
 	loc = ir_mem_malloc(len * 2 * sizeof(int32_t) + len * sizeof(int8_t));
 	pred = loc + len;
 	types = (int8_t*)(pred + len);

+	len = ir_bitset_len(len);
+	todo = ir_mem_malloc(len * IR_BITSET_BITS / 8 * 4);
+	memset(todo, 0, len * IR_BITSET_BITS / 8 * 2);
+	srcs = todo + len;
+	ready = srcs + len;
+	visited = ready + len;
+
 	for (i = 0; i < count; i++) {
 		from = copies[i].from;
 		to = copies[i].to;
@@ -754,24 +841,23 @@ static int ir_dessa_parallel_copy(ir_ctx *ctx, ir_dessa_copy *copies, int count,
 	IR_ASSERT(tmp_fp_reg == IR_REG_NONE || !ir_bitset_in(srcs, tmp_fp_reg));

 	/* first we resolve all "windmill blades" - trees, that don't set temporary registers */
-	ready = ir_bitset_malloc(len);
-	ir_bitset_copy(ready, todo, ir_bitset_len(len));
-	ir_bitset_difference(ready, srcs, ir_bitset_len(len));
+	ir_bitset_copy(ready, todo, len);
+	ir_bitset_difference(ready, srcs, len);
 	if (tmp_reg != IR_REG_NONE) {
 		ir_bitset_excl(ready, tmp_reg);
 	}
 	if (tmp_fp_reg != IR_REG_NONE) {
 		ir_bitset_excl(ready, tmp_fp_reg);
 	}
-	while ((to = ir_bitset_pop_first(ready, ir_bitset_len(len))) >= 0) {
+	while ((to = ir_bitset_pop_first(ready, len)) >= 0) {
 		ir_bitset_excl(todo, to);
 		type = types[to];
 		from = pred[to];
 		if (IR_IS_CONST_REF(from)) {
-			ir_emit_dessa_move(ctx, type, to, from, tmp_reg, tmp_fp_reg);
+			ir_emit_dessa_move(ctx, mem_slots, type, to, from, tmp_reg, tmp_fp_reg);
 		} else {
 			int32_t r = loc[from];
-			ir_emit_dessa_move(ctx, type, to, r, tmp_reg, tmp_fp_reg);
+			ir_emit_dessa_move(ctx, mem_slots, type, to, r, tmp_reg, tmp_fp_reg);
 			loc[from] = to;
 			if (from == r && ir_bitset_in(todo, from) && from != tmp_reg && from != tmp_fp_reg) {
 				ir_bitset_incl(ready, from);
@@ -780,40 +866,47 @@ static int ir_dessa_parallel_copy(ir_ctx *ctx, ir_dessa_copy *copies, int count,
 	}

 	/* then we resolve all "windmill axles" - cycles (this requres temporary registers) */
-	visited = ir_bitset_malloc(len);
-	ir_bitset_copy(ready, todo, ir_bitset_len(len));
-	ir_bitset_intersection(ready, srcs, ir_bitset_len(len));
-	while ((to = ir_bitset_first(ready, ir_bitset_len(len))) >= 0) {
-		ir_bitset_clear(visited, ir_bitset_len(len));
+	ir_bitset_copy(ready, todo, len);
+	ir_bitset_intersection(ready, srcs, len);
+	while ((to = ir_bitset_first(ready, len)) >= 0) {
+		ir_bitset_clear(visited, len);
 		ir_bitset_incl(visited, to);
 		to = pred[to];
 		while (!IR_IS_CONST_REF(to) && ir_bitset_in(ready, to)) {
 			to = pred[to];
-			if (IR_IS_CONST_REF(to)) {
-				break;
-			} else if (ir_bitset_in(visited, to)) {
+			IR_ASSERT(!IR_IS_CONST_REF(to));
+			if (ir_bitset_in(visited, to)) {
 				/* We found a cycle. Resolve it. */
 				ir_bitset_incl(visited, to);
-				ir_dessa_resolve_cycle(ctx, pred, loc, types, todo, to, tmp_reg, tmp_fp_reg);
+				ir_dessa_resolve_cycle(ctx, mem_slots, pred, loc, types, todo, to, tmp_reg, tmp_fp_reg);
 				break;
 			}
 			ir_bitset_incl(visited, to);
 		}
-		ir_bitset_difference(ready, visited, ir_bitset_len(len));
+		ir_bitset_difference(ready, visited, len);
 	}

 	/* finally we resolve remaining "windmill blades" - trees that set temporary registers */
-	ir_bitset_copy(ready, todo, ir_bitset_len(len));
-	ir_bitset_difference(ready, srcs, ir_bitset_len(len));
-	while ((to = ir_bitset_pop_first(ready, ir_bitset_len(len))) >= 0) {
+	ir_bitset_copy(ready, todo, len);
+	ir_bitset_difference(ready, srcs, len);
+	while ((to = ir_bitset_pop_first(ready, len)) >= 0) {
 		ir_bitset_excl(todo, to);
 		type = types[to];
 		from = pred[to];
+#ifdef IR_DEBUG
+		/* If destionation is set, it can't be used as temporary anymore */
+		if (to == tmp_reg) {
+			tmp_reg = IR_REG_NONE;
+		}
+		if (to == tmp_fp_reg) {
+			tmp_fp_reg = IR_REG_NONE;
+		}
+#endif
 		if (IR_IS_CONST_REF(from)) {
-			ir_emit_dessa_move(ctx, type, to, from, tmp_reg, tmp_fp_reg);
+			ir_emit_dessa_move(ctx, mem_slots, type, to, from, tmp_reg, tmp_fp_reg);
 		} else {
 			int32_t r = loc[from];
-			ir_emit_dessa_move(ctx, type, to, r, tmp_reg, tmp_fp_reg);
+			ir_emit_dessa_move(ctx, mem_slots, type, to, r, tmp_reg, tmp_fp_reg);
 			loc[from] = to;
 			if (from == r && ir_bitset_in(todo, from)) {
 				ir_bitset_incl(ready, from);
@@ -821,16 +914,25 @@ static int ir_dessa_parallel_copy(ir_ctx *ctx, ir_dessa_copy *copies, int count,
 		}
 	}

-	IR_ASSERT(ir_bitset_empty(todo, ir_bitset_len(len)));
+	IR_ASSERT(ir_bitset_empty(todo, len));

-	ir_mem_free(visited);
-	ir_mem_free(ready);
-	ir_mem_free(loc);
-	ir_mem_free(srcs);
 	ir_mem_free(todo);
+	ir_mem_free(loc);
 	return 1;
 }

+static uint32_t _find_mem_slot(ir_mem *mem_slots, uint32_t *mem_slots_count, ir_mem mem)
+{
+	uint32_t j, n = *mem_slots_count;
+
+	for (j = 0; j < n; j++) {
+		if (IR_MEM_VAL(mem_slots[j]) == IR_MEM_VAL(mem)) return j;
+	}
+	mem_slots[n] = mem;
+	*mem_slots_count = n + 1;
+	return n;
+}
+
 static void ir_emit_dessa_moves(ir_ctx *ctx, int b, ir_block *bb)
 {
 	uint32_t succ, k, n = 0;
@@ -838,6 +940,8 @@ static void ir_emit_dessa_moves(ir_ctx *ctx, int b, ir_block *bb)
 	ir_use_list *use_list;
 	ir_ref i, *p;
 	ir_dessa_copy *copies;
+	ir_mem *mem_slots;
+	uint32_t mem_slots_count = 0;
 	ir_reg tmp_reg = ctx->regs[bb->end][0];
 	ir_reg tmp_fp_reg = ctx->regs[bb->end][1];

@@ -848,7 +952,13 @@ static void ir_emit_dessa_moves(ir_ctx *ctx, int b, ir_block *bb)
 	use_list = &ctx->use_lists[succ_bb->start];
 	k = ir_phi_input_number(ctx, succ_bb, b);

-	copies = alloca(use_list->count * sizeof(ir_dessa_copy));
+#if IR_X86_I64
+	copies = alloca((use_list->count - 1) * 2 * sizeof(ir_dessa_copy));
+	mem_slots = alloca((use_list->count - 1) * 2 * 2 * sizeof(ir_mem));
+#else
+	copies = alloca((use_list->count - 1) * sizeof(ir_dessa_copy));
+	mem_slots = alloca((use_list->count - 1) * 2 * sizeof(ir_mem));
+#endif

 	for (i = use_list->count, p = &ctx->use_edges[use_list->refs]; i > 0; p++, i--) {
 		ir_ref ref = *p;
@@ -859,6 +969,9 @@ static void ir_emit_dessa_moves(ir_ctx *ctx, int b, ir_block *bb)
 			ir_reg src = ir_get_alocated_reg(ctx, ref, k);
 			ir_reg dst = ctx->regs[ref][0];
 			ir_ref from, to;
+#if IR_X86_I64
+			ir_reg src_hi = IR_REG_NONE, dst_hi = IR_REG_NONE;
+#endif

 			IR_ASSERT(dst == IR_REG_NONE || !IR_REG_SPILLED(dst));
 			if (IR_IS_CONST_REF(input)) {
@@ -866,21 +979,69 @@ static void ir_emit_dessa_moves(ir_ctx *ctx, int b, ir_block *bb)
 			} else if (ir_rule(ctx, input) == IR_STATIC_ALLOCA) {
 				/* encode local variable address */
 				from = -(ctx->consts_count + input);
+			} else if (src != IR_REG_NONE && !IR_REG_SPILLED(src)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					src_hi = IR_REG_I64_HI(src);
+					src = IR_REG_I64_LO(src);
+				}
+#endif
+				from = src;
 			} else {
-				from = (src != IR_REG_NONE && !IR_REG_SPILLED(src)) ?
-					(ir_ref)src : (ir_ref)(IR_REG_NUM + ctx->vregs[input]);
+				ir_mem mem = ir_vreg_spill_slot(ctx, ctx->vregs[input]);
+
+				from = IR_REG_NUM + _find_mem_slot(mem_slots, &mem_slots_count, mem);
 			}
-			to = (dst != IR_REG_NONE) ?
-				(ir_ref)dst : (ir_ref)(IR_REG_NUM + ctx->vregs[ref]);
-			if (to != from) {
-				if (to >= IR_REG_NUM
-				 && from >= IR_REG_NUM
-				 && IR_MEM_VAL(ir_vreg_spill_slot(ctx, from - IR_REG_NUM)) ==
-						IR_MEM_VAL(ir_vreg_spill_slot(ctx, to - IR_REG_NUM))) {
-					/* It's possible that different virtual registers share the same special spill slot */
-					// TODO: See ext/opcache/tests/jit/gh11917.phpt failure on Linux 32-bit
+			if (dst != IR_REG_NONE) {
+				IR_ASSERT(!IR_REG_SPILLED(dst));
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					dst_hi = IR_REG_I64_HI(dst);
+					dst = IR_REG_I64_LO(dst);
+				}
+#endif
+				to = dst;
+			} else {
+				ir_mem mem = ir_vreg_spill_slot(ctx, ctx->vregs[ref]);
+
+				to = IR_REG_NUM + _find_mem_slot(mem_slots, &mem_slots_count, mem);
+			}
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				if (from != to) {
+					copies[n].type = IR_U32;
+					copies[n].from = from;
+					copies[n].to = to;
+					n++;
+				} else if (from >= IR_REG_NUM) {
 					continue;
 				}
+				if (from < 0) {
+					/* pass */
+				} else if (from < IR_REG_NUM) {
+					from = src_hi;
+				} else {
+					ir_mem mem = IR_MEM_I64_HI(mem_slots[from - IR_REG_NUM]);
+
+					from = IR_REG_NUM + _find_mem_slot(mem_slots, &mem_slots_count, mem);
+				}
+				if (to < IR_REG_NUM) {
+					to = dst_hi;
+				} else {
+					ir_mem mem = IR_MEM_I64_HI(mem_slots[to - IR_REG_NUM]);
+
+					to = IR_REG_NUM + _find_mem_slot(mem_slots, &mem_slots_count, mem);
+				}
+				if (from != to) {
+					copies[n].type = (from < 0) ? IR_U32_HI : IR_U32;
+					copies[n].from = from;
+					copies[n].to = to;
+					n++;
+				}
+				continue;
+			}
+#endif
+			if (to != from) {
 				copies[n].type = insn->type;
 				copies[n].from = from;
 				copies[n].to = to;
@@ -890,7 +1051,254 @@ static void ir_emit_dessa_moves(ir_ctx *ctx, int b, ir_block *bb)
 	}

 	if (n > 0) {
-		ir_dessa_parallel_copy(ctx, copies, n, tmp_reg, tmp_fp_reg);
+		ir_dessa_parallel_copy(ctx, copies, n, mem_slots, mem_slots_count, tmp_reg, tmp_fp_reg);
+	}
+}
+
+/* TAILCALL optimization */
+static bool ir_may_be_local_addr(ir_ctx *ctx, ir_insn *insn)
+{
+	if (insn->op == IR_PARAM) return 0;
+
+	return 1;
+}
+
+static bool ir_try_tailcall(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+{
+	ir_ref proto_ref = IR_UNUSED;
+	const ir_proto_t *proto = NULL;
+	const ir_call_conv_dsc *cc;
+	int32_t params_stack_size, copy_stack;
+
+	if (IR_IS_CONST_REF(insn->op2)) {
+		const ir_insn *func = &ctx->ir_base[insn->op2];
+
+		if (func->op == IR_FUNC && func->proto) {
+			uint32_t rule = ir_match_builtin_call(ctx, func);
+
+			if (rule) {
+				ctx->rules[ref] = rule;
+				return 0;
+			}
+			proto_ref = func->proto;
+		} else if (func->op == IR_FUNC_ADDR) {
+			proto_ref = func->proto;
+		}
+	} else if (ctx->ir_base[insn->op2].op == IR_PROTO) {
+		proto_ref = ctx->ir_base[insn->op2].op2;
+	}
+
+	if (!proto_ref) return 0;
+	proto = (const ir_proto_t *)ir_get_str(ctx, proto_ref);
+
+	if ((proto->flags & IR_CALL_CONV_MASK) != (ctx->flags & IR_CALL_CONV_MASK)) return 0;
+
+	cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
+	copy_stack = 0;
+	params_stack_size = ir_call_used_stack(ctx, insn, cc, &copy_stack);
+	if (cc->shadow_store_size && params_stack_size == cc->shadow_store_size) {
+		params_stack_size = 0;
+	}
+
+	// TODO: "params_stack_size" must match the "args_stack_size"
+	if (params_stack_size) return 0;
+
+	/* check for passing addresses of local variable */
+	uint32_t n = insn->inputs_count;
+	for (uint32_t i = 3; i <= n; i++) {
+		ir_ref input = ir_insn_op(insn, i);
+		if (!IR_IS_CONST_REF(input) && ctx->ir_base[input].type == IR_ADDR) {
+			/* Passing addrss of local varible to TAILCALL is disallowd */
+			if (ir_may_be_local_addr(ctx, &ctx->ir_base[input])) {
+				return 0;
+			}
+		}
+	}
+
+#if defined(IR_TARGET_X64) || defined(IR_TARGET_X86)
+	if (!IR_IS_CONST_REF(insn->op2)) {
+		if (ctx->ir_base[insn->op2].op == IR_PROTO) {
+			if (IR_IS_CONST_REF(ctx->ir_base[insn->op2].op1)) {
+				ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_PROTO;
+			} else {
+				ir_match_fuse_load(ctx, ctx->ir_base[insn->op2].op1, ref);
+				if (ctx->rules[ctx->ir_base[insn->op2].op1] & IR_FUSED) {
+					ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_PROTO;
+				}
+		   }
+		} else {
+			ir_match_fuse_load(ctx, insn->op2, ref);
+		}
+	}
+#endif
+
+	ctx->rules[ref] = IR_TAILCALL | IR_NO_REG;
+
+	return 1;
+}
+
+#if 0
+static bool ir_try_tailcalls(ir_ctx *ctx, ir_insn *merge_insn)
+{
+	ir_ref count = 0, n = merge_insn->inputs_count;
+	ir_ref end, ref, *p = merge_insn->ops + 1;
+	ir_insn *insn;
+
+	do {
+		end = *p;
+		insn = &ctx->ir_base[end];
+		IR_ASSERT(insn->op == IR_END);
+		ref = insn->op1;
+		insn = &ctx->ir_base[ref];
+		if (insn->op == IR_CALL) {
+			if (ir_try_tailcall(ctx, ref, insn)) {
+				ctx->rules[end] = IR_SKIPPED | IR_NOP;
+				count++;
+			}
+		} else if (insn->op == IR_MERGE) {
+			if (ir_try_tailcalls(ctx, insn)) {
+				ctx->rules[end] = IR_SKIPPED | IR_NOP;
+				ctx->rules[ref] = IR_SKIPPED | IR_NOP;
+				count++;
+			}
+		}
+		p++;
+	} while (--n != 0);
+
+	return count == merge_insn->inputs_count;
+}
+#endif
+
+static size_t ir_calc_args_stack(const ir_ctx *ctx)
+{
+	ir_use_list *use_list = &ctx->use_lists[1];
+	ir_insn *insn;
+	ir_ref i, n, *p, use;
+	int int_param_num = 0;
+	int fp_param_num = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param_num = 0;
+#endif
+	ir_reg src_reg;
+	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(ctx->flags);
+	int32_t stack_offset = 0;
+
+	n = use_list->count;
+	for (i = 0, p = &ctx->use_edges[use_list->refs]; i < n; i++, p++) {
+		use = *p;
+		insn = &ctx->ir_base[use];
+		if (insn->op == IR_PARAM) {
+			if (IR_IS_TYPE_INT(insn->type)) {
+				if (ctx->value_params && ctx->value_params[insn->op3 - 1].align) {
+					/* struct passed by value on stack */
+					uint32_t align = ctx->value_params[insn->op3 - 1].align;
+
+					align = IR_MAX(sizeof(void*), align);
+					stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
+					stack_offset += ctx->value_params[insn->op3 - 1].size;
+					stack_offset = IR_ALIGNED_SIZE(stack_offset, sizeof(void*));
+					continue;
+				} else if (int_param_num < cc->int_param_regs_count) {
+					src_reg = cc->int_param_regs[int_param_num];
+#if IR_X86_I64
+					if (src_reg != IR_REG_NONE && (insn->type == IR_I64 || insn->type == IR_U64)) {
+						if (int_param_num + 1 < cc->int_param_regs_count) {
+							int_param_num++;
+							if (cc->shadow_param_regs) {
+								fp_param_num++;
+							}
+						}
+						src_reg = IR_REG_NONE;
+					}
+#endif
+				} else {
+					src_reg = IR_REG_NONE;
+				}
+				int_param_num++;
+				if (cc->shadow_param_regs) {
+					fp_param_num++;
+				}
+#if IR_SIMD && defined(IR_TARGET_X86)
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (vector_param_num < cc->vector_param_regs_count) {
+					src_reg = cc->vector_param_regs[vector_param_num];
+				} else {
+					src_reg = IR_REG_NONE;
+				}
+				vector_param_num++;
+#endif
+			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(insn->type) || IR_IS_TYPE_VECTOR(insn->type));
+				if (fp_param_num < cc->fp_param_regs_count) {
+					src_reg = cc->fp_param_regs[fp_param_num];
+				} else {
+					src_reg = IR_REG_NONE;
+				}
+				fp_param_num++;
+				if (cc->shadow_param_regs) {
+					int_param_num++;
+				}
+			}
+			if (src_reg == IR_REG_NONE) {
+				if (sizeof(void*) == 8) {
+					stack_offset += sizeof(void*);
+				} else {
+					stack_offset += IR_MAX(sizeof(void*), ir_get_type_size(insn->type));
+				}
+			}
+		}
+	}
+
+	return stack_offset;
+}
+
+static void ir_match_tailcalls(ir_ctx *ctx)
+{
+	ir_ref ref;
+	ir_insn *insn;
+	size_t args_stack_size = (size_t)-1;
+
+	ref = ctx->ir_base[1].op1;
+	while (ref) {
+		insn = &ctx->ir_base[ref];
+		if (insn->op == IR_RETURN) {
+			if (insn->op1 == insn->op2) {
+				if (ctx->ir_base[insn->op1].op == IR_CALL) {
+					if (args_stack_size == (size_t)-1) {
+						args_stack_size = ir_calc_args_stack(ctx);
+						// TODO: "args_stack_size" must match the "params_stack_size"
+						if (args_stack_size) return;
+					}
+					if (ir_try_tailcall(ctx, insn->op1, &ctx->ir_base[insn->op1])) {
+						ctx->rules[ref] = IR_SKIPPED | IR_NOP;
+					}
+				}
+			} else if (insn->op2 == IR_UNUSED) {
+				if (ctx->ir_base[insn->op1].op == IR_CALL) {
+					if (args_stack_size == (size_t)-1) {
+						args_stack_size = ir_calc_args_stack(ctx);
+						// TODO: "args_stack_size" must match the "params_stack_size"
+						if (args_stack_size) return;
+					}
+					if (ir_try_tailcall(ctx, insn->op1, &ctx->ir_base[insn->op1])) {
+						ctx->rules[ref] = IR_SKIPPED | IR_NOP;
+					}
+#if 0
+				} else if (ctx->ir_base[insn->op1].op == IR_MERGE) {
+					if (args_stack_size == (size_t)-1) {
+						args_stack_size = ir_calc_args_stack(ctx);
+						// TODO: "args_stack_size" must match the "params_stack_size"
+						if (args_stack_size) return;
+					}
+					if (ir_try_tailcalls(ctx, &ctx->ir_base[insn->op1])) {
+						ctx->rules[insn->op1] = IR_SKIPPED | IR_NOP;
+						ctx->rules[ref] = IR_SKIPPED | IR_NOP;
+					}
+#endif
+				}
+			}
+		}
+		ref = insn->op3;
 	}
 }

@@ -914,6 +1322,12 @@ int ir_match(ir_ctx *ctx)
 		ctx->entries = ir_mem_malloc(ctx->entries_count * sizeof(ir_ref));
 	}

+	if ((ctx->flags & IR_OPT_TAILCALL)
+	 && (ctx->flags & IR_FUNCTION)
+	 && !(ctx->flags & IR_VARARG_FUNC)) {
+		ir_match_tailcalls(ctx);
+	}
+
 	for (b = ctx->cfg_blocks_count, bb = ctx->cfg_blocks + b; b > 0; b--, bb--) {
 		IR_ASSERT(!(bb->flags & IR_BB_UNREACHABLE));
 		start = bb->start;
@@ -1013,3 +1427,549 @@ const ir_call_conv_dsc *ir_get_call_conv_dsc(uint32_t flags)
 	IR_ASSERT((flags & IR_CALL_CONV_MASK) == IR_CC_DEFAULT || (flags & IR_CALL_CONV_MASK) == IR_CC_BUILTIN);
 	return &ir_call_conv_default;
 }
+
+/* Simple Register Allocator */
+typedef struct {
+	int32_t  num;
+	ir_regset preserved_regs;
+	ir_regset clobbered[IR_SUB_REFS_COUNT];
+	struct {
+		uint8_t type;
+		int8_t  start;
+		int8_t  end;
+		int8_t  hint;
+		int8_t  flags;
+		ir_ref  root;
+		ir_ref  ref;
+		ir_ref  op;
+	} regs[32];
+} ir_reg_alloc_simple_data;
+
+static void _add_scratch(ir_reg_alloc_simple_data *x, ir_reg reg, int8_t start, int8_t end)
+{
+	int8_t j;
+
+	if (start < 0) start = 0; // TODO: ARGVAL support ???
+	IR_ASSERT(start >= 0 && end <= IR_SUB_REFS_COUNT);
+	if (reg >= IR_REG_NUM) {
+		for (j = start; j < end; j++) {
+			x->clobbered[j] = IR_REGSET_UNION(x->clobbered[j], ir_scratch_regset[reg - IR_REG_NUM]);
+		}
+	} else {
+		for (j = start; j < end; j++) {
+			IR_REGSET_INCL(x->clobbered[j], reg);
+		}
+	}
+}
+
+static void _add_reg(ir_reg_alloc_simple_data *x, ir_type type,
+                     int8_t start, int8_t end, ir_reg hint, int8_t flags,
+                     ir_ref root, ir_ref ref, ir_ref op)
+{
+	IR_ASSERT(start >= 0 && end <= IR_SUB_REFS_COUNT && x->num < 32);
+	x->regs[x->num].type = type;
+	x->regs[x->num].start = start;
+	x->regs[x->num].end = end;
+	x->regs[x->num].hint = hint;
+	x->regs[x->num].flags = flags;
+	x->regs[x->num].root = root;
+	x->regs[x->num].ref = ref;
+	x->regs[x->num].op = op;
+	x->num++;
+}
+
+static ir_reg _get_free_reg(ir_type type, ir_regset available)
+{
+	if (IR_IS_TYPE_INT(type)) {
+		available = IR_REGSET_INTERSECTION(available, IR_REGSET_GP);
+	} else {
+		IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
+		available = IR_REGSET_INTERSECTION(available, IR_REGSET_FP);
+	}
+	if (!IR_REGSET_IS_EMPTY(available)) {
+		return IR_REGSET_FIRST(available);
+	} else {
+		return IR_REG_NONE;
+	}
+}
+
+static ir_reg _get_free_reg2(ir_ctx *ctx, ir_type type, ir_reg_alloc_simple_data *x, int j)
+{
+	int n;
+	ir_regset available;
+	ir_reg reg;
+
+	if (IR_IS_TYPE_INT(type)) {
+		available = IR_REGSET_GP;
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			IR_REGSET_EXCL(available, IR_REG_FRAME_POINTER);
+		}
+
+#if defined(IR_TARGET_X86)
+		if (ir_type_size[type] == 1) {
+			/* TODO: if no registers avialivle, we may use of one this register for already allocated interval ??? */
+			IR_REGSET_EXCL(available, IR_REG_RBP);
+			IR_REGSET_EXCL(available, IR_REG_RSI);
+			IR_REGSET_EXCL(available, IR_REG_RDI);
+		}
+#endif
+	} else {
+		IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
+		available = IR_REGSET_FP;
+	}
+	for (n = x->regs[j].start; n < x->regs[j].end; n++) {
+		available = IR_REGSET_DIFFERENCE(available, x->clobbered[n]);
+	}
+	if (IR_REGSET_IS_EMPTY(available)) {
+		fprintf(stderr, "Internal Error: No registers available. Allocation is not possible\n");
+		IR_ASSERT(0);
+		exit(-1);
+	}
+
+	reg = IR_REGSET_FIRST(available);
+	if (IR_REGSET_IN(x->preserved_regs, reg)) {
+		IR_REGSET_INCL(ctx->used_preserved_regs, reg);
+	}
+	return reg;
+}
+
+static void ir_set_fused_reg(ir_ctx *ctx, ir_ref root, ir_ref ref_and_op, int8_t reg)
+{
+	char key[10];
+
+	if (!ctx->fused_regs) {
+		ctx->fused_regs = ir_mem_malloc(sizeof(ir_strtab));
+		ir_strtab_init(ctx->fused_regs, 8, 128);
+	}
+	memcpy(key, &root, sizeof(ir_ref));
+	memcpy(key + 4, &ref_and_op, sizeof(ir_ref));
+	ir_strtab_lookup(ctx->fused_regs, key, 8, 0x10000000 | (uint8_t)reg);
+}
+
+static bool ir_load_may_reuse_var_slot(ir_ctx *ctx, ir_block *bb, ir_ref var, ir_ref load)
+{
+	ir_use_list *use_list = &ctx->use_lists[load];
+	ir_ref *p, use, i, n = use_list->count;
+	ir_ref last_use = IR_UNUSED;
+	ir_insn *insn;
+
+	if (n) {
+		for (p = ctx->use_edges + use_list->refs; n > 0; p++, n--) {
+			use = *p;
+			if (use < load || use > bb->end) return 0;
+			if (use > last_use) last_use = use;
+		}
+		for (i = load + 1, insn = &ctx->ir_base[i]; i < last_use;) {
+			if ((insn->op == IR_VSTORE || insn->op == IR_VSTORE_v) && insn->op2 == var) {
+				return 0;
+			}
+			n = ir_insn_len(insn);
+			i += n;
+			insn += n;
+		}
+	}
+	return 1;
+}
+
+static bool ir_store_may_reuse_var_slot(ir_ctx *ctx, ir_block *bb, ir_ref var, ir_ref store, ir_ref val)
+{
+	ir_ref i, n;
+	ir_insn *insn;
+
+	if (val < bb->start && val > store) return 0;
+
+	for (i = val, insn = &ctx->ir_base[i]; i < store;) {
+		if ((insn->op == IR_VLOAD || insn->op == IR_VLOAD_v || insn->op == IR_VSTORE || insn->op == IR_VSTORE_v)
+		 && insn->op2 == var) {
+			return 0;
+		}
+		n = ir_insn_len(insn);
+		i += n;
+		insn += n;
+	}
+	return 1;
+}
+
+static void ir_add_fusion_data(ir_ctx *ctx, ir_ref ref, ir_ref input, ir_reg_alloc_simple_data *x)
+{
+	ir_ref stack[4];
+	int stack_pos = 0;
+	ir_target_constraints constraints;
+	ir_insn *insn;
+	uint32_t j, n, flags, def_flags;
+	ir_ref *p, child;
+
+	while (1) {
+		IR_ASSERT(input > 0 && ctx->rules[input] & IR_FUSED);
+
+		if (!(ctx->rules[input] & IR_SIMPLE)) {
+			def_flags = ir_get_target_constraints(ctx, input, &constraints);
+			n = constraints.tmps_count;
+			while (n > 0) {
+				n--;
+				if (constraints.tmp_regs[n].type) {
+					ir_reg flags = 0;
+					ir_ref op = constraints.tmp_regs[n].num;
+
+					if (op > 0 && op <= ctx->ir_base[input].inputs_count) {
+						ir_ref *ops = ctx->ir_base[input].ops;
+
+						if (IR_IS_CONST_REF(ops[op])) {
+							/* rematerialization */
+							flags = IR_REG_SPILL_LOAD;
+						} else if (ctx->rules[ops[op]] == IR_STATIC_ALLOCA) {
+							/* local address rematerialization */
+							flags = IR_REG_SPILL_LOAD;
+						}
+					}
+					_add_reg(x, constraints.tmp_regs[n].type,
+						constraints.tmp_regs[n].start, constraints.tmp_regs[n].end, IR_REG_NONE, flags,
+						IR_UNUSED, input, op);
+				} else {
+					_add_scratch(x, constraints.tmp_regs[n].reg,
+						constraints.tmp_regs[n].start, constraints.tmp_regs[n].end);
+				}
+			}
+		} else {
+			def_flags = IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG | IR_OP3_MUST_BE_IN_REG;
+			constraints.hints_count = 0;
+		}
+
+		insn = &ctx->ir_base[input];
+		flags = ir_op_flags[insn->op];
+		n = IR_INPUT_EDGES_COUNT(flags);
+		j = 1;
+		p = insn->ops + j;
+		if (flags & (IR_OP_FLAG_CONTROL|IR_OP_FLAG_PINNED)) {
+			j++;
+			p++;
+		}
+		for (; j <= n; j++, p++) {
+			IR_ASSERT(IR_OPND_KIND(flags, j) == IR_OPND_DATA);
+			child = *p;
+			if (child > 0) {
+				if (ctx->vregs[child]) {
+					if (IR_USE_FLAGS(def_flags, j) & IR_USE_MUST_BE_IN_REG) {
+						ir_reg reg = (j < constraints.hints_count) ? constraints.hints[j] : IR_REG_NONE;
+						int8_t use_pos = EXPECTED(reg == IR_REG_NONE) ? IR_USE_SUB_REF : IR_LOAD_SUB_REF;
+
+						_add_reg(x, ctx->ir_base[child].type, IR_LOAD_SUB_REF, use_pos, reg, IR_REG_SPILL_LOAD,
+							ref, input, j);
+					}
+				} else if (ctx->rules[child] & IR_FUSED) {
+					IR_ASSERT(stack_pos < (int)(sizeof(stack)/sizeof(stack_pos)));
+					stack[stack_pos++] = child;
+				} else if (ctx->rules[child] == (IR_SKIPPED|IR_RLOAD)) {
+					ctx->regs[input][j] = ctx->ir_base[child].op2;
+				}
+			}
+		}
+		if (!stack_pos) {
+			break;
+		}
+		input = stack[--stack_pos];
+	}
+}
+
+int ir_reg_alloc_simple(ir_ctx *ctx)
+{
+	ir_reg_alloc_data data;
+	uint32_t b;
+	ir_block *bb;
+	ir_insn *insn;
+	ir_ref i, n, j, *p;
+	uint32_t *rule, insn_flags;
+	ir_target_constraints constraints;
+	uint32_t def_flags;
+	ir_reg reg;
+	ir_regset scratch;
+	ir_reg_alloc_simple_data x;
+
+	memset(&data, 0, sizeof(data));
+	data.cc = ir_get_call_conv_dsc(ctx->flags);
+	ctx->data = &data;
+
+	ctx->stack_frame_size = 0;
+	ctx->call_stack_size = 0;
+	ctx->used_preserved_regs = 0;
+	ctx->used_preserved_regs = ctx->fixed_save_regset;
+
+	scratch = ir_scratch_regset[data.cc->scratch_reg - IR_REG_NUM];
+	x.preserved_regs = IR_REGSET_DIFFERENCE(data.cc->preserved_regs, ctx->fixed_save_regset);
+
+	ctx->regs = ir_mem_malloc(sizeof(ir_regs) * ctx->insns_count);
+	memset(ctx->regs, IR_REG_NONE, sizeof(ir_regs) * ctx->insns_count);
+
+	/* vregs + tmp + fixed + SRATCH + ALL */
+	ctx->live_intervals = ir_mem_calloc(ctx->vregs_count + 1 + IR_REG_SET_NUM, sizeof(ir_live_interval*));
+
+	if (!ctx->arena) {
+		ctx->arena = ir_arena_create(16 * 1024);
+	}
+
+	for (b = 1, bb = ctx->cfg_blocks + b; b <= ctx->cfg_blocks_count; b++, bb++) {
+		IR_ASSERT(!(bb->flags & IR_BB_UNREACHABLE));
+		for (i = bb->start, insn = ctx->ir_base + i, rule = ctx->rules + i; i <= bb->end;) {
+			if (*rule & (IR_FUSED|IR_SKIPPED)) {
+				if ((*rule & IR_RULE_MASK) == IR_ALLOCA) {
+					if (insn->op == IR_VAR) {
+						if (ctx->use_lists[i].count > 0) {
+							insn->op3 = ir_allocate_spill_slot(ctx, insn->type);
+						}
+					} else if (insn->op == IR_ALLOCA) {
+						if (ctx->use_lists[i].count > 0) {
+							ir_insn *val = &ctx->ir_base[insn->op2];
+
+							IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+							IR_ASSERT(IR_IS_TYPE_INT(val->type));
+							IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+							IR_ASSERT(IR_IS_TYPE_UNSIGNED(val->type) || val->val.i64 >= 0);
+							IR_ASSERT(val->val.i64 < 0x7fffffff);
+							insn->op3 = ir_allocate_big_spill_slot(ctx, val->val.i32);
+						}
+					} else if (insn->op == IR_VADDR) {
+						insn->op3 = ctx->ir_base[insn->op1].op3;
+					}
+				}
+			} else {
+				x.num = 0;
+				for (j = 0; j < IR_SUB_REFS_COUNT; j++) {
+					x.clobbered[j] = IR_REGSET_EMPTY;
+				}
+
+				def_flags = ir_get_target_constraints(ctx, i, &constraints);
+				n = constraints.tmps_count;
+				while (n) {
+					n--;
+
+					IR_ASSERT(constraints.tmp_regs[n].start >= 0 && constraints.tmp_regs[n].end < IR_SUB_REFS_COUNT);
+					if (constraints.tmp_regs[n].type) {
+						ir_reg flags = 0;
+						ir_ref op = constraints.tmp_regs[n].num;
+
+						if (op > 0 && op <= insn->inputs_count) {
+							ir_ref *ops = insn->ops;
+
+							if (IR_IS_CONST_REF(ops[op])) {
+								/* rematerialization */
+								flags = IR_REG_SPILL_LOAD;
+							} else if (ctx->rules[ops[op]] == IR_STATIC_ALLOCA) {
+								/* local address rematerialization */
+								flags = IR_REG_SPILL_LOAD;
+							}
+						}
+						_add_reg(&x, constraints.tmp_regs[n].type,
+							constraints.tmp_regs[n].start, constraints.tmp_regs[n].end, IR_REG_NONE, flags,
+							IR_UNUSED, i, op);
+					} else {
+						_add_scratch(&x, constraints.tmp_regs[n].reg,
+							constraints.tmp_regs[n].start, constraints.tmp_regs[n].end);
+					}
+				}
+
+				if (ctx->vregs[i]) {
+					reg = constraints.def_reg;
+					if (!ctx->live_intervals[ctx->vregs[i]]) {
+						ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
+						memset(ival, 0, sizeof(ir_live_interval));
+						ctx->live_intervals[ctx->vregs[i]] = ival;
+						ival->type = insn->type;
+						ival->reg = IR_REG_NONE;
+						ival->vreg = ctx->vregs[i];
+						ival->stack_spill_pos = -1;
+						if ((insn->op == IR_VLOAD || insn->op == IR_VLOAD_v)
+						 && ir_load_may_reuse_var_slot(ctx, bb, insn->op2, i)) {
+							ival->stack_spill_pos = ctx->ir_base[insn->op2].op3;
+							reg = IR_REG_NONE;
+							def_flags &= ~IR_USE_MUST_BE_IN_REG;
+						} else if (insn->op == IR_PARAM && reg == IR_REG_NONE) {
+							ival->flags |= IR_LIVE_INTERVAL_MEM_PARAM;
+						} else if (ctx->use_lists[i].count == 1) {
+							ir_ref use = ctx->use_edges[ctx->use_lists[i].refs];
+							ir_insn *use_insn = &ctx->ir_base[use];
+
+							if ((use_insn->op == IR_VSTORE || use_insn->op == IR_VSTORE_v)
+							 && use_insn->op3 == i
+							 && ir_store_may_reuse_var_slot(ctx, bb, use_insn->op2, use, i)) {
+								if (use_insn->op2 < i) {
+									ival->stack_spill_pos = ctx->ir_base[use_insn->op2].op3;
+								} else {
+									ival->stack_spill_pos = ctx->ir_base[use_insn->op2].op3 =
+										ir_allocate_spill_slot(ctx, ival->type);
+								}
+							} else {
+								ival->stack_spill_pos = ir_allocate_spill_slot(ctx, ival->type);
+							}
+						} else {
+							ival->stack_spill_pos = ir_allocate_spill_slot(ctx, ival->type);
+						}
+					} else if (insn->op == IR_PARAM) {
+						IR_ASSERT(0 && "unexpected PARAM");
+						return 0;
+					}
+
+					if (def_flags & IR_USE_MUST_BE_IN_REG) {
+						ir_live_pos def_pos;
+
+						if (reg != IR_REG_NONE) {
+							def_pos = IR_SAVE_SUB_REF;
+						} else if (def_flags & IR_DEF_REUSES_OP1_REG) {
+							if (def_flags & IR_DEF_CONFLICTS_WITH_INPUT_REGS) {
+								def_pos = IR_USE_SUB_REF;
+							} else {
+								def_pos = IR_LOAD_SUB_REF;
+							}
+						} else if (def_flags & IR_DEF_CONFLICTS_WITH_INPUT_REGS) {
+							def_pos = IR_LOAD_SUB_REF;
+						} else {
+							if (insn->op == IR_PARAM) {
+								/* We may reuse parameter stack slot for spilling */
+								ctx->live_intervals[ctx->vregs[i]]->flags |= IR_LIVE_INTERVAL_MEM_PARAM;
+							}
+							def_pos = IR_DEF_SUB_REF;
+						}
+
+						_add_reg(&x, insn->type, def_pos, IR_SUB_REFS_COUNT, reg, IR_REG_SPILL_STORE,
+							IR_UNUSED, i, 0);
+					}
+				}
+
+				n = insn->inputs_count;
+				insn_flags = ir_op_flags[insn->op];
+				j = 1;
+				p = insn->ops + 1;
+				if (insn_flags & (IR_OP_FLAG_CONTROL|IR_OP_FLAG_MEM|IR_OP_FLAG_PINNED)) {
+					j++;
+					p++;
+				}
+				for (; j <= n; j++, p++) {
+					ir_ref input = *p;
+					ir_reg reg = (j < constraints.hints_count) ? constraints.hints[j] : IR_REG_NONE;
+					ir_live_pos use_pos;
+					uint32_t use_flags = IR_USE_FLAGS(def_flags, j);
+
+					if (input > 0) {
+						if (ctx->vregs[input]) {
+							use_pos = IR_USE_SUB_REF;
+							if (reg != IR_REG_NONE) {
+								use_pos = IR_LOAD_SUB_REF;
+#if IR_X86_I64
+								if (use_flags & IR_HINT_TWO_REGS) {
+									IR_REGSET_INCL(x.clobbered[IR_LOAD_SUB_REF], IR_REG_I64_LO(reg));
+									IR_REGSET_INCL(x.clobbered[IR_LOAD_SUB_REF], IR_REG_I64_HI(reg));
+								} else
+#endif
+								IR_REGSET_INCL(x.clobbered[IR_LOAD_SUB_REF], reg);
+							} else if (def_flags & IR_DEF_REUSES_OP1_REG) {
+								if (j == 1) {
+									if (def_flags & IR_DEF_CONFLICTS_WITH_INPUT_REGS) {
+										use_pos = IR_USE_SUB_REF;
+									} else {
+										use_pos = IR_LOAD_SUB_REF;
+									}
+								} else if (input == insn->op1) {
+									/* Input is the same as "op1" */
+									use_pos = IR_LOAD_SUB_REF;
+								}
+							}
+							if (use_flags & IR_USE_MUST_BE_IN_REG) {
+								_add_reg(&x, ctx->ir_base[input].type, IR_LOAD_SUB_REF, use_pos, reg, IR_REG_SPILL_LOAD,
+									IR_UNUSED, i, j);
+							}
+						} else {
+							if ((ctx->rules[input] & (IR_FUSED|IR_SKIPPED)) == IR_FUSED) {
+								ir_add_fusion_data(ctx, i, input, &x);
+							} else if (ctx->rules[input] == (IR_SKIPPED|IR_RLOAD)) {
+								ctx->regs[i][j] = ctx->ir_base[input].op2;
+							}
+						}
+					}
+				}
+
+				for (j = 0; j < x.num; j++) {
+					ir_regset available = scratch;
+#if IR_X86_I64
+					ir_reg reg2;
+#endif
+
+					for (n = x.regs[j].start; n < x.regs[j].end; n++) {
+						available = IR_REGSET_DIFFERENCE(available, x.clobbered[n]);
+					}
+					reg = x.regs[j].hint;
+#if IR_X86_I64
+					reg2 = IR_REG_NONE;
+					if (reg != IR_REG_NONE && (x.regs[j].type == IR_I64 || x.regs[j].type == IR_U64)) {
+						reg2 = IR_REG_I64_HI(reg);
+						reg = IR_REG_I64_LO(reg);
+					}
+#endif
+					if (reg == IR_REG_NONE || !IR_REGSET_IN(available, reg)) {
+						reg = _get_free_reg(x.regs[j].type, available);
+						if (UNEXPECTED(reg == IR_REG_NONE)) {
+							reg = _get_free_reg2(ctx, x.regs[j].type, &x, j);
+						}
+					}
+					for (n = x.regs[j].start; n < x.regs[j].end; n++) {
+						IR_REGSET_INCL(x.clobbered[n], reg);
+					}
+#if IR_X86_I64
+					if (x.regs[j].type == IR_I64 || x.regs[j].type == IR_U64) {
+						IR_REGSET_EXCL(available, reg);
+						if (reg2 == IR_REG_NONE || !IR_REGSET_IN(available, reg2)) {
+							reg2 = _get_free_reg(x.regs[j].type, available);
+							if (UNEXPECTED(reg2 == IR_REG_NONE)) {
+								reg2 = _get_free_reg2(ctx, x.regs[j].type, &x, j);
+							}
+						}
+						for (n = x.regs[j].start; n < x.regs[j].end; n++) {
+							IR_REGSET_INCL(x.clobbered[n], reg2);
+						}
+						if (reg > reg2) {
+							SWAP_REGS(reg, reg2);
+						}
+						reg = IR_REG_I64_PAIR(reg, reg2);
+					}
+#endif
+					reg = reg | x.regs[j].flags;
+					if (x.regs[j].op == 4 && insn->inputs_count < 4) {
+						if (!ctx->tmp_regs) {
+							ctx->tmp_regs = ir_mem_malloc(ctx->insns_count);
+							memset(ctx->tmp_regs, -1, ctx->insns_count);
+						}
+						ctx->tmp_regs[x.regs[j].ref] = reg;
+					} else if (!x.regs[j].root || ctx->regs[x.regs[j].ref][x.regs[j].op] == IR_REG_NONE) {
+						ctx->regs[x.regs[j].ref][x.regs[j].op] = reg;
+					} else if (ctx->regs[x.regs[j].ref][x.regs[j].op] != reg) {
+						ctx->rules[x.regs[j].ref] |= IR_FUSED_REG;
+						ir_set_fused_reg(ctx, x.regs[j].root, x.regs[j].ref * sizeof(ir_ref) + x.regs[j].op, reg);
+					}
+				}
+			}
+
+			n = ir_insn_len(insn);
+			i += n;
+			insn += n;
+			rule += n;
+		}
+		if (bb->flags & IR_BB_DESSA_MOVES) {
+			ir_gen_dessa_moves(ctx, b, ir_fix_dessa_tmps, (void*)(intptr_t)b);
+		}
+	}
+
+#ifdef IR_TARGET_X86
+	if (ctx->flags2 & IR_HAS_FP_RET_SLOT) {
+		ctx->ret_slot = ir_allocate_spill_slot(ctx, IR_DOUBLE);
+	} else if ((ctx->ret_type == IR_FLOAT || ctx->ret_type == IR_DOUBLE)
+			&& data.cc->fp_ret_reg == IR_REG_NONE) {
+		ctx->ret_slot = ir_allocate_spill_slot(ctx, ctx->ret_type);
+	} else {
+		ctx->ret_slot = -1;
+	}
+#endif
+
+	ctx->flags |= IR_NO_STACK_COMBINE;
+	ir_fix_stack_frame(ctx);
+	ctx->data = NULL;
+
+	return 1;
+}
diff --git a/ext/opcache/jit/ir/ir_fold.h b/ext/opcache/jit/ir/ir_fold.h
index cbe049be932..c5fd2f28894 100644
--- a/ext/opcache/jit/ir/ir_fold.h
+++ b/ext/opcache/jit/ir/ir_fold.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Folding engine rules)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * Based on Mike Pall's implementation for LuaJIT.
@@ -60,12 +60,28 @@ IR_FOLD(NE(C_FLOAT, C_FLOAT))

 IR_FOLD(LT(C_BOOL, C_BOOL))
 IR_FOLD(LT(C_U8, C_U8))
+{
+	IR_FOLD_BOOL(op1_insn->val.i8 < op2_insn->val.i8);
+}
+
 IR_FOLD(LT(C_U16, C_U16))
+{
+	IR_FOLD_BOOL(op1_insn->val.i16 < op2_insn->val.i16);
+}
+
 IR_FOLD(LT(C_U32, C_U32))
+{
+	IR_FOLD_BOOL(op1_insn->val.i32 < op2_insn->val.i32);
+}
+
 IR_FOLD(LT(C_U64, C_U64))
+{
+	IR_FOLD_BOOL(op1_insn->val.i64 < op2_insn->val.i64);
+}
+
 IR_FOLD(LT(C_ADDR, C_ADDR))
 {
-	IR_FOLD_BOOL(op1_insn->val.u64 < op2_insn->val.u64);
+	IR_FOLD_BOOL(op1_insn->val.addr < op2_insn->val.addr);
 }

 IR_FOLD(LT(C_CHAR, C_CHAR))
@@ -89,12 +105,28 @@ IR_FOLD(LT(C_FLOAT, C_FLOAT))

 IR_FOLD(GE(C_BOOL, C_BOOL))
 IR_FOLD(GE(C_U8, C_U8))
+{
+	IR_FOLD_BOOL(op1_insn->val.i8 >= op2_insn->val.i8);
+}
+
 IR_FOLD(GE(C_U16, C_U16))
+{
+	IR_FOLD_BOOL(op1_insn->val.i16 >= op2_insn->val.i16);
+}
+
 IR_FOLD(GE(C_U32, C_U32))
+{
+	IR_FOLD_BOOL(op1_insn->val.i32 >= op2_insn->val.i32);
+}
+
 IR_FOLD(GE(C_U64, C_U64))
+{
+	IR_FOLD_BOOL(op1_insn->val.i64 >= op2_insn->val.i64);
+}
+
 IR_FOLD(GE(C_ADDR, C_ADDR))
 {
-	IR_FOLD_BOOL(op1_insn->val.u64 >= op2_insn->val.u64);
+	IR_FOLD_BOOL(op1_insn->val.addr >= op2_insn->val.addr);
 }

 IR_FOLD(GE(C_CHAR, C_CHAR))
@@ -118,12 +150,28 @@ IR_FOLD(GE(C_FLOAT, C_FLOAT))

 IR_FOLD(LE(C_BOOL, C_BOOL))
 IR_FOLD(LE(C_U8, C_U8))
+{
+	IR_FOLD_BOOL(op1_insn->val.i8 <= op2_insn->val.i8);
+}
+
 IR_FOLD(LE(C_U16, C_U16))
+{
+	IR_FOLD_BOOL(op1_insn->val.i16 <= op2_insn->val.i16);
+}
+
 IR_FOLD(LE(C_U32, C_U32))
+{
+	IR_FOLD_BOOL(op1_insn->val.i32 <= op2_insn->val.i32);
+}
+
 IR_FOLD(LE(C_U64, C_U64))
+{
+	IR_FOLD_BOOL(op1_insn->val.i64 <= op2_insn->val.i64);
+}
+
 IR_FOLD(LE(C_ADDR, C_ADDR))
 {
-	IR_FOLD_BOOL(op1_insn->val.u64 <= op2_insn->val.u64);
+	IR_FOLD_BOOL(op1_insn->val.addr <= op2_insn->val.addr);
 }

 IR_FOLD(LE(C_CHAR, C_CHAR))
@@ -147,14 +195,32 @@ IR_FOLD(LE(C_FLOAT, C_FLOAT))

 IR_FOLD(GT(C_BOOL, C_BOOL))
 IR_FOLD(GT(C_U8, C_U8))
+{
+	IR_FOLD_BOOL(op1_insn->val.i8 > op2_insn->val.i8);
+}
+
 IR_FOLD(GT(C_U16, C_U16))
+{
+	IR_FOLD_BOOL(op1_insn->val.i16 > op2_insn->val.i16);
+}
+
+
 IR_FOLD(GT(C_U32, C_U32))
+{
+	IR_FOLD_BOOL(op1_insn->val.i32 > op2_insn->val.i32);
+}
+
 IR_FOLD(GT(C_U64, C_U64))
+{
+	IR_FOLD_BOOL(op1_insn->val.i64 > op2_insn->val.i64);
+}
+
 IR_FOLD(GT(C_ADDR, C_ADDR))
 {
-	IR_FOLD_BOOL(op1_insn->val.u64 > op2_insn->val.u64);
+	IR_FOLD_BOOL(op1_insn->val.addr > op2_insn->val.addr);
 }

+
 IR_FOLD(GT(C_CHAR, C_CHAR))
 IR_FOLD(GT(C_I8, C_I8))
 IR_FOLD(GT(C_I16, C_I16))
@@ -524,7 +590,7 @@ IR_FOLD(MUL(C_U8, C_U8))
 IR_FOLD(MUL(C_U16, C_U16))
 {
 	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
-	IR_FOLD_CONST_U(op1_insn->val.u16 * op2_insn->val.u16);
+	IR_FOLD_CONST_U((uint32_t)op1_insn->val.u16 * (uint32_t)op2_insn->val.u16);
 }

 IR_FOLD(MUL(C_U32, C_U32))
@@ -706,39 +772,83 @@ IR_FOLD(MOD(C_ADDR, C_ADDR))
 }

 IR_FOLD(MOD(C_I8, C_I8))
+{
+	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
+	if (op2_insn->val.i64 == 0
+	 || (op2_insn->val.i64 == -1 && op1_insn->val.i8 == INT8_MIN)) {
+		/* division by zero or signed overflow */
+		IR_FOLD_EMIT;
+	}
+	IR_FOLD_CONST_I(op1_insn->val.i8 % op2_insn->val.i8);
+}
+
 IR_FOLD(MOD(C_I16, C_I16))
+{
+	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
+	if (op2_insn->val.i64 == 0
+	 || (op2_insn->val.i64 == -1 && op1_insn->val.i16 == INT16_MIN)) {
+		/* division by zero or signed overflow */
+		IR_FOLD_EMIT;
+	}
+	IR_FOLD_CONST_I(op1_insn->val.i16 % op2_insn->val.i16);
+}
+
 IR_FOLD(MOD(C_I32, C_I32))
+{
+	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
+	if (op2_insn->val.i64 == 0
+	 || (op2_insn->val.i64 == -1 && op1_insn->val.i32 == INT32_MIN)) {
+		/* division by zero or signed overflow */
+		IR_FOLD_EMIT;
+	}
+	IR_FOLD_CONST_I(op1_insn->val.i32 % op2_insn->val.i32);
+}
+
 IR_FOLD(MOD(C_I64, C_I64))
 {
 	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
-	if (op2_insn->val.i64 == 0) {
-		/* division by zero */
+	if (op2_insn->val.i64 == 0
+	 || (op2_insn->val.i64 == -1 && op1_insn->val.i64 == INT64_MIN)) {
+		/* division by zero or signed overflow */
 		IR_FOLD_EMIT;
 	}
 	IR_FOLD_CONST_I(op1_insn->val.i64 % op2_insn->val.i64);
 }

 IR_FOLD(NEG(C_I8))
+IR_FOLD(NEG(C_CHAR))
 {
-	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
 	IR_FOLD_CONST_I((int8_t)(0 - op1_insn->val.u8));
 }

 IR_FOLD(NEG(C_I16))
 {
-	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
 	IR_FOLD_CONST_I((int16_t)(0 -op1_insn->val.u16));
 }

 IR_FOLD(NEG(C_I32))
 {
-	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
 	IR_FOLD_CONST_I((int32_t)(0 - op1_insn->val.u32));
 }

+IR_FOLD(NEG(C_U8))
+{
+	IR_FOLD_CONST_U((uint8_t)(0 - op1_insn->val.u8));
+}
+
+IR_FOLD(NEG(C_U16))
+{
+	IR_FOLD_CONST_U((uint16_t)(0 -op1_insn->val.u16));
+}
+
+IR_FOLD(NEG(C_U32))
+{
+	IR_FOLD_CONST_U((uint32_t)(0 - op1_insn->val.u32));
+}
+
 IR_FOLD(NEG(C_I64))
+IR_FOLD(NEG(C_U64))
 {
-	IR_ASSERT(IR_OPT_TYPE(opt) == op1_insn->type);
 	IR_FOLD_CONST_I(0 - op1_insn->val.u64);
 }

@@ -1440,8 +1550,30 @@ IR_FOLD(BITCAST(C_BOOL))
 IR_FOLD(BITCAST(C_CHAR))
 IR_FOLD(BITCAST(C_ADDR))
 {
-	IR_ASSERT(ir_type_size[IR_OPT_TYPE(opt)] == ir_type_size[op1_insn->type]);
-	switch (IR_OPT_TYPE(opt)) {
+	ir_type dst_type = IR_OPT_TYPE(opt);
+
+	if (IR_IS_TYPE_VECTOR(dst_type)) {
+		uint32_t size = IR_VECTOR_SIZE(dst_type);
+
+		if (size == ir_get_type_size(op1_insn->type)) {
+			ir_ref vec = ir_const_vector(ctx, dst_type);
+			void *dst = ir_long_const_ptr(ctx, vec);
+			op1_insn = &ctx->ir_base[op1];
+			switch (ir_type_size[op1_insn->type]) {
+				case 8: memcpy(dst, &op1_insn->val.u64, 8); break;
+				case 4: memcpy(dst, &op1_insn->val.u32, 4); break;
+				case 2: memcpy(dst, &op1_insn->val.u16, 2); break;
+				case 1: memcpy(dst, &op1_insn->val.u8, 1);  break;
+				default:
+					IR_ASSERT(0);
+			}
+			vec = ir_long_const_commit(ctx, vec);
+			IR_FOLD_COPY(vec);
+		}
+		IR_FOLD_NEXT;
+	}
+	IR_ASSERT(ir_type_size[dst_type] == ir_type_size[op1_insn->type]);
+	switch (dst_type) {
 		default:
 			IR_ASSERT(0);
 		case IR_BOOL:
@@ -1473,6 +1605,45 @@ IR_FOLD(BITCAST(C_ADDR))
 	}
 }

+IR_FOLD(BITCAST(LONG_CONST))
+{
+	ir_type dst_type = IR_OPT_TYPE(opt);
+	ir_type src_type = op1_insn->type;
+	uint32_t size = IR_VECTOR_SIZE(src_type);
+
+	if (IR_IS_TYPE_VECTOR(dst_type)) {
+		if (size == IR_VECTOR_SIZE(dst_type)) {
+			ir_ref vec = ir_const_vector(ctx, dst_type);
+			void *dst = ir_long_const_ptr(ctx, vec);
+			void *src = ir_long_const_ptr(ctx, op1);
+			memcpy(dst, src, size);
+			vec = ir_long_const_commit(ctx, vec);
+			IR_FOLD_COPY(vec);
+		}
+	} else {
+		if (size == ir_type_size[dst_type]) {
+			void *ptr = ir_long_const_ptr(ctx, op1);
+
+			switch (dst_type) {
+				case IR_CHAR:
+				case IR_I8:     IR_FOLD_CONST_I(*(int8_t*)ptr);
+				case IR_I16:    IR_FOLD_CONST_I(*(int16_t*)ptr);
+				case IR_I32:    IR_FOLD_CONST_I(*(int32_t*)ptr);
+				case IR_I64:    IR_FOLD_CONST_I(*(int64_t*)ptr);
+				case IR_U8:     IR_FOLD_CONST_U(*(uint8_t*)ptr);
+				case IR_U16:    IR_FOLD_CONST_U(*(uint16_t*)ptr);
+				case IR_U32:    IR_FOLD_CONST_U(*(uint32_t*)ptr);
+				case IR_U64:    IR_FOLD_CONST_U(*(uint64_t*)ptr);
+				case IR_DOUBLE: IR_FOLD_CONST_D(*(double*)ptr);
+				case IR_FLOAT:  IR_FOLD_CONST_F(*(float*)ptr);
+				default:
+					IR_ASSERT(0);
+			}
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
 IR_FOLD(INT2FP(C_I8))
 IR_FOLD(INT2FP(C_I16))
 IR_FOLD(INT2FP(C_I32))
@@ -1569,6 +1740,238 @@ IR_FOLD(FP2FP(C_DOUBLE))
 	}
 }

+IR_FOLD(SPLAT(C_I8))
+IR_FOLD(SPLAT(C_U8))
+IR_FOLD(SPLAT(C_CHAR))
+{
+	uint8_t v = op1_insn->val.u8;
+	ir_type type = IR_OPT_TYPE(opt);
+	int n = IR_VECTOR_LENGTH(type);
+	ir_ref vec = ir_const_vector(ctx, type);
+	uint8_t *ptr = (uint8_t*)ir_long_const_ptr(ctx, vec);
+
+	for (;n > 0; ptr++, n--) {
+		*ptr = v;
+	}
+	vec = ir_long_const_commit(ctx, vec);
+	IR_FOLD_COPY(vec);
+}
+
+IR_FOLD(SPLAT(C_I16))
+IR_FOLD(SPLAT(C_U16))
+{
+	uint16_t v = op1_insn->val.u16;
+	ir_type type = IR_OPT_TYPE(opt);
+	int n = IR_VECTOR_LENGTH(type);
+	ir_ref vec = ir_const_vector(ctx, type);
+	uint16_t *ptr = (uint16_t*)ir_long_const_ptr(ctx, vec);
+
+	for (;n > 0; ptr++, n--) {
+		*ptr = v;
+	}
+	vec = ir_long_const_commit(ctx, vec);
+	IR_FOLD_COPY(vec);
+}
+
+IR_FOLD(SPLAT(C_I32))
+IR_FOLD(SPLAT(C_U32))
+{
+	uint32_t v = op1_insn->val.u32;
+	ir_type type = IR_OPT_TYPE(opt);
+	int n = IR_VECTOR_LENGTH(type);
+	ir_ref vec = ir_const_vector(ctx, type);
+	uint32_t *ptr = (uint32_t*)ir_long_const_ptr(ctx, vec);
+
+	for (;n > 0; ptr++, n--) {
+		*ptr = v;
+	}
+	vec = ir_long_const_commit(ctx, vec);
+	IR_FOLD_COPY(vec);
+}
+
+IR_FOLD(SPLAT(C_I64))
+IR_FOLD(SPLAT(C_U64))
+{
+	uint64_t v = op1_insn->val.u64;
+	ir_type type = IR_OPT_TYPE(opt);
+	int n = IR_VECTOR_LENGTH(type);
+	ir_ref vec = ir_const_vector(ctx, type);
+	uint64_t *ptr = (uint64_t*)ir_long_const_ptr(ctx, vec);
+
+	for (;n > 0; ptr++, n--) {
+		*ptr = v;
+	}
+	vec = ir_long_const_commit(ctx, vec);
+	IR_FOLD_COPY(vec);
+}
+
+IR_FOLD(SPLAT(C_FLOAT))
+{
+	float v = op1_insn->val.f;
+	ir_type type = IR_OPT_TYPE(opt);
+	int n = IR_VECTOR_LENGTH(type);
+	ir_ref vec = ir_const_vector(ctx, type);
+	float *ptr = (float*)ir_long_const_ptr(ctx, vec);
+
+	for (;n > 0; ptr++, n--) {
+		*ptr = v;
+	}
+	vec = ir_long_const_commit(ctx, vec);
+	IR_FOLD_COPY(vec);
+}
+
+IR_FOLD(SPLAT(C_DOUBLE))
+{
+	double v = op1_insn->val.d;
+	ir_type type = IR_OPT_TYPE(opt);
+	int n = IR_VECTOR_LENGTH(type);
+	ir_ref vec = ir_const_vector(ctx, type);
+	double *ptr = (double*)ir_long_const_ptr(ctx, vec);
+
+	for (;n > 0; ptr++, n--) {
+		*ptr = v;
+	}
+	vec = ir_long_const_commit(ctx, vec);
+	IR_FOLD_COPY(vec);
+}
+
+IR_FOLD(REPLACE(LONG_CONST, _))
+{
+	if (IR_IS_CONST_REF(op3)
+	 && IR_IS_CONST_REF(op2)
+	 && IR_IS_TYPE_INT(op2_insn->type)
+	 && op2_insn->val.i64 >= 0
+	 && op2_insn->val.i64 < IR_VECTOR_LENGTH(op1_insn->type)) {
+		IR_ASSERT(IR_IS_TYPE_VECTOR(op1_insn->type)
+		  && (IR_VECTOR_BASE_TYPE(op1_insn->type) == op3_insn->type
+		   || (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(op1_insn->type))
+		    && IR_IS_TYPE_INT(op3_insn->type)
+		    && ir_type_size[IR_VECTOR_BASE_TYPE(op1_insn->type)] == ir_type_size[op3_insn->type])));
+		uint32_t idx = op2_insn->val.u32;
+
+		if (op3_insn->type == IR_U8 || op3_insn->type == IR_I8 || op3_insn->type == IR_CHAR) {
+			uint8_t v = op3_insn->val.u8;
+			uint8_t *src = (uint8_t*)ir_long_const_ptr(ctx, op1);
+			if (src[idx] == v) {
+				IR_FOLD_COPY(op1);
+			} else {
+				ir_ref vec = ir_const_vector(ctx, IR_OPT_TYPE(opt));
+				uint8_t *src = (uint8_t*)ir_long_const_ptr(ctx, op1);
+				uint8_t *dst = (uint8_t*)ir_long_const_ptr(ctx, vec);
+				memcpy(dst, src, IR_VECTOR_SIZE(IR_OPT_TYPE(opt)));
+				dst[idx] = v;
+				vec = ir_long_const_commit(ctx, vec);
+				IR_FOLD_COPY(vec);
+			}
+		} else if (op3_insn->type == IR_U16 || op3_insn->type == IR_I16) {
+			uint16_t v = op3_insn->val.u16;
+			uint16_t *src = (uint16_t*)ir_long_const_ptr(ctx, op1);
+			if (src[idx] == v) {
+				IR_FOLD_COPY(op1);
+			} else {
+				ir_ref vec = ir_const_vector(ctx, IR_OPT_TYPE(opt));
+				uint16_t *src = (uint16_t*)ir_long_const_ptr(ctx, op1);
+				uint16_t *dst = (uint16_t*)ir_long_const_ptr(ctx, vec);
+				memcpy(dst, src, IR_VECTOR_SIZE(IR_OPT_TYPE(opt)));
+				dst[idx] = v;
+				vec = ir_long_const_commit(ctx, vec);
+				IR_FOLD_COPY(vec);
+			}
+		} else if (op3_insn->type == IR_U32 || op3_insn->type == IR_I32) {
+			uint32_t v = op3_insn->val.u32;
+			uint32_t *src = (uint32_t*)ir_long_const_ptr(ctx, op1);
+			if (src[idx] == v) {
+				IR_FOLD_COPY(op1);
+			} else {
+				ir_ref vec = ir_const_vector(ctx, IR_OPT_TYPE(opt));
+				uint32_t *src = (uint32_t*)ir_long_const_ptr(ctx, op1);
+				uint32_t *dst = (uint32_t*)ir_long_const_ptr(ctx, vec);
+				memcpy(dst, src, IR_VECTOR_SIZE(IR_OPT_TYPE(opt)));
+				dst[idx] = v;
+				vec = ir_long_const_commit(ctx, vec);
+				IR_FOLD_COPY(vec);
+			}
+		} else if (op3_insn->type == IR_U64 || op3_insn->type == IR_I64) {
+			uint64_t v = op3_insn->val.u64;
+			uint64_t *src = (uint64_t*)ir_long_const_ptr(ctx, op1);
+			if (src[idx] == v) {
+				IR_FOLD_COPY(op1);
+			} else {
+				ir_ref vec = ir_const_vector(ctx, IR_OPT_TYPE(opt));
+				uint64_t *src = (uint64_t*)ir_long_const_ptr(ctx, op1);
+				uint64_t *dst = (uint64_t*)ir_long_const_ptr(ctx, vec);
+				memcpy(dst, src, IR_VECTOR_SIZE(IR_OPT_TYPE(opt)));
+				dst[idx] = v;
+				vec = ir_long_const_commit(ctx, vec);
+				IR_FOLD_COPY(vec);
+			}
+		} else if (op3_insn->type == IR_DOUBLE) {
+			double v = op3_insn->val.d;
+			double *src = (double*)ir_long_const_ptr(ctx, op1);
+			if (src[idx] == v) {
+				IR_FOLD_COPY(op1);
+			} else {
+				ir_ref vec = ir_const_vector(ctx, IR_OPT_TYPE(opt));
+				double *src = (double*)ir_long_const_ptr(ctx, op1);
+				double *dst = (double*)ir_long_const_ptr(ctx, vec);
+				memcpy(dst, src, IR_VECTOR_SIZE(IR_OPT_TYPE(opt)));
+				dst[idx] = v;
+				vec = ir_long_const_commit(ctx, vec);
+				IR_FOLD_COPY(vec);
+			}
+		} else if (op3_insn->type == IR_FLOAT) {
+			float v = op3_insn->val.f;
+			float *src = (float*)ir_long_const_ptr(ctx, op1);
+			if (src[idx] == v) {
+				IR_FOLD_COPY(op1);
+			} else {
+				ir_ref vec = ir_const_vector(ctx, IR_OPT_TYPE(opt));
+				float *src = (float*)ir_long_const_ptr(ctx, op1);
+				float *dst = (float*)ir_long_const_ptr(ctx, vec);
+				memcpy(dst, src, IR_VECTOR_SIZE(IR_OPT_TYPE(opt)));
+				dst[idx] = v;
+				vec = ir_long_const_commit(ctx, vec);
+				IR_FOLD_COPY(vec);
+			}
+		}
+	}
+	IR_FOLD_EMIT;
+}
+
+IR_FOLD(EXTRACT(LONG_CONST, _))
+{
+	if (IR_IS_CONST_REF(op2)
+	 && IR_IS_TYPE_INT(op2_insn->type)
+	 && op2_insn->val.i64 >= 0
+	 && op2_insn->val.i64 < IR_VECTOR_LENGTH(op1_insn->type)) {
+		uint32_t idx = op2_insn->val.u32;
+		void *ptr;
+
+		IR_ASSERT(IR_IS_TYPE_VECTOR(op1_insn->type)
+			&& (IR_VECTOR_BASE_TYPE(op1_insn->type) == IR_OPT_TYPE(opt)
+				|| (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(op1_insn->type))
+					&& IR_IS_TYPE_INT(IR_OPT_TYPE(opt))
+					&& ir_type_size[IR_VECTOR_BASE_TYPE(op1_insn->type)] == ir_type_size[IR_OPT_TYPE(opt)])));
+		ptr = (uint8_t*)ir_long_const_ptr(ctx, op1);
+		switch (IR_OPT_TYPE(opt)) {
+			case IR_CHAR:
+			case IR_I8:     IR_FOLD_CONST_I(((int8_t*)ptr)[idx]);
+			case IR_I16:    IR_FOLD_CONST_I(((int16_t*)ptr)[idx]);
+			case IR_I32:    IR_FOLD_CONST_I(((int32_t*)ptr)[idx]);
+			case IR_I64:    IR_FOLD_CONST_I(((int64_t*)ptr)[idx]);
+			case IR_U8:     IR_FOLD_CONST_U(((uint8_t*)ptr)[idx]);
+			case IR_U16:    IR_FOLD_CONST_U(((uint16_t*)ptr)[idx]);
+			case IR_U32:    IR_FOLD_CONST_U(((uint32_t*)ptr)[idx]);
+			case IR_U64:    IR_FOLD_CONST_U(((uint64_t*)ptr)[idx]);
+			case IR_DOUBLE: IR_FOLD_CONST_D(((double*)ptr)[idx]);
+			case IR_FLOAT:  IR_FOLD_CONST_F(((float*)ptr)[idx]);
+			default:
+				IR_ASSERT(0);
+		}
+	}
+	IR_FOLD_EMIT;
+}
+
 // TODO: constant functions (e.g.  sin, cos)

 /* Copy Propagation */
@@ -1665,6 +2068,62 @@ IR_FOLD(NE(_, C_BOOL))
 	}
 }

+IR_FOLD(EQ(BITCAST, C_U8))
+IR_FOLD(EQ(BITCAST, C_U16))
+IR_FOLD(EQ(BITCAST, C_U32))
+IR_FOLD(EQ(BITCAST, C_U64))
+IR_FOLD(EQ(BITCAST, C_I8))
+IR_FOLD(EQ(BITCAST, C_I16))
+IR_FOLD(EQ(BITCAST, C_I32))
+IR_FOLD(EQ(BITCAST, C_I64))
+IR_FOLD(EQ(BITCAST, C_ADDR))
+IR_FOLD(EQ(BITCAST, C_CHAR))
+IR_FOLD(NE(BITCAST, C_U8))
+IR_FOLD(NE(BITCAST, C_U16))
+IR_FOLD(NE(BITCAST, C_U32))
+IR_FOLD(NE(BITCAST, C_U64))
+IR_FOLD(NE(BITCAST, C_I8))
+IR_FOLD(NE(BITCAST, C_I16))
+IR_FOLD(NE(BITCAST, C_I32))
+IR_FOLD(NE(BITCAST, C_I64))
+IR_FOLD(NE(BITCAST, C_ADDR))
+IR_FOLD(NE(BITCAST, C_CHAR))
+{
+	ir_type type = ctx->ir_base[op1_insn->op1].type;
+	if (type == IR_BOOL) {
+		if ((((opt & IR_OPT_OP_MASK) == IR_NE) && (op2_insn->val.u64 == 1))
+		 || (((opt & IR_OPT_OP_MASK) == IR_EQ) && (op2_insn->val.u64 == 0))) {
+			opt = IR_OPT(IR_NOT, IR_BOOL);
+			op1 = op1_insn->op1;
+			op2 = IR_UNUSED;
+			IR_FOLD_RESTART;
+		} else if ((((opt & IR_OPT_OP_MASK) == IR_NE) && (op2_insn->val.u64 == 0))
+				|| (((opt & IR_OPT_OP_MASK) == IR_EQ) && (op2_insn->val.u64 == 1))) {
+			IR_FOLD_COPY(op1_insn->op1);
+		}
+	} else if (IR_IS_TYPE_INT(type)) {
+		op1 = op1_insn->op1;
+		if (IR_IS_TYPE_SIGNED(type)) {
+			switch (ir_type_size[type]) {
+				case 1:  val.i64 = op2_insn->val.i8;  break;
+				case 2:  val.i64 = op2_insn->val.i16; break;
+				case 4:  val.i64 = op2_insn->val.i32; break;
+				default: val.u64 = op2_insn->val.u64; break;
+			 }
+		} else {
+			switch (ir_type_size[type]) {
+				case 1:  val.u64 = op2_insn->val.u8;  break;
+				case 2:  val.u64 = op2_insn->val.u16; break;
+				case 4:  val.u64 = op2_insn->val.u32; break;
+				default: val.u64 = op2_insn->val.u64; break;
+			 }
+		}
+		op2 = ir_const(ctx, val, type);
+		IR_FOLD_RESTART;
+	}
+	IR_FOLD_NEXT;
+}
+
 IR_FOLD(EQ(ZEXT, C_U16))
 IR_FOLD(EQ(ZEXT, C_U32))
 IR_FOLD(EQ(ZEXT, C_U64))
@@ -1755,13 +2214,15 @@ IR_FOLD(GT(SEXT, C_ADDR))
 	} else {
 		ir_type type = ctx->ir_base[op1_insn->op1].type;

-		if (type == IR_BOOL && op2_insn->val.u64 == 0) {
-			if ((opt & IR_OPT_OP_MASK) == IR_EQ) {
+		if (type == IR_BOOL) {
+			if ((((opt & IR_OPT_OP_MASK) == IR_NE) && (op2_insn->val.u64 == 1))
+			 || (((opt & IR_OPT_OP_MASK) == IR_EQ) && (op2_insn->val.u64 == 0))) {
 				opt = IR_OPT(IR_NOT, IR_BOOL);
 				op1 = op1_insn->op1;
 				op2 = IR_UNUSED;
 				IR_FOLD_RESTART;
-			} else if ((opt & IR_OPT_OP_MASK) == IR_NE) {
+			} else if ((((opt & IR_OPT_OP_MASK) == IR_NE) && (op2_insn->val.u64 == 0))
+					|| (((opt & IR_OPT_OP_MASK) == IR_EQ) && (op2_insn->val.u64 == 1))) {
 				IR_FOLD_COPY(op1_insn->op1);
 			}
 		}
@@ -2216,7 +2677,7 @@ IR_FOLD(DIV(NEG, C_I32))
 IR_FOLD(DIV(NEG, C_I64))
 {
 	op1 = op1_insn->op1;
-	val.i64 = -op2_insn->val.i64;
+	val.i64 = -(uint64_t)op2_insn->val.i64;
 	op2 = ir_const(ctx, val, op2_insn->type);
 	IR_FOLD_RESTART;
 }
@@ -2319,6 +2780,12 @@ IR_FOLD(DIV(_, C_U64))
 {
 	if (op2_insn->val.u64 == 1) {
 		IR_FOLD_COPY(op1);
+	} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
+		/* a / C => a >> log2(C) ; C is power of 2 */
+		val.u64 = IR_LOG2(op2_insn->val.u64);;
+		op2 = ir_const(ctx, val, IR_OPT_TYPE(opt));
+		opt = IR_SHR | (opt & IR_OPT_TYPE_MASK);
+		IR_FOLD_RESTART;
 	}
 	IR_FOLD_NEXT;
 }
@@ -2335,6 +2802,15 @@ IR_FOLD(DIV(_, C_I64))
 		/* a / -1 => -a */
 		opt = IR_NEG | (opt & IR_OPT_TYPE_MASK);
 		op2 = IR_UNUSED;
+		op3 = IR_UNUSED;
+		IR_FOLD_RESTART;
+		IR_FOLD_COPY(op1);
+	} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64) && op1_insn->op == IR_ZEXT) {
+		/* a / C => a >> log2(C) ; C is power of 2 */
+		val.u64 = IR_LOG2(op2_insn->val.u64);;
+		op2 = ir_const(ctx, val, IR_OPT_TYPE(opt));
+		op3 = IR_UNUSED;
+		opt = IR_SHR | (opt & IR_OPT_TYPE_MASK);
 		IR_FOLD_RESTART;
 	}
 	IR_FOLD_NEXT;
@@ -2352,6 +2828,14 @@ IR_FOLD(MOD(_, C_I64))
 	if (op2_insn->val.i64 == 1) {
 		/* a % 1 => 0 */
 		IR_FOLD_CONST_U(0);
+	} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)
+	 && (IR_IS_TYPE_UNSIGNED(IR_OPT_TYPE(opt)) || op1_insn->op == IR_ZEXT)) {
+		/* a % C => a & (C - 1) ; C is power of 2 */
+		val.u64 = op2_insn->val.u64 - 1;
+		op2 = ir_const(ctx, val, IR_OPT_TYPE(opt));
+		op3 = IR_UNUSED;
+		opt = IR_AND | (opt & IR_OPT_TYPE_MASK);
+		IR_FOLD_RESTART;
 	}
 	IR_FOLD_NEXT;
 }
@@ -2635,6 +3119,15 @@ IR_FOLD(ROR(_, C_I64))
 	IR_FOLD_NEXT;
 }

+IR_FOLD(SHL(_, SPLAT))
+IR_FOLD(SHR(_, SPLAT))
+IR_FOLD(SAR(_, SPLAT))
+{
+	/* a << SPLAT(b) => a << b */
+	op1 = op1_insn->op1;
+	IR_FOLD_RESTART;
+}
+
 IR_FOLD(SHL(C_U8, _))
 IR_FOLD(SHL(C_U16, _))
 IR_FOLD(SHL(C_U32, _))
@@ -2756,7 +3249,9 @@ IR_FOLD(FP2INT(INT2FP))
 	ir_type dst_type = IR_OPT_TYPE(opt);
 	ir_type src_type = ctx->ir_base[op1_insn->op1].type;

-	if (ir_type_size[src_type] >= ir_type_size[op1_insn->type]) {
+	if (!IR_IS_TYPE_INT(IR_OPT_TYPE(opt))) {
+		IR_FOLD_NEXT;
+	} else if (ir_type_size[src_type] >= ir_type_size[op1_insn->type]) {
 		/* source integer type can not fit into intermediate floating point */
 		IR_FOLD_NEXT;
 	}
@@ -2776,11 +3271,11 @@ IR_FOLD(TRUNC(SEXT))
 	/* (int32_t)(int64_t)i => i */
 	if (src_type == dst_type) {
 		IR_FOLD_COPY(op1_insn->op1);
-	} else if (ir_type_size[src_type] == ir_type_size[dst_type]) {
+	} else if (ir_get_type_size(src_type) == ir_get_type_size(dst_type)) {
 		opt = IR_OPT(IR_BITCAST, dst_type);
 		op1 = op1_insn->op1;
 		IR_FOLD_RESTART;
-	} else if (ir_type_size[src_type] > ir_type_size[dst_type]) {
+	} else if (ir_get_type_size(src_type) > ir_get_type_size(dst_type)) {
 		opt = IR_OPT(IR_TRUNC, dst_type);
 		op1 = op1_insn->op1;
 		IR_FOLD_RESTART;
@@ -2796,7 +3291,8 @@ IR_FOLD(TRUNC(BITCAST))
 IR_FOLD(ZEXT(BITCAST))
 IR_FOLD(SEXT(BITCAST))
 {
-	if (IR_IS_TYPE_INT(ctx->ir_base[op1_insn->op1].type)) {
+	if (IR_IS_TYPE_INT(IR_OPT_TYPE(opt))
+	 && IR_IS_TYPE_INT(ctx->ir_base[op1_insn->op1].type)) {
 		op1 = op1_insn->op1;
 		IR_FOLD_RESTART;
 	}
@@ -2810,11 +3306,10 @@ IR_FOLD(BITCAST(BITCAST))

 	if (src_type == dst_type) {
 		IR_FOLD_COPY(op1_insn->op1);
-	} else if (IR_IS_TYPE_INT(src_type) == IR_IS_TYPE_INT(dst_type)) {
+	} else {
 		op1 = op1_insn->op1;
 		IR_FOLD_RESTART;
 	}
-	IR_FOLD_NEXT;
 }

 IR_FOLD(TRUNC(TRUNC))
@@ -2834,7 +3329,8 @@ IR_FOLD(SEXT(ZEXT))

 IR_FOLD(SEXT(AND))
 {
-	if (IR_IS_CONST_REF(op1_insn->op2)
+	if (IR_IS_TYPE_INT(IR_OPT_TYPE(opt))
+	 && IR_IS_CONST_REF(op1_insn->op2)
 	 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)
 	 && !(ctx->ir_base[op1_insn->op2].val.u64
 			& (1ULL << ((ir_type_size[op1_insn->type] * 8) - 1)))) {
@@ -2847,7 +3343,8 @@ IR_FOLD(SEXT(AND))

 IR_FOLD(SEXT(SHR))
 {
-	if (IR_IS_CONST_REF(op1_insn->op2)
+	if (IR_IS_TYPE_INT(IR_OPT_TYPE(opt))
+	 && IR_IS_CONST_REF(op1_insn->op2)
 	 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)
 	 && ctx->ir_base[op1_insn->op2].val.u64 != 0) {
 		opt = IR_OPT(IR_ZEXT, IR_OPT_TYPE(opt));
@@ -2858,7 +3355,8 @@ IR_FOLD(SEXT(SHR))

 IR_FOLD(TRUNC(AND))
 {
-	if (IR_IS_CONST_REF(op1_insn->op2)) {
+	if (IR_IS_TYPE_INT(IR_OPT_TYPE(opt))
+	 && IR_IS_CONST_REF(op1_insn->op2)) {
 		size_t size = ir_type_size[IR_OPT_TYPE(opt)];
 		uint64_t mask = ctx->ir_base[op1_insn->op2].val.u64;

@@ -2920,11 +3418,13 @@ IR_FOLD(AND(SEXT, C_ADDR))
 	}
 	IR_FOLD_NEXT;
 }
+
 IR_FOLD(AND(SHR, C_I8))
 IR_FOLD(AND(SHR, C_U8))
 {
 	if (IR_IS_CONST_REF(op1_insn->op2)) {
-		if (((uint8_t)-1) >> ctx->ir_base[op1_insn->op2].val.u8 == op2_insn->val.u8) {
+		if (((uint8_t)-1) >> (ctx->ir_base[op1_insn->op2].val.u8 & 0x7) == op2_insn->val.u8) {
+			/* (x >> N) & (0xff >> N) => x >> N */
 			IR_FOLD_COPY(op1);
 		}
 	}
@@ -2935,7 +3435,7 @@ IR_FOLD(AND(SHR, C_I16))
 IR_FOLD(AND(SHR, C_U16))
 {
 	if (IR_IS_CONST_REF(op1_insn->op2)) {
-		if (((uint16_t)-1) >> ctx->ir_base[op1_insn->op2].val.u16 == op2_insn->val.u16) {
+		if (((uint16_t)-1) >> (ctx->ir_base[op1_insn->op2].val.u16 & 0xf) == op2_insn->val.u16) {
 			IR_FOLD_COPY(op1);
 		}
 	}
@@ -2946,7 +3446,7 @@ IR_FOLD(AND(SHR, C_I32))
 IR_FOLD(AND(SHR, C_U32))
 {
 	if (IR_IS_CONST_REF(op1_insn->op2)) {
-		if (((uint32_t)-1) >> ctx->ir_base[op1_insn->op2].val.u32 == op2_insn->val.u32) {
+		if (((uint32_t)-1) >> (ctx->ir_base[op1_insn->op2].val.u32 & 0x1f) == op2_insn->val.u32) {
 			IR_FOLD_COPY(op1);
 		}
 	}
@@ -2957,13 +3457,156 @@ IR_FOLD(AND(SHR, C_I64))
 IR_FOLD(AND(SHR, C_U64))
 {
 	if (IR_IS_CONST_REF(op1_insn->op2)) {
-		if (((uint64_t)-1) >> ctx->ir_base[op1_insn->op2].val.u64 == op2_insn->val.u64) {
+		if (((uint64_t)-1) >> (ctx->ir_base[op1_insn->op2].val.u64 & 0x3f) == op2_insn->val.u64) {
+			IR_FOLD_COPY(op1);
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(AND(SHL, C_I8))
+IR_FOLD(AND(SHL, C_U8))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint8_t)-1) << (ctx->ir_base[op1_insn->op2].val.u8 & 0x7) == op2_insn->val.u8) {
+			/* (x << N) & (0xff << N) => x << N */
+			IR_FOLD_COPY(op1);
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(AND(SHL, C_I16))
+IR_FOLD(AND(SHL, C_U16))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint16_t)-1) << (ctx->ir_base[op1_insn->op2].val.u16 & 0xf) == op2_insn->val.u16) {
+			IR_FOLD_COPY(op1);
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(AND(SHL, C_I32))
+IR_FOLD(AND(SHL, C_U32))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint32_t)-1) << (ctx->ir_base[op1_insn->op2].val.u32 & 0x1f) == op2_insn->val.u32) {
+			IR_FOLD_COPY(op1);
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(AND(SHL, C_I64))
+IR_FOLD(AND(SHL, C_U64))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint64_t)-1) << (ctx->ir_base[op1_insn->op2].val.u64 & 0x3f) == op2_insn->val.u64) {
 			IR_FOLD_COPY(op1);
 		}
 	}
 	IR_FOLD_NEXT;
 }

+IR_FOLD(SHL(AND, C_I8))
+IR_FOLD(SHL(AND, C_U8))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint8_t)-1) >> (op2_insn->val.u8 & 0x7) == ctx->ir_base[op1_insn->op2].val.u8) {
+			/* (x & (0xff >> N) << N) => x << N */
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHL(AND, C_I16))
+IR_FOLD(SHL(AND, C_U16))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint16_t)-1) >> (op2_insn->val.u16 & 0xf) == ctx->ir_base[op1_insn->op2].val.u16) {
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHL(AND, C_I32))
+IR_FOLD(SHL(AND, C_U32))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint32_t)-1) >> (op2_insn->val.u32 & 0x1f) == ctx->ir_base[op1_insn->op2].val.u32) {
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHL(AND, C_I64))
+IR_FOLD(SHL(AND, C_U64))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint64_t)-1) >> (op2_insn->val.u64 & 0x3f) == ctx->ir_base[op1_insn->op2].val.u64) {
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHR(AND, C_I8))
+IR_FOLD(SHR(AND, C_U8))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint8_t)-1) << (op2_insn->val.u8 & 0x7) == ctx->ir_base[op1_insn->op2].val.u8) {
+			/* (x & (0xff << N) >> N) => x >> N */
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHR(AND, C_I16))
+IR_FOLD(SHR(AND, C_U16))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint16_t)-1) << (op2_insn->val.u16 & 0xf) == ctx->ir_base[op1_insn->op2].val.u16) {
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHR(AND, C_I32))
+IR_FOLD(SHR(AND, C_U32))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint32_t)-1) << (op2_insn->val.u32 & 0x1f) == ctx->ir_base[op1_insn->op2].val.u32) {
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
+IR_FOLD(SHR(AND, C_I64))
+IR_FOLD(SHR(AND, C_U64))
+{
+	if (IR_IS_CONST_REF(op1_insn->op2)) {
+		if (((uint64_t)-1) << (op2_insn->val.u64 & 0x3f) == ctx->ir_base[op1_insn->op2].val.u64) {
+			op1 = op1_insn->op1;
+			IR_FOLD_RESTART;
+		}
+	}
+	IR_FOLD_NEXT;
+}
+
 IR_FOLD(EQ(FP2FP, C_DOUBLE))
 IR_FOLD(NE(FP2FP, C_DOUBLE))
 IR_FOLD(LT(FP2FP, C_DOUBLE))
@@ -3354,7 +3997,7 @@ IR_FOLD(OR(SHR, SHL))
 IR_FOLD(ADD(SHL, SHR))
 IR_FOLD(ADD(SHR, SHL))
 {
-	if (op1_insn->op1 == op2_insn->op1) {
+	if (IR_IS_TYPE_INT(IR_OPT_TYPE(opt)) && op1_insn->op1 == op2_insn->op1) {
 		if (IR_IS_CONST_REF(op1_insn->op2) && IR_IS_CONST_REF(op2_insn->op2)) {
 			if (ctx->ir_base[op1_insn->op2].val.u64 + ctx->ir_base[op2_insn->op2].val.u64 ==
 					ir_type_size[IR_OPT_TYPE(opt)] * 8) {
@@ -3500,8 +4143,10 @@ IR_FOLD(ULE(_, _))
 IR_FOLD(UGT(_, _))
 {
 	if (op1 == op2) {
-		/* a >= a => true (two low bits are differ) */
-		IR_FOLD_BOOL((opt ^ (opt >> 1)) & 1);
+		if (IR_IS_TYPE_SCALAR(IR_OPT_TYPE(opt))) {
+			/* a >= a => true (two low bits are differ) */
+			IR_FOLD_BOOL((opt ^ (opt >> 1)) & 1);
+		}
 	} else if (op1 < op2) {  /* move lower ref to op2 */
 		SWAP_REFS(op1, op2);
 		opt ^= 3; /* [U]LT <-> [U]GT, [U]LE <-> [U]GE */
diff --git a/ext/opcache/jit/ir/ir_gcm.c b/ext/opcache/jit/ir/ir_gcm.c
index b194eeb8177..60422f12f6b 100644
--- a/ext/opcache/jit/ir/ir_gcm.c
+++ b/ext/opcache/jit/ir/ir_gcm.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (GCM - Global Code Motion and Scheduler)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * The GCM algorithm is based on Cliff Click's publication
@@ -216,6 +216,10 @@ static bool ir_split_partially_dead_node(ir_ctx *ctx, ir_ref ref, uint32_t b)
 	n = use_list->count;
 	for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
 		use = *p;
+		i = ctx->cfg_map[use];
+		if (!i) {
+			continue;
+		}
 		insn = &ctx->ir_base[use];
 		if (insn->op == IR_PHI) {
 			ir_ref *p = insn->ops + 2; /* PHI data inputs */
@@ -233,10 +237,6 @@ static bool ir_split_partially_dead_node(ir_ctx *ctx, ir_ref ref, uint32_t b)
 				}
 			}
 		} else {
-			i = ctx->cfg_map[use];
-			if (!i) {
-				continue;
-			}
 			IR_ASSERT(i > 0 && i <= ctx->cfg_blocks_count);
 			if (!ir_sparse_set_in(&data->totally_useful, i)) {
 				if (i == b) return 0; /* node is totally-useful in the scheduled block */
@@ -595,7 +595,7 @@ static void ir_gcm_schedule_late(ir_ctx *ctx, ir_ref ref, uint32_t b)
 			ir_use_list *use_list = &ctx->use_lists[ref];
 			ir_ref n, *p, use;

-			for (n = use_list->count, p = &ctx->use_edges[use_list->refs]; n < 0; p++, n--) {
+			for (n = use_list->count, p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
 				use = *p;
 				if (ctx->ir_base[use].op == IR_OVERFLOW) {
 					ctx->cfg_map[use] = b;
@@ -821,10 +821,18 @@ static void ir_xlat_binding(ir_ctx *ctx, ir_ref *_xlat)
 	binding->count = n2;
 }

-IR_ALWAYS_INLINE ir_ref ir_count_constant(ir_ref *_xlat, ir_ref ref)
+IR_ALWAYS_INLINE ir_ref ir_count_constant(const ir_ctx *ctx, ir_ref *_xlat, ir_ref ref)
 {
 	if (!_xlat[ref]) {
 		_xlat[ref] = ref; /* this is only a "used constant" marker */
+		if (ctx->ir_base[ref].op == IR_LONG_CONST) {
+				ir_ref i, n = IR_ALIGNED_SIZE(ctx->ir_base[ref].long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+
+			for (i = 1; i <= n; i++) {
+				_xlat[ref + i] = ref + i; /* this is only a "used constant" marker */
+			}
+			return n + 1;
+		}
 		return 1;
 	}
 	return 0;
@@ -851,7 +859,7 @@ IR_ALWAYS_INLINE bool ir_is_good_bb_order(ir_ctx *ctx, uint32_t b, ir_block *bb,
 				} else if ((bb->flags & IR_BB_LOOP_HEADER)
 				  && (input_b == b || ctx->cfg_blocks[input_b].loop_header == b)) {
 					/* back-edge of reducible loop */
-				} else if ((bb->flags & IR_BB_IRREDUCIBLE_LOOP)
+				} else if (UNEXPECTED(bb->flags & IR_BB_IRREDUCIBLE_LOOP)
 				  && (ctx->cfg_blocks[input_b].loop_header == bb->loop_header)) {
 					/* closing edge of irreducible loop */
 				} else {
@@ -863,6 +871,37 @@ IR_ALWAYS_INLINE bool ir_is_good_bb_order(ir_ctx *ctx, uint32_t b, ir_block *bb,
 	}
 }

+static bool ir_belongs_to_loop(ir_ctx *ctx, uint32_t loop_header, uint32_t b)
+{
+	uint32_t loop_depth = ctx->cfg_blocks[loop_header].loop_depth;
+	ir_block *bb = &ctx->cfg_blocks[b];
+
+	if (bb->loop_depth < loop_depth) {
+		return 0;
+	} else if (bb->loop_depth == loop_depth) {
+		if (bb->flags & IR_BB_LOOP_HEADER) {
+			return b == loop_header;
+		} else {
+			return bb->loop_header == loop_header;
+		}
+	} else {
+		while (bb->loop_depth > loop_depth) {
+			b = bb->loop_header;
+			bb = &ctx->cfg_blocks[b];
+		}
+		return bb->loop_header == loop_header;
+	}
+}
+
+static bool ir_is_irreducable_loop_side_entry(ir_ctx *ctx, ir_block *entry, uint32_t from)
+{
+	if (entry->flags & IR_BB_LOOP_HEADER) {
+		return 0;
+	} else {
+		return !ir_belongs_to_loop(ctx, entry->loop_header, from);
+	}
+}
+
 static IR_NEVER_INLINE void ir_fix_bb_order(ir_ctx *ctx, ir_ref *_prev, ir_ref *_next)
 {
 	uint32_t b, succ, count, *q, *xlat;
@@ -896,10 +935,8 @@ static IR_NEVER_INLINE void ir_fix_bb_order(ir_ctx *ctx, ir_ref *_prev, ir_ref *
 			succ = ctx->cfg_edges[bb->successors];
 			if (ir_bitset_in(worklist.visited, succ)) {
 				/* already processed */
-			} else if ((ctx->cfg_blocks[succ].flags & IR_BB_IRREDUCIBLE_LOOP)
-					&& ((ctx->cfg_blocks[b].flags & IR_BB_LOOP_HEADER) ?
-						(ctx->cfg_blocks[succ].loop_header != b) :
-						(ctx->cfg_blocks[succ].loop_header != ctx->cfg_blocks[b].loop_header))) {
+			} else if (UNEXPECTED(ctx->cfg_blocks[succ].flags & IR_BB_IRREDUCIBLE_ENTRY)
+					&& ir_is_irreducable_loop_side_entry(ctx, &ctx->cfg_blocks[succ], b)) {
 				/* "side" entry of irreducible loop (ignore) */
 			} else if (ir_worklist_push(&worklist, succ)) {
 				goto next;
@@ -914,10 +951,8 @@ static IR_NEVER_INLINE void ir_fix_bb_order(ir_ctx *ctx, ir_ref *_prev, ir_ref *
 				succ = *q;
 				if (ir_bitset_in(worklist.visited, succ)) {
 					/* already processed */
-				} else if ((ctx->cfg_blocks[succ].flags & IR_BB_IRREDUCIBLE_LOOP)
-						&& ((ctx->cfg_blocks[b].flags & IR_BB_LOOP_HEADER) ?
-							(ctx->cfg_blocks[succ].loop_header != b) :
-							(ctx->cfg_blocks[succ].loop_header != ctx->cfg_blocks[b].loop_header))) {
+				} else if (UNEXPECTED(ctx->cfg_blocks[succ].flags & IR_BB_IRREDUCIBLE_ENTRY)
+						&& ir_is_irreducable_loop_side_entry(ctx, &ctx->cfg_blocks[succ], b)) {
 					/* "side" entry of irreducible loop (ignore) */
 				} else if (!best) {
 					best = succ;
@@ -1063,7 +1098,7 @@ static void ir_schedule_topsort(const ir_ctx *ctx, uint32_t b, const ir_block *b
 						goto restart;
 					}
 				} else if (input < IR_TRUE) {
-					*consts_count += ir_count_constant(_xlat, input);
+					*consts_count += ir_count_constant(ctx, _xlat, input);
 				}
 			}
 		}
@@ -1160,16 +1195,16 @@ int ir_schedule(ir_ctx *ctx)
 		insn = &ctx->ir_base[i];
 		if (insn->op == IR_BEGIN) {
 			if (insn->op2) {
-				consts_count += ir_count_constant(_xlat, insn->op2);
+				consts_count += ir_count_constant(ctx, _xlat, insn->op2);
 			}
 		} else if (insn->op == IR_CASE_VAL) {
 			IR_ASSERT(insn->op2 < IR_TRUE);
-			consts_count += ir_count_constant(_xlat, insn->op2);
+			consts_count += ir_count_constant(ctx, _xlat, insn->op2);
 		} else if (insn->op == IR_CASE_RANGE) {
 			IR_ASSERT(insn->op2 < IR_TRUE);
-			consts_count += ir_count_constant(_xlat, insn->op2);
+			consts_count += ir_count_constant(ctx, _xlat, insn->op2);
 			IR_ASSERT(insn->op3 < IR_TRUE);
-			consts_count += ir_count_constant(_xlat, insn->op3);
+			consts_count += ir_count_constant(ctx, _xlat, insn->op3);
 		}
 		n = insn->inputs_count;
 		insns_count += ir_insn_inputs_to_len(n);
@@ -1196,7 +1231,7 @@ int ir_schedule(ir_ctx *ctx)
 				for (j = n, p = insn->ops + 2; j > 0; p++, j--) {
 					input = *p;
 					if (input < IR_TRUE) {
-						consts_count += ir_count_constant(_xlat, input);
+						consts_count += ir_count_constant(ctx, _xlat, input);
 					}
 				}
 				i = _next[i];
@@ -1237,7 +1272,7 @@ int ir_schedule(ir_ctx *ctx)
 								for (j = n, q = use_insn->ops + 2; j > 0; q++, j--) {
 									ir_ref input = *q;
 									if (input < IR_TRUE) {
-										consts_count += ir_count_constant(_xlat, input);
+										consts_count += ir_count_constant(ctx, _xlat, input);
 									}
 								}
 							} else {
@@ -1268,7 +1303,7 @@ int ir_schedule(ir_ctx *ctx)
 		insns_count++;
 		if (IR_INPUT_EDGES_COUNT(ir_op_flags[insn->op]) == 2) {
 			if (insn->op2 < IR_TRUE) {
-				consts_count += ir_count_constant(_xlat, insn->op2);
+				consts_count += ir_count_constant(ctx, _xlat, insn->op2);
 			}
 		}
 	}
@@ -1322,10 +1357,23 @@ int ir_schedule(ir_ctx *ctx)
 		while (i < IR_TRUE) {
 			if (_xlat[i]) {
 				*dst = *src;
-				dst->prev_const = 0;
 				_xlat[i] = j;
-				dst++;
-				j++;
+				if (dst->op == IR_LONG_CONST) {
+					uintptr_t n;
+
+					memset(dst + 1, 0, dst->long_const_size);
+					memcpy(dst + 1, src + 1, dst->long_const_size);
+					n = IR_ALIGNED_SIZE(dst->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+					dst += n + 1;
+					src += n + 1;
+					i += n + 1;
+					j += n + 1;
+					continue;
+				} else {
+					dst->prev_const = 0;
+					dst++;
+					j++;
+				}
 			}
 			src++;
 			i++;
diff --git a/ext/opcache/jit/ir/ir_gdb.c b/ext/opcache/jit/ir/ir_gdb.c
index 41141bd2871..5f1bde3fff0 100644
--- a/ext/opcache/jit/ir/ir_gdb.c
+++ b/ext/opcache/jit/ir/ir_gdb.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (GDB interface)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * Based on Mike Pall's implementation of GDB interface for LuaJIT.
@@ -198,9 +198,12 @@ static uint32_t ir_gdbjit_strz(ir_gdbjit_ctx *ctx, const char *str)
 {
 	uint8_t *p = ctx->p;
 	uint32_t ofs = (uint32_t)(p - ctx->startp);
-	do {
-		*p++ = (uint8_t)*str;
-	} while (*str++);
+	while (*str) {
+		IR_ASSERT(p < ctx->obj.space + sizeof(ctx->obj.space));
+		*p++ = (uint8_t)*str++;
+	}
+	IR_ASSERT(p < ctx->obj.space + sizeof(ctx->obj.space));
+	*p++ = '\0';
 	ctx->p = p;
 	return ofs;
 }
@@ -209,6 +212,7 @@ static uint32_t ir_gdbjit_strz(ir_gdbjit_ctx *ctx, const char *str)
 static void ir_gdbjit_uleb128(ir_gdbjit_ctx *ctx, uint32_t v)
 {
 	uint8_t *p = ctx->p;
+	IR_ASSERT(p + 5 <= ctx->obj.space + sizeof(ctx->obj.space));
 	for (; v >= 0x80; v >>= 7)
 		*p++ = (uint8_t)((v & 0x7f) | 0x80);
 	*p++ = (uint8_t)v;
@@ -219,6 +223,7 @@ static void ir_gdbjit_uleb128(ir_gdbjit_ctx *ctx, uint32_t v)
 static void ir_gdbjit_sleb128(ir_gdbjit_ctx *ctx, int32_t v)
 {
 	uint8_t *p = ctx->p;
+	IR_ASSERT(p + 5 <= ctx->obj.space + sizeof(ctx->obj.space));
 	for (; (uint32_t)(v+0x40) >= 0x80; v >>= 7)
 		*p++ = (uint8_t)((v & 0x7f) | 0x80);
 	*p++ = (uint8_t)(v & 0x7f);
@@ -229,6 +234,7 @@ static void ir_gdbjit_secthdr(ir_gdbjit_ctx *ctx)
 {
 	ir_elf_sectheader *sect;

+	IR_ASSERT(ctx->p < ctx->obj.space + sizeof(ctx->obj.space));
 	*ctx->p++ = '\0';

 #define SECTDEF(id, tp, al)                       \
@@ -267,6 +273,7 @@ static void ir_gdbjit_symtab(ir_gdbjit_ctx *ctx)
 {
 	ir_elf_symbol *sym;

+	IR_ASSERT(ctx->p < ctx->obj.space + sizeof(ctx->obj.space));
 	*ctx->p++ = '\0';

 	sym = &ctx->obj.sym[GDBJIT_SYM_FILE];
@@ -459,6 +466,7 @@ static void ir_gdbjit_initsect(ir_gdbjit_ctx *ctx, int sect)
 static void ir_gdbjit_initsect_done(ir_gdbjit_ctx *ctx, int sect)
 {
 	ctx->obj.sect[sect].size = (uintptr_t)(ctx->p - ctx->startp);
+	IR_ASSERT(ctx->p <= ctx->obj.space + sizeof(ctx->obj.space));
 }

 static void ir_gdbjit_buildobj(ir_gdbjit_ctx *ctx, uint32_t sp_offset, uint32_t sp_adjustment)
diff --git a/ext/opcache/jit/ir/ir_patch.c b/ext/opcache/jit/ir/ir_patch.c
index 39e08eb46a5..849cf779dc5 100644
--- a/ext/opcache/jit/ir/ir_patch.c
+++ b/ext/opcache/jit/ir/ir_patch.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Native code patcher)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * Based on Mike Pall's implementation for LuaJIT.
diff --git a/ext/opcache/jit/ir/ir_perf.c b/ext/opcache/jit/ir/ir_perf.c
index c0561ff86ac..52be820088a 100644
--- a/ext/opcache/jit/ir/ir_perf.c
+++ b/ext/opcache/jit/ir/ir_perf.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Linux perf interface)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * 1) Profile using perf-<pid>.map
diff --git a/ext/opcache/jit/ir/ir_php.h b/ext/opcache/jit/ir/ir_php.h
index 370611f1ac3..4f19fb2a7ed 100644
--- a/ext/opcache/jit/ir/ir_php.h
+++ b/ext/opcache/jit/ir/ir_php.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (IR/PHP integration)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

diff --git a/ext/opcache/jit/ir/ir_private.h b/ext/opcache/jit/ir/ir_private.h
index 6d8f31a8b7e..4179eb21ee5 100644
--- a/ext/opcache/jit/ir/ir_private.h
+++ b/ext/opcache/jit/ir/ir_private.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (Common data structures and non public definitions)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -254,23 +254,30 @@ IR_ALWAYS_INLINE void ir_arena_free(ir_arena *arena)
 	} while (arena);
 }

-IR_ALWAYS_INLINE void* ir_arena_alloc(ir_arena **arena_ptr, size_t size)
+IR_ALWAYS_INLINE void* ir_arena_alloc_aligned(ir_arena **arena_ptr, size_t size, size_t align)
 {
 	ir_arena *arena = *arena_ptr;
-	char *ptr = (char*)IR_ALIGNED_SIZE((uintptr_t)arena->ptr, 8);
+	char *ptr;

+	if (align < 8) {
+		align = 8;
+	}
+	ptr = (char*)IR_ALIGNED_SIZE((uintptr_t)arena->ptr, align);
 	if (EXPECTED((ptrdiff_t)size <= (ptrdiff_t)(arena->end - ptr))) {
 		arena->ptr = ptr + size;
 	} else {
+		size_t hdr_size = IR_ALIGNED_SIZE(sizeof(ir_arena), align);
 		size_t arena_size =
-			UNEXPECTED((size + IR_ALIGNED_SIZE(sizeof(ir_arena), 8)) > (size_t)(arena->end - (char*) arena)) ?
-				(size + IR_ALIGNED_SIZE(sizeof(ir_arena), 8)) :
-				(size_t)(arena->end - (char*) arena);
-		ir_arena *new_arena = (ir_arena*)ir_mem_malloc(arena_size);
+			UNEXPECTED((size + hdr_size) > (size_t)(arena->end - (char*) arena)) ?
+				(size + hdr_size) :
+				(size_t)(arena->end - (char*)arena);
+		ir_arena *new_arena;

+		if (align > 16) arena_size += IR_ALIGNED_SIZE(align, 16);
+		new_arena = (ir_arena*)ir_mem_malloc(arena_size);
 		if (UNEXPECTED(!new_arena)) return NULL;
-		ptr = (char*) new_arena + IR_ALIGNED_SIZE(sizeof(ir_arena), 8);
-		new_arena->ptr = (char*) new_arena + IR_ALIGNED_SIZE(sizeof(ir_arena), 8) + size;
+		ptr = (char*)IR_ALIGNED_SIZE((uintptr_t)new_arena + sizeof(ir_arena), align);
+		new_arena->ptr = (char*) ptr + size;
 		new_arena->end = (char*) new_arena + arena_size;
 		new_arena->prev = arena;
 		*arena_ptr = new_arena;
@@ -279,6 +286,11 @@ IR_ALWAYS_INLINE void* ir_arena_alloc(ir_arena **arena_ptr, size_t size)
 	return (void*) ptr;
 }

+IR_ALWAYS_INLINE void* ir_arena_alloc(ir_arena **arena_ptr, size_t size)
+{
+	return ir_arena_alloc_aligned(arena_ptr, size, 8);
+}
+
 IR_ALWAYS_INLINE void* ir_arena_checkpoint(ir_arena *arena)
 {
 	return arena->ptr;
@@ -878,11 +890,34 @@ void ir_addrtab_free(ir_hashtab *tab);
 ir_ref ir_addrtab_find(const ir_hashtab *tab, uint64_t key);
 void ir_addrtab_set(ir_hashtab *tab, uint64_t key, ir_ref val);

-/*** IR OP info ***/
+/*** IR Type info ***/
 extern const uint8_t ir_type_flags[IR_LAST_TYPE];
 extern const char *ir_type_name[IR_LAST_TYPE];
 extern const char *ir_type_cname[IR_LAST_TYPE];
 extern const uint8_t ir_type_size[IR_LAST_TYPE];
+
+#define IR_VECTOR_SIZE(t)                 (ir_type_size[IR_VECTOR_BASE_TYPE(t)] * IR_VECTOR_LENGTH(t))
+#define IR_MAKE_VECTOR_TYPE(base, length) ir_make_vector_type(base, length)
+
+IR_ALWAYS_INLINE ir_type ir_make_vector_type(ir_type base, uint8_t length)
+{
+	IR_ASSERT(IR_IS_TYPE_SCALAR(base) && length > 0 && length <= 64 && (length & (length - 1)) == 0);
+
+	if (base == IR_CHAR) {
+		base = IR_I8;
+	}
+	return base | ((ir_ntz(length) + 1) << 4);
+}
+
+IR_ALWAYS_INLINE uint32_t ir_get_type_size(ir_type type)
+{
+#if IR_SIMD
+	if (IR_IS_TYPE_VECTOR(type)) return IR_VECTOR_SIZE(type);
+#endif
+	return ir_type_size[type];
+}
+
+/*** IR OP info ***/
 extern const uint32_t ir_op_flags[IR_LAST_OP];
 extern const char *ir_op_name[IR_LAST_OP];

@@ -946,17 +981,17 @@ IR_ALWAYS_INLINE bool ir_ref_is_true(const ir_ctx *ctx, ir_ref ref)
 #define IR_OP_FLAG_MEM_ALLOC      ((1<<6)|(1<<7))
 #define IR_OP_FLAG_MEM_MASK       ((1<<6)|(1<<7))

-#define IR_OPND_UNUSED            0x0
-#define IR_OPND_DATA              0x1
-#define IR_OPND_CONTROL           0x2
-#define IR_OPND_LABEL_REF         0x3
-#define IR_OPND_CONTROL_DEP       0x4
-#define IR_OPND_CONTROL_REF       0x5
-#define IR_OPND_CONTROL_GUARD     0x6
-#define IR_OPND_STR               0x7
-#define IR_OPND_NUM               0x8
-#define IR_OPND_PROB              0x9
-#define IR_OPND_PROTO             0xa
+#define IR_OPND_UNUSED            0x0U
+#define IR_OPND_DATA              0x1U
+#define IR_OPND_CONTROL           0x2U
+#define IR_OPND_LABEL_REF         0x3U
+#define IR_OPND_CONTROL_DEP       0x4U
+#define IR_OPND_CONTROL_REF       0x5U
+#define IR_OPND_CONTROL_GUARD     0x6U
+#define IR_OPND_STR               0x7U
+#define IR_OPND_NUM               0x8U
+#define IR_OPND_PROB              0x9U
+#define IR_OPND_PROTO             0xaU

 #define IR_OP_FLAGS(op_flags, op1_flags, op2_flags, op3_flags) \
 	((op_flags) | ((op1_flags) << 20) | ((op2_flags) << 24) | ((op3_flags) << 28))
@@ -1020,7 +1055,9 @@ IR_ALWAYS_INLINE uint32_t ir_insn_len(const ir_insn *insn)
 #define IR_16B_FRAME_ALIGNMENT (1<<11)
 #define IR_HAS_BLOCK_ADDR      (1<<12)
 #define IR_PREALLOCATED_STACK  (1<<13)
-
+#define IR_RECURSIVE_TAILCALL  (1<<14)
+#define IR_HAS_MEMCPY          (1<<15)
+#define IR_HAS_LONG_CONSTANTS  (1<<16)

 /* Temporary: MEM2SSA -> SCCP */
 #define IR_MEM2SSA_VARS        (1<<25)
@@ -1089,6 +1126,12 @@ IR_ALWAYS_INLINE ir_ref ir_next_control(const ir_ctx *ctx, ir_ref ref)
 		_ref2 = _tmp; \
 	} while (0)

+#define SWAP_REGS(_reg1, _reg2) do { \
+		ir_reg _tmp = _reg1; \
+		_reg1 = _reg2; \
+		_reg2 = _tmp; \
+	} while (0)
+
 #define SWAP_INSNS(_insn1, _insn2) do { \
 		ir_insn *_tmp = _insn1; \
 		_insn1 = _insn2; \
@@ -1101,7 +1144,6 @@ void ir_update_op(ir_ctx *ctx, ir_ref ref, uint32_t idx, ir_ref new_val);
 /*** Iterative Optimization ***/
 void ir_iter_add_uses(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist);
 void ir_iter_replace(ir_ctx *ctx, ir_ref ref, ir_ref new_ref, ir_bitqueue *worklist);
-void ir_iter_update_op(ir_ctx *ctx, ir_ref ref, uint32_t idx, ir_ref new_val, ir_bitqueue *worklist);
 void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist);
 void ir_iter_cleanup(ir_ctx *ctx);

@@ -1115,25 +1157,29 @@ void ir_iter_cleanup(ir_ctx *ctx);
 #define IR_IS_BB_END(op) \
 	((ir_op_flags[op] & IR_OP_FLAG_BB_END) != 0)

-#define IR_BB_UNREACHABLE      (1<<0)
-#define IR_BB_START            (1<<1)
-#define IR_BB_ENTRY            (1<<2)
-#define IR_BB_LOOP_HEADER      (1<<3)
-#define IR_BB_IRREDUCIBLE_LOOP (1<<4)
-#define IR_BB_DESSA_MOVES      (1<<5) /* translation out of SSA requires MOVEs */
-#define IR_BB_EMPTY            (1<<6)
-#define IR_BB_PREV_EMPTY_ENTRY (1<<7)
-#define IR_BB_OSR_ENTRY_LOADS  (1<<8) /* OSR Entry-point with register LOADs   */
-#define IR_BB_LOOP_WITH_ENTRY  (1<<9) /* set together with LOOP_HEADER if there is an ENTRY in the loop */
+#define IR_BB_UNREACHABLE       (1<<0)
+#define IR_BB_START             (1<<1)
+#define IR_BB_ENTRY             (1<<2)
+#define IR_BB_LOOP_HEADER       (1<<3)
+#define IR_BB_IRREDUCIBLE_LOOP  (1<<4)
+#define IR_BB_IRREDUCIBLE_ENTRY (1<<5)
+#define IR_BB_DESSA_MOVES       (1<<6) /* translation out of SSA requires MOVEs */
+#define IR_BB_EMPTY             (1<<7)
+#define IR_BB_PREV_EMPTY_ENTRY  (1<<8)
+#define IR_BB_OSR_ENTRY_LOADS   (1<<9) /* OSR Entry-point with register LOADs   */
+#define IR_BB_LOOP_WITH_ENTRY   (1<<10) /* set together with LOOP_HEADER if there is an ENTRY in the loop */

 /* The following flags are set by GCM */
-#define IR_BB_HAS_PHI          (1<<10)
-#define IR_BB_HAS_PI           (1<<11)
-#define IR_BB_HAS_PARAM        (1<<12)
-#define IR_BB_HAS_VAR          (1<<13)
+#define IR_BB_HAS_PHI           (1<<11)
+#define IR_BB_HAS_PI            (1<<12)
+#define IR_BB_HAS_PARAM         (1<<13)
+#define IR_BB_HAS_VAR           (1<<14)

 /* The following flags are set by BB scheduler */
-#define IR_BB_ALIGN_LOOP       (1<<14)
+#define IR_BB_ALIGN_LOOP        (1<<15)
+
+#define IR_BB_DESSA_TMP_INT     (1<<16) /* translation out of SSA may need temporary genral purpose register */
+#define IR_BB_DESSA_TMP_FP      (1<<17) /* translation out of SSA may need temporary floating point register */

 struct _ir_block {
 	uint32_t flags;
@@ -1157,6 +1203,7 @@ struct _ir_block {
 	union {
 		uint32_t loop_depth;
 		uint32_t next_succ;      /* used temporary for iterative Post Ordering */
+		uint32_t next_loop;      /* used temporary for loop nesting tree       */
 	};
 };

@@ -1225,21 +1272,25 @@ typedef struct _ir_use_pos       ir_use_pos;
 /* ir_use_pos.flags bits */
 #define IR_USE_MUST_BE_IN_REG            (1<<0)
 #define IR_USE_SHOULD_BE_IN_REG          (1<<1)
-#define IR_DEF_REUSES_OP1_REG            (1<<2)
-#define IR_DEF_CONFLICTS_WITH_INPUT_REGS (1<<3)
-#define IR_EXTEND_INPUTS_TO_NEXT         (1<<4) /* used for SNAPSHOT followed by GUARD */
+#define IR_HINT_TWO_REGS                 (1<<2)
+#define IR_DEF_REUSES_OP1_REG            (1<<3)
+#define IR_DEF_CONFLICTS_WITH_INPUT_REGS (1<<4)
+#define IR_EXTEND_INPUTS_TO_NEXT         (1<<5) /* used for SNAPSHOT followed by GUARD */

 #define IR_FUSED_USE                     (1<<6)
 #define IR_PHI_USE                       (1<<7)

 #define IR_OP1_MUST_BE_IN_REG            (1<<8)
 #define IR_OP1_SHOULD_BE_IN_REG          (1<<9)
-#define IR_OP2_MUST_BE_IN_REG            (1<<10)
-#define IR_OP2_SHOULD_BE_IN_REG          (1<<11)
-#define IR_OP3_MUST_BE_IN_REG            (1<<12)
-#define IR_OP3_SHOULD_BE_IN_REG          (1<<13)
+#define IR_OP1_HINT_TWO_REGS             (1<<10)
+#define IR_OP2_MUST_BE_IN_REG            (1<<11)
+#define IR_OP2_SHOULD_BE_IN_REG          (1<<12)
+#define IR_OP2_HINT_TWO_REGS             (1<<13)
+#define IR_OP3_MUST_BE_IN_REG            (1<<14)
+#define IR_OP3_SHOULD_BE_IN_REG          (1<<15)
+#define IR_OP3_HINT_TWO_REGS             (1<<16)

-#define IR_USE_FLAGS(def_flags, op_num)  (((def_flags) >> (6 + (IR_MIN((op_num), 3) * 2))) & 3)
+#define IR_USE_FLAGS(def_flags, op_num) (((def_flags) >> (5 + (IR_MIN((op_num), 3) * 3))) & 7)

 struct _ir_use_pos {
 	uint16_t       op_num; /* 0 - means result */
@@ -1266,10 +1317,14 @@ struct _ir_live_range {
 #define IR_LIVE_INTERVAL_SPILL_SPECIAL   (1<<6) /* spill slot is pre-allocated in a special area (see ir_ctx.spill_reserved_base) */
 #define IR_LIVE_INTERVAL_SPILLED         (1<<7)
 #define IR_LIVE_INTERVAL_SPLIT_CHILD     (1<<8)
+#define IR_LIVE_INTERVAL_TWO_REGS        (1<<9)

 struct _ir_live_interval {
 	uint8_t           type;
 	int8_t            reg;
+#if IR_X86_I64
+	int8_t            reg_hi;
+#endif
 	uint16_t          flags;
 	union {
 		int32_t       vreg;
@@ -1386,12 +1441,18 @@ struct _ir_call_conv_dsc {
 	uint8_t       shadow_store_size;          /* reserved stack space to keep arguemnts passed in registers (WIN64) */
 	uint8_t       int_param_regs_count;       /* number of registers for INT parameters */
 	uint8_t       fp_param_regs_count;        /* number of registers for FP parameters */
+	uint8_t       vector_param_regs_count;    /* number of registers for SIMD vector parameters */
 	int8_t        int_ret_reg;                /* register to return INT value */
+	int8_t        int_ret2_reg;               /* register to return second INT value (used to return I64 on 32-bit) */
 	int8_t        fp_ret_reg;                 /* register to return FP value */
+	int8_t        fp_ret2_reg;                /* register to return second FP value */
+	int8_t        vector_ret_reg;             /* register to return SIMD vector value */
+	int8_t        vector_ret2_reg;            /* register to return second SIMD vector value */
 	int8_t        fp_varargs_reg;             /* register to pass number of fp register arguments into vararg func */
 	int8_t        scratch_reg;                /* pseudo register to reffer srcatch regset (clobbered by call) */
 	const int8_t *int_param_regs;             /* registers for INT parameters */
 	const int8_t *fp_param_regs;              /* registers for FP parameters */
+	const int8_t *vector_param_regs;          /* registers for SIMD vector parameters */
 	ir_regset     preserved_regs;             /* preserved or callee-saved registers */
 };

@@ -1413,15 +1474,32 @@ typedef struct _ir_reg_alloc_data {
 } ir_reg_alloc_data;

 int32_t ir_allocate_spill_slot(ir_ctx *ctx, ir_type type);
+int32_t ir_allocate_big_spill_slot(ir_ctx *ctx, int32_t size);
+void ir_dump_reg(const ir_ctx *ctx, int8_t reg, ir_ref ref, bool store, FILE *f);

 IR_ALWAYS_INLINE void ir_set_alocated_reg(ir_ctx *ctx, ir_ref ref, int op_num, int8_t reg)
 {
 	int8_t *regs = ctx->regs[ref];

-	if (op_num > 0) {
-		/* regs[] is not limited by the declared boundary 4, the real boundary checked below */
-		IR_ASSERT(op_num <= IR_MAX(3, ctx->ir_base[ref].inputs_count));
+	/* regs[] is not limited by the declared boundary 4, the real boundary checked below */
+	IR_ASSERT(op_num >=0 && op_num <= IR_MAX(3, ctx->ir_base[ref].inputs_count));
+	regs[op_num] = reg;
+}
+
+IR_ALWAYS_INLINE void ir_set_alocated_tmp_reg(ir_ctx *ctx, ir_ref ref, int op_num, int8_t reg)
+{
+	int8_t *regs = ctx->regs[ref];
+
+	if (UNEXPECTED(op_num == 4)) {
+		/* Used for COND(I64, _, _) and SIMD instructions */
+		if (!ctx->tmp_regs) {
+			ctx->tmp_regs = ir_mem_malloc(ctx->insns_count);
+			memset(ctx->tmp_regs, -1, ctx->insns_count);
+		}
+		ctx->tmp_regs[ref] = reg;
+		return;
 	}
+	IR_ASSERT(op_num >= 0 && op_num <= 3);
 	regs[op_num] = reg;
 }

@@ -1430,7 +1508,7 @@ IR_ALWAYS_INLINE int8_t ir_get_alocated_reg(const ir_ctx *ctx, ir_ref ref, int o
 	int8_t *regs = ctx->regs[ref];

 	/* regs[] is not limited by the declared boundary 4, the real boundary checked below */
-	IR_ASSERT(op_num <= IR_MAX(3, ctx->ir_base[ref].inputs_count));
+	IR_ASSERT(op_num >= 0 && op_num <= IR_MAX(3, ctx->ir_base[ref].inputs_count));
 	return regs[op_num];
 }

@@ -1439,12 +1517,14 @@ IR_ALWAYS_INLINE int8_t ir_get_alocated_reg(const ir_ctx *ctx, ir_ref ref, int o
 /* ctx->rules[] flags */
 #define IR_FUSED     (1U<<31) /* Insn is fused into others (code is generated as part of the fusion root) */
 #define IR_SKIPPED   (1U<<30) /* Insn is skipped (code is not generated) */
-#define IR_SIMPLE    (1U<<29) /* Insn doesn't have any target constraints */
-#define IR_FUSED_REG (1U<<28) /* Register assignemnt may be stored in ctx->fused_regs instead of ctx->regs */
-#define IR_MAY_SWAP  (1U<<27) /* Allow swapping operands for better register allocation */
-#define IR_MAY_REUSE (1U<<26) /* Result may reuse register of the source */
+#define IR_NO_REG    (1U<<29) /* Result doesn't need register (used for TAILCALL) */
+#define IR_SIMPLE    (1U<<28) /* Insn doesn't have any target constraints */
+#define IR_FUSED_REG (1U<<27) /* Register assignemnt may be stored in ctx->fused_regs instead of ctx->regs */
+#define IR_MAY_SWAP  (1U<<26) /* Allow swapping operands for better register allocation */
+#define IR_MAY_REUSE (1U<<25) /* Result may reuse register of the source */
+#define IR_TWO_REGS  (1U<<24) /* Result needs two registers (used for 64-bit integers on x86) */

-#define IR_RULE_MASK 0xff
+#define IR_RULE_MASK 0xffff

 #define IR_MAX_REG_ARGS 64

@@ -1464,7 +1544,7 @@ typedef struct {
 	int8_t      def_reg;
 	uint8_t     tmps_count;
 	uint8_t     hints_count;
-	ir_tmp_reg  tmp_regs[3];
+	ir_tmp_reg  tmp_regs[4];
 	int8_t      hints[IR_MAX_REG_ARGS + 3];
 } ir_target_constraints;

diff --git a/ext/opcache/jit/ir/ir_ra.c b/ext/opcache/jit/ir/ir_ra.c
index f22e0608378..17f41319bb1 100644
--- a/ext/opcache/jit/ir/ir_ra.c
+++ b/ext/opcache/jit/ir/ir_ra.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (RA - Register Allocation, Liveness, Coalescing, SSA Deconstruction)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * See: "Linear Scan Register Allocation on SSA Form", Christian Wimmer and
@@ -64,7 +64,7 @@ static int ir_assign_virtual_registers_slow(ir_ctx *ctx)
 			flags = ir_op_flags[insn->op];
 			if (((flags & IR_OP_FLAG_DATA) && insn->op != IR_VAR && (insn->op != IR_PARAM || ctx->use_lists[i].count > 0))
 			 || ((flags & IR_OP_FLAG_MEM) && ctx->use_lists[i].count > 1)) {
-				if (!ctx->rules || !(ctx->rules[i] & (IR_FUSED|IR_SKIPPED))) {
+				if (!ctx->rules || !(ctx->rules[i] & (IR_FUSED|IR_SKIPPED|IR_NO_REG))) {
 					vregs[i] = ++vregs_count;
 				}
 			}
@@ -96,7 +96,7 @@ int ir_assign_virtual_registers(ir_ctx *ctx)
 	for (i = 1, insn = &ctx->ir_base[1]; i < ctx->insns_count; i++, insn++) {
 		uint32_t v = 0;

-		if (ctx->rules[i] && !(ctx->rules[i] & (IR_FUSED|IR_SKIPPED))) {
+		if (ctx->rules[i] && !(ctx->rules[i] & (IR_FUSED|IR_SKIPPED|IR_NO_REG))) {
 			uint32_t flags = ir_op_flags[insn->op];

 			if ((flags & IR_OP_FLAG_DATA)
@@ -121,6 +121,9 @@ static ir_live_interval *ir_new_live_range(ir_ctx *ctx, int v, ir_live_pos start

 	ival->type = IR_VOID;
 	ival->reg = IR_REG_NONE;
+#if IR_X86_I64
+	ival->reg_hi = IR_REG_NONE;
+#endif
 	ival->flags = 0;
 	ival->vreg = v;
 	ival->stack_spill_pos = -1; // not allocated
@@ -232,6 +235,9 @@ static void ir_add_fixed_live_range(ir_ctx *ctx, ir_reg reg, ir_live_pos start,
 		ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
 		ival->type = IR_VOID;
 		ival->reg = reg;
+#if IR_X86_I64
+		ival->reg_hi = IR_REG_NONE;
+#endif
 		ival->flags = IR_LIVE_INTERVAL_FIXED;
 		ival->vreg = v;
 		ival->stack_spill_pos = -1; // not allocated
@@ -270,6 +276,9 @@ static void ir_add_tmp(ir_ctx *ctx, ir_ref ref, ir_ref tmp_ref, int32_t tmp_op_n

 	ival->type = tmp_reg.type;
 	ival->reg = IR_REG_NONE;
+#if IR_X86_I64
+	ival->reg_hi = IR_REG_NONE;
+#endif
 	ival->flags = IR_LIVE_INTERVAL_TEMP;
 	ival->tmp_ref = tmp_ref;
 	ival->tmp_op_num = tmp_op_num;
@@ -298,21 +307,6 @@ static void ir_add_tmp(ir_ctx *ctx, ir_ref ref, ir_ref tmp_ref, int32_t tmp_op_n
 	return;
 }

-static bool ir_has_tmp(ir_ctx *ctx, ir_ref ref, int32_t op_num)
-{
-	ir_live_interval *ival = ctx->live_intervals[0];
-
-	if (ival) {
-		while (ival && IR_LIVE_POS_TO_REF(ival->range.start) <= ref) {
-			if (ival->tmp_ref == ref && ival->tmp_op_num == op_num) {
-				return 1;
-			}
-			ival = ival->next;
-		}
-	}
-	return 0;
-}
-
 static ir_live_interval *ir_fix_live_range(ir_ctx *ctx, int v, ir_live_pos old_start, ir_live_pos new_start)
 {
 	ir_live_interval *ival = ctx->live_intervals[v];
@@ -385,16 +379,15 @@ static void ir_add_phi_use(ir_ctx *ctx, ir_live_interval *ival, int op_num, ir_l
 	ir_add_use_pos(ctx, ival, use_pos);
 }

-static void ir_add_hint(ir_ctx *ctx, ir_ref ref, ir_live_pos pos, ir_reg hint)
+static void ir_add_hint(ir_ctx *ctx, ir_live_interval *ival, ir_live_pos pos, ir_reg hint, uint8_t flags)
 {
-	ir_live_interval *ival = ctx->live_intervals[ctx->vregs[ref]];
-
 	if (!(ival->flags & IR_LIVE_INTERVAL_HAS_HINT_REGS)) {
 		ir_use_pos *use_pos = ival->use_pos;

 		while (use_pos) {
 			if (use_pos->pos == pos) {
 				if (use_pos->hint == IR_REG_NONE) {
+					use_pos->flags |= flags;
 					use_pos->hint = hint;
 					ival->flags |= IR_LIVE_INTERVAL_HAS_HINT_REGS;
 				}
@@ -424,7 +417,19 @@ static void ir_hint_propagation(ir_ctx *ctx)
 					}
 				} else if (use_pos->hint != IR_REG_NONE) {
 					if (hint_use_pos) {
-						ir_add_hint(ctx, hint_use_pos->hint_ref, hint_use_pos->pos, use_pos->hint);
+						ir_live_interval *hint_ival = ctx->live_intervals[ctx->vregs[hint_use_pos->hint_ref]];
+
+#if IR_X86_I64
+						if (use_pos->flags & IR_HINT_TWO_REGS) {
+
+							if (hint_ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+								ir_add_hint(ctx, hint_ival, hint_use_pos->pos, use_pos->hint, IR_HINT_TWO_REGS);
+							} else {
+								ir_add_hint(ctx, hint_ival, hint_use_pos->pos, IR_REG_I64_LO(use_pos->hint), 0);
+							}
+						} else
+#endif
+						ir_add_hint(ctx, hint_ival, hint_use_pos->pos, use_pos->hint, 0);
 						hint_use_pos = NULL;
 					}
 				}
@@ -535,7 +540,7 @@ static void ir_add_fusion_ranges(ir_ctx *ctx, ir_ref ref, ir_ref input, ir_block
 		n = IR_INPUT_EDGES_COUNT(flags);
 		j = 1;
 		p = insn->ops + j;
-		if (flags & IR_OP_FLAG_CONTROL) {
+		if (flags & (IR_OP_FLAG_CONTROL|IR_OP_FLAG_PINNED)) {
 			j++;
 			p++;
 		}
@@ -717,12 +722,21 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 			if (ctx->rules) {
 				int n;

+#if IR_X86_I64
+				if (ctx->rules[ref] & IR_TWO_REGS) {
+					v = ctx->vregs[ref];
+					if (v) {
+						IR_ASSERT(ctx->live_intervals[v]);
+						ctx->live_intervals[v]->flags |= IR_LIVE_INTERVAL_TWO_REGS;
+					}
+				}
+#endif
+
 				if (ctx->rules[ref] & (IR_FUSED|IR_SKIPPED)) {
-					if (((ctx->rules[ref] & IR_RULE_MASK) == IR_VAR
-					  || (ctx->rules[ref] & IR_RULE_MASK) == IR_ALLOCA)
+					if (((ctx->rules[ref] & IR_RULE_MASK) == IR_ALLOCA)
 					 && ctx->use_lists[ref].count > 0) {
 						insn = &ctx->ir_base[ref];
-						if (insn->op != IR_VADDR) {
+						if (insn->op == IR_VAR || insn->op == IR_ALLOCA) {
 							insn->op3 = ctx->vars;
 							ctx->vars = ref;
 						}
@@ -763,6 +777,12 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						def_pos = IR_SAVE_LIVE_POS_FROM_REF(ref);
 						if (insn->op == IR_PARAM || insn->op == IR_RLOAD) {
 							/* parameter register must be kept before it's copied */
+#if IR_X86_I64
+							if (def_flags & IR_HINT_TWO_REGS) {
+								ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), IR_START_LIVE_POS_FROM_REF(bb->start), def_pos);
+								ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), IR_START_LIVE_POS_FROM_REF(bb->start), def_pos);
+							} else
+#endif
 							ir_add_fixed_live_range(ctx, reg, IR_START_LIVE_POS_FROM_REF(bb->start), def_pos);
 						}
 					} else if (def_flags & IR_DEF_REUSES_OP1_REG) {
@@ -843,6 +863,7 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 				ir_live_pos use_pos;
 				ir_ref hint_ref = 0;
 				uint32_t v;
+				uint32_t use_flags = IR_USE_FLAGS(def_flags, j);

 				if (input > 0) {
 					v = ctx->vregs[input];
@@ -850,6 +871,12 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						use_pos = IR_USE_LIVE_POS_FROM_REF(ref);
 						if (reg != IR_REG_NONE) {
 							use_pos = IR_LOAD_LIVE_POS_FROM_REF(ref);
+#if IR_X86_I64
+							if (use_flags & IR_HINT_TWO_REGS) {
+								ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), use_pos, use_pos + IR_USE_SUB_REF);
+								ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), use_pos, use_pos + IR_USE_SUB_REF);
+							} else
+#endif
 							ir_add_fixed_live_range(ctx, reg, use_pos, use_pos + IR_USE_SUB_REF);
 						} else if (def_flags & IR_DEF_REUSES_OP1_REG) {
 							if (j == 1) {
@@ -869,7 +896,7 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						} else {
 							ival = ctx->live_intervals[v];
 						}
-						ir_add_use(ctx, ival, j, use_pos, reg, IR_USE_FLAGS(def_flags, j), hint_ref);
+						ir_add_use(ctx, ival, j, use_pos, reg, use_flags, hint_ref);
 					} else {
 						if (ctx->rules) {
 							if ((ctx->rules[input] & (IR_FUSED|IR_SKIPPED)) == IR_FUSED) {
@@ -880,11 +907,23 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						}
 						if (reg != IR_REG_NONE) {
 							use_pos = IR_LOAD_LIVE_POS_FROM_REF(ref);
+#if IR_X86_I64
+							if (use_flags & IR_HINT_TWO_REGS) {
+								ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), use_pos, use_pos + IR_USE_SUB_REF);
+								ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), use_pos, use_pos + IR_USE_SUB_REF);
+							} else
+#endif
 							ir_add_fixed_live_range(ctx, reg, use_pos, use_pos + IR_USE_SUB_REF);
 						}
 					}
 				} else if (reg != IR_REG_NONE) {
 					use_pos = IR_LOAD_LIVE_POS_FROM_REF(ref);
+#if IR_X86_I64
+					if (use_flags & IR_HINT_TWO_REGS) {
+						ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), use_pos, use_pos + IR_USE_SUB_REF);
+						ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), use_pos, use_pos + IR_USE_SUB_REF);
+					} else
+#endif
 					ir_add_fixed_live_range(ctx, reg, use_pos, use_pos + IR_USE_SUB_REF);
 				}
 			}
@@ -1366,12 +1405,21 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 			if (ctx->rules) {
 				int n;

+#if IR_X86_I64
+				if (ctx->rules[ref] & IR_TWO_REGS) {
+					v = ctx->vregs[ref];
+					if (v) {
+						IR_ASSERT(ctx->live_intervals[v]);
+						ctx->live_intervals[v]->flags |= IR_LIVE_INTERVAL_TWO_REGS;
+					}
+				}
+#endif
+
 				if (ctx->rules[ref] & (IR_FUSED|IR_SKIPPED)) {
-					if (((ctx->rules[ref] & IR_RULE_MASK) == IR_VAR
-					  || (ctx->rules[ref] & IR_RULE_MASK) == IR_ALLOCA)
+					if (((ctx->rules[ref] & IR_RULE_MASK) == IR_ALLOCA)
 					 && ctx->use_lists[ref].count > 0) {
 						insn = &ctx->ir_base[ref];
-						if (insn->op != IR_VADDR && insn->op != IR_PARAM) {
+						if (insn->op == IR_VAR || insn->op == IR_ALLOCA) {
 							insn->op3 = ctx->vars;
 							ctx->vars = ref;
 						}
@@ -1410,6 +1458,12 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						def_pos = IR_SAVE_LIVE_POS_FROM_REF(ref);
 						if (insn->op == IR_PARAM || insn->op == IR_RLOAD) {
 							/* parameter register must be kept before it's copied */
+#if IR_X86_I64
+							if (def_flags & IR_HINT_TWO_REGS) {
+								ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), IR_START_LIVE_POS_FROM_REF(bb->start), def_pos);
+								ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), IR_START_LIVE_POS_FROM_REF(bb->start), def_pos);
+							} else
+#endif
 							ir_add_fixed_live_range(ctx, reg, IR_START_LIVE_POS_FROM_REF(bb->start), def_pos);
 						}
 					} else if (def_flags & IR_DEF_REUSES_OP1_REG) {
@@ -1490,6 +1544,7 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 				ir_live_pos use_pos;
 				ir_ref hint_ref = 0;
 				uint32_t v;
+				uint32_t use_flags = IR_USE_FLAGS(def_flags, j);

 				if (input > 0) {
 					v = ctx->vregs[input];
@@ -1497,6 +1552,12 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						use_pos = IR_USE_LIVE_POS_FROM_REF(ref);
 						if (reg != IR_REG_NONE) {
 							use_pos = IR_LOAD_LIVE_POS_FROM_REF(ref);
+#if IR_X86_I64
+							if (use_flags & IR_HINT_TWO_REGS) {
+								ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), use_pos, use_pos + IR_USE_SUB_REF);
+								ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), use_pos, use_pos + IR_USE_SUB_REF);
+							} else
+#endif
 							ir_add_fixed_live_range(ctx, reg, use_pos, use_pos + IR_USE_SUB_REF);
 						} else if (def_flags & IR_DEF_REUSES_OP1_REG) {
 							if (j == 1) {
@@ -1520,7 +1581,7 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						} else {
 							ival = ctx->live_intervals[v];
 						}
-						ir_add_use(ctx, ival, j, use_pos, reg, IR_USE_FLAGS(def_flags, j), hint_ref);
+						ir_add_use(ctx, ival, j, use_pos, reg, use_flags, hint_ref);
 					} else {
 						if (ctx->rules) {
 							if ((ctx->rules[input] & (IR_FUSED|IR_SKIPPED)) == IR_FUSED) {
@@ -1531,11 +1592,23 @@ int ir_compute_live_ranges(ir_ctx *ctx)
 						}
 						if (reg != IR_REG_NONE) {
 							use_pos = IR_LOAD_LIVE_POS_FROM_REF(ref);
+#if IR_X86_I64
+							if (use_flags & IR_HINT_TWO_REGS) {
+								ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), use_pos, use_pos + IR_USE_SUB_REF);
+								ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), use_pos, use_pos + IR_USE_SUB_REF);
+							} else
+#endif
 							ir_add_fixed_live_range(ctx, reg, use_pos, use_pos + IR_USE_SUB_REF);
 						}
 					}
 				} else if (reg != IR_REG_NONE) {
 					use_pos = IR_LOAD_LIVE_POS_FROM_REF(ref);
+#if IR_X86_I64
+					if (use_flags & IR_HINT_TWO_REGS) {
+						ir_add_fixed_live_range(ctx, IR_REG_I64_LO(reg), use_pos, use_pos + IR_USE_SUB_REF);
+						ir_add_fixed_live_range(ctx, IR_REG_I64_HI(reg), use_pos, use_pos + IR_USE_SUB_REF);
+					} else
+#endif
 					ir_add_fixed_live_range(ctx, reg, use_pos, use_pos + IR_USE_SUB_REF);
 				}
 			}
@@ -1737,11 +1810,16 @@ static void ir_vregs_coalesce(ir_ctx *ctx, uint32_t v1, uint32_t v2, ir_ref from
 	}
 }

-static void ir_add_phi_move(ir_ctx *ctx, uint32_t b, ir_ref from, ir_ref to)
+static void ir_add_phi_move(ir_ctx *ctx, uint32_t b, ir_type type, ir_ref from, ir_ref to)
 {
 	if (IR_IS_CONST_REF(from) || ctx->vregs[from] != ctx->vregs[to]) {
 		ctx->cfg_blocks[b].flags &= ~IR_BB_EMPTY;
 		ctx->cfg_blocks[b].flags |= IR_BB_DESSA_MOVES;
+		if (IR_IS_TYPE_INT(type)) {
+			ctx->cfg_blocks[b].flags |= IR_BB_DESSA_TMP_INT;
+		} else {
+			ctx->cfg_blocks[b].flags |= IR_BB_DESSA_TMP_FP;
+		}
 		ctx->flags2 |= IR_LR_HAVE_DESSA_MOVES;
 #if 0
 		fprintf(stderr, "BB%d: MOV %d -> %d\n", b, from, to);
@@ -2022,12 +2100,12 @@ int ir_coalesce(ir_ctx *ctx)
 								}
 							}
 #endif
-							ir_add_phi_move(ctx, b, input, use);
+							ir_add_phi_move(ctx, b, insn->type, input, use);
 						}
 					}
 				} else {
 					/* Move for constant input */
-					ir_add_phi_move(ctx, b, input, use);
+					ir_add_phi_move(ctx, b, insn->type, input, use);
 				}
 			}
 		}
@@ -2137,10 +2215,17 @@ int ir_compute_dessa_moves(ir_ctx *ctx)
 					insn = &ctx->ir_base[use];
 					if (insn->op == IR_PHI) {
 						for (j = 2; j <= k; j++) {
-							if (IR_IS_CONST_REF(ir_insn_op(insn, j)) || ctx->vregs[ir_insn_op(insn, j)] != ctx->vregs[use]) {
+							ir_ref input = ir_insn_op(insn, j);
+
+							if (IR_IS_CONST_REF(input) || ctx->vregs[input] != ctx->vregs[use]) {
 								int pred = ctx->cfg_edges[bb->predecessors + (j-2)];
 								ctx->cfg_blocks[pred].flags &= ~IR_BB_EMPTY;
 								ctx->cfg_blocks[pred].flags |= IR_BB_DESSA_MOVES;
+								if (IR_IS_TYPE_INT(insn->type)) {
+									ctx->cfg_blocks[pred].flags |= IR_BB_DESSA_TMP_INT;
+								} else {
+									ctx->cfg_blocks[pred].flags |= IR_BB_DESSA_TMP_FP;
+								}
 								ctx->flags2 |= IR_LR_HAVE_DESSA_MOVES;
 							}
 						}
@@ -2185,6 +2270,20 @@ int ir_gen_dessa_moves(ir_ctx *ctx, uint32_t b, emit_copy_t emit_copy, void *dat

 	k = ir_phi_input_number(ctx, succ_bb, b);

+	if (use_list->count == 2) {
+		/* Simple version for BB with single PHI */
+		ref = ctx->use_edges[use_list->refs];
+		insn = &ctx->ir_base[ref];
+		if (insn->op != IR_PHI) {
+			ref = ctx->use_edges[use_list->refs + 1];
+			insn = &ctx->ir_base[ref];
+		}
+		IR_ASSERT(insn->op == IR_PHI);
+		input = ir_insn_op(insn, k);
+		emit_copy(ctx, insn->type, input, ref, data);
+		return 1;
+	}
+
 	loc = ir_mem_malloc((ctx->vregs_count + 1) * 4 * sizeof(ir_ref));
 	pred = loc + ctx->vregs_count + 1;
 	src = pred + ctx->vregs_count + 1;
@@ -2279,6 +2378,18 @@ int ir_gen_dessa_moves(ir_ctx *ctx, uint32_t b, emit_copy_t emit_copy, void *dat
 /* Linear Scan Register Allocation */

 #ifdef IR_DEBUG
+# if IR_X86_I64
+#  define IR_REG_NAME_FMT        "%s%s%s"
+#  define IR_REG_NAME_VAL(_ival) ((_ival->flags & IR_LIVE_INTERVAL_TWO_REGS) ? \
+									ir_reg_name((_ival)->reg, IR_U32) : \
+									ir_reg_name((_ival)->reg, (_ival)->type)), \
+                                 ((_ival->flags & IR_LIVE_INTERVAL_TWO_REGS) ? " and " : ""), \
+                                 ((_ival->flags & IR_LIVE_INTERVAL_TWO_REGS) ? \
+									ir_reg_name((_ival)->reg_hi, IR_U32) : "")
+# else
+#  define IR_REG_NAME_FMT        "%s"
+#  define IR_REG_NAME_VAL(_ival) ir_reg_name((_ival)->reg, (_ival)->type)
+# endif
 # define IR_LOG_LSRA(action, ival, comment) do { \
 		if (ctx->flags & IR_DEBUG_RA) { \
 			ir_live_interval *_ival = (ival); \
@@ -2295,11 +2406,11 @@ int ir_gen_dessa_moves(ir_ctx *ctx, uint32_t b, emit_copy_t emit_copy, void *dat
 			ir_live_interval *_ival = (ival); \
 			ir_live_pos _start = _ival->range.start; \
 			ir_live_pos _end = _ival->end; \
-			fprintf(stderr, action " R%d [%d.%d...%d.%d) to %s" comment "\n", \
+			fprintf(stderr, action " R%d [%d.%d...%d.%d) to " IR_REG_NAME_FMT comment "\n", \
 				(_ival->flags & IR_LIVE_INTERVAL_TEMP) ? 0 : _ival->vreg, \
 				IR_LIVE_POS_TO_REF(_start), IR_LIVE_POS_TO_SUB_REF(_start), \
 				IR_LIVE_POS_TO_REF(_end), IR_LIVE_POS_TO_SUB_REF(_end), \
-				ir_reg_name(_ival->reg, _ival->type)); \
+				IR_REG_NAME_VAL(_ival)); \
 		} \
 	} while (0)
 # define IR_LOG_LSRA_SPLIT(ival, pos) do { \
@@ -2321,11 +2432,11 @@ int ir_gen_dessa_moves(ir_ctx *ctx, uint32_t b, emit_copy_t emit_copy, void *dat
 			ir_live_pos _start = _ival->range.start; \
 			ir_live_pos _end = _ival->end; \
 			ir_live_pos _pos = (pos); \
-			fprintf(stderr, action " R%d [%d.%d...%d.%d) assigned to %s at %d.%d\n", \
+			fprintf(stderr, action " R%d [%d.%d...%d.%d) assigned to " IR_REG_NAME_FMT " at %d.%d\n", \
 				(_ival->flags & IR_LIVE_INTERVAL_TEMP) ? 0 : _ival->vreg, \
 				IR_LIVE_POS_TO_REF(_start), IR_LIVE_POS_TO_SUB_REF(_start), \
 				IR_LIVE_POS_TO_REF(_end), IR_LIVE_POS_TO_SUB_REF(_end), \
-				ir_reg_name(_ival->reg, _ival->type), \
+				IR_REG_NAME_VAL(_ival), \
 				IR_LIVE_POS_TO_REF(_pos), IR_LIVE_POS_TO_SUB_REF(_pos)); \
 		} \
 	} while (0)
@@ -2515,7 +2626,12 @@ static ir_live_interval *ir_split_interval_at(ir_ctx *ctx, ir_live_interval *iva
 	child = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
 	child->type = ival->type;
 	child->reg = IR_REG_NONE;
+#if IR_X86_I64
+	child->reg_hi = IR_REG_NONE;
+	child->flags = (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) | IR_LIVE_INTERVAL_SPLIT_CHILD;
+#else
 	child->flags = IR_LIVE_INTERVAL_SPLIT_CHILD;
+#endif
 	child->vreg = ival->vreg;
 	child->stack_spill_pos = -1; // not allocated
 	child->range.start = pos;
@@ -2560,12 +2676,17 @@ static ir_live_interval *ir_split_interval_at(ir_ctx *ctx, ir_live_interval *iva
 static int32_t ir_allocate_small_spill_slot(ir_ctx *ctx, size_t size)
 {
 	ir_reg_alloc_data *data = ctx->data;
-	int32_t ret;
+	int32_t ret, n;
+
+	if (size == 0) {
+		return IR_NULL;
+	}

-	IR_ASSERT(size == 0 || size == 1 || size == 2 || size == 4 || size == 8);
-	if (data->handled && data->handled[size]) {
-		ret = data->handled[size]->stack_spill_pos;
-		data->handled[size] = data->handled[size]->list_next;
+	IR_ASSERT(size == 1 || size == 2 || size == 4 || size == 8);
+	n = ir_ntz(size);
+	if (data->handled && data->handled[n]) {
+		ret = data->handled[n]->stack_spill_pos;
+		data->handled[n] = data->handled[n]->list_next;
 	} else if (size == 8) {
 		ret = ctx->stack_frame_size;
 		ctx->stack_frame_size += 8;
@@ -2573,9 +2694,9 @@ static int32_t ir_allocate_small_spill_slot(ir_ctx *ctx, size_t size)
 		if (data->unused_slot_4) {
 			ret = data->unused_slot_4;
 			data->unused_slot_4 = 0;
-	    } else if (data->handled && data->handled[8]) {
-			ret = data->handled[8]->stack_spill_pos;
-			data->handled[8] = data->handled[8]->list_next;
+	    } else if (data->handled && data->handled[3]) {
+			ret = data->handled[3]->stack_spill_pos;
+			data->handled[3] = data->handled[3]->list_next;
 			data->unused_slot_4 = ret + 4;
 		} else {
 			ret = ctx->stack_frame_size;
@@ -2594,13 +2715,13 @@ static int32_t ir_allocate_small_spill_slot(ir_ctx *ctx, size_t size)
 			ret = data->unused_slot_4;
 			data->unused_slot_2 = data->unused_slot_4 + 2;
 			data->unused_slot_4 = 0;
-	    } else if (data->handled && data->handled[4]) {
-			ret = data->handled[4]->stack_spill_pos;
-			data->handled[4] = data->handled[4]->list_next;
+	    } else if (data->handled && data->handled[2]) {
+			ret = data->handled[2]->stack_spill_pos;
+			data->handled[2] = data->handled[2]->list_next;
 			data->unused_slot_2 = ret + 2;
-	    } else if (data->handled && data->handled[8]) {
-			ret = data->handled[8]->stack_spill_pos;
-			data->handled[8] = data->handled[8]->list_next;
+	    } else if (data->handled && data->handled[3]) {
+			ret = data->handled[3]->stack_spill_pos;
+			data->handled[3] = data->handled[3]->list_next;
 			data->unused_slot_2 = ret + 2;
 			data->unused_slot_4 = ret + 4;
 		} else {
@@ -2626,18 +2747,18 @@ static int32_t ir_allocate_small_spill_slot(ir_ctx *ctx, size_t size)
 			data->unused_slot_1 = data->unused_slot_4 + 1;
 			data->unused_slot_2 = data->unused_slot_4 + 2;
 			data->unused_slot_4 = 0;
+	    } else if (data->handled && data->handled[1]) {
+			ret = data->handled[1]->stack_spill_pos;
+			data->handled[1] = data->handled[1]->list_next;
+			data->unused_slot_1 = ret + 1;
 	    } else if (data->handled && data->handled[2]) {
 			ret = data->handled[2]->stack_spill_pos;
 			data->handled[2] = data->handled[2]->list_next;
 			data->unused_slot_1 = ret + 1;
-	    } else if (data->handled && data->handled[4]) {
-			ret = data->handled[4]->stack_spill_pos;
-			data->handled[4] = data->handled[4]->list_next;
-			data->unused_slot_1 = ret + 1;
 			data->unused_slot_2 = ret + 2;
-	    } else if (data->handled && data->handled[8]) {
-			ret = data->handled[8]->stack_spill_pos;
-			data->handled[8] = data->handled[8]->list_next;
+	    } else if (data->handled && data->handled[3]) {
+			ret = data->handled[3]->stack_spill_pos;
+			data->handled[3] = data->handled[3]->list_next;
 			data->unused_slot_1 = ret + 1;
 			data->unused_slot_2 = ret + 2;
 			data->unused_slot_4 = ret + 4;
@@ -2658,12 +2779,7 @@ static int32_t ir_allocate_small_spill_slot(ir_ctx *ctx, size_t size)
 	return ret;
 }

-int32_t ir_allocate_spill_slot(ir_ctx *ctx, ir_type type)
-{
-	return ir_allocate_small_spill_slot(ctx, ir_type_size[type]);
-}
-
-static int32_t ir_allocate_big_spill_slot(ir_ctx *ctx, int32_t size)
+int32_t ir_allocate_big_spill_slot(ir_ctx *ctx, int32_t size)
 {
 	int32_t ret;

@@ -2676,6 +2792,17 @@ static int32_t ir_allocate_big_spill_slot(ir_ctx *ctx, int32_t size)
 		return ir_allocate_small_spill_slot(ctx, size);
 	}

+	if (size <= 64 && (size & (size - 1)) == 0) {
+		uint32_t n = ir_ntz(size);
+		ir_reg_alloc_data *data = ctx->data;
+
+		if (data->handled && data->handled[n]) {
+			ret = data->handled[n]->stack_spill_pos;
+			data->handled[n] = data->handled[n]->list_next;
+			return ret;
+		}
+	}
+
 	/* Align stack allocated data to 16 byte */
 	ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
 	ret = IR_ALIGNED_SIZE(ctx->stack_frame_size, 16);
@@ -2685,6 +2812,21 @@ static int32_t ir_allocate_big_spill_slot(ir_ctx *ctx, int32_t size)
 	return ret;
 }

+int32_t ir_allocate_spill_slot(ir_ctx *ctx, ir_type type)
+{
+	if (IR_IS_TYPE_SCALAR(type)) {
+		return ir_allocate_small_spill_slot(ctx, ir_type_size[type]);
+	} else {
+		int32_t size;
+
+		IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+		size = IR_VECTOR_SIZE(type);
+		size = IR_MAX(size, 4);
+		return ir_allocate_big_spill_slot(ctx, size);
+	}
+}
+
+
 static ir_reg ir_get_first_reg_hint(ir_ctx *ctx, ir_live_interval *ival, ir_regset available)
 {
 	ir_use_pos *use_pos;
@@ -2693,8 +2835,27 @@ static ir_reg ir_get_first_reg_hint(ir_ctx *ctx, ir_live_interval *ival, ir_regs
 	use_pos = ival->use_pos;
 	while (use_pos) {
 		reg = use_pos->hint;
-		if (reg >= 0 && IR_REGSET_IN(available, reg)) {
-			return reg;
+		if (reg >= 0) {
+#if IR_X86_I64
+			if (use_pos->flags & IR_HINT_TWO_REGS) {
+				ir_reg reg_hi = IR_REG_I64_HI(reg);
+				ir_reg reg_lo = IR_REG_I64_LO(reg);
+
+				IR_ASSERT(ival->flags & IR_LIVE_INTERVAL_TWO_REGS);
+				if (IR_REGSET_IN(available, reg_lo) && IR_REGSET_IN(available, reg_hi)) {
+					return reg;
+				}
+			} else
+#endif
+			if (IR_REGSET_IN(available, reg)) {
+#if IR_X86_I64
+				if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+					/* Use the same reg for reg_hi */
+					return IR_REG_I64_PAIR(reg, reg);
+				} else
+#endif
+				return reg;
+			}
 		}
 		use_pos = use_pos->next;
 	}
@@ -2711,10 +2872,31 @@ static ir_reg ir_try_allocate_preferred_reg(ir_ctx *ctx, ir_live_interval *ival,
 		use_pos = ival->use_pos;
 		while (use_pos) {
 			reg = use_pos->hint;
-			if (reg >= 0 && IR_REGSET_IN(available, reg)) {
-				if (ival->end <= freeUntilPos[reg]) {
-					/* register available for the whole interval */
-					return reg;
+			if (reg >= 0) {
+#if IR_X86_I64
+				if (use_pos->flags & IR_HINT_TWO_REGS) {
+					ir_reg reg_hi = IR_REG_I64_HI(reg);
+					ir_reg reg_lo = IR_REG_I64_LO(reg);
+
+					IR_ASSERT(ival->flags & IR_LIVE_INTERVAL_TWO_REGS);
+					if (IR_REGSET_IN(available, reg_hi) && IR_REGSET_IN(available, reg_lo)) {
+						if (ival->end <= freeUntilPos[reg_lo] && ival->end <= freeUntilPos[reg_hi]) {
+							return reg;
+						}
+					}
+				} else
+#endif
+				if (IR_REGSET_IN(available, reg)) {
+					if (ival->end <= freeUntilPos[reg]) {
+						/* register available for the whole interval */
+#if IR_X86_I64
+						if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+							/* Use the same reg for reg_hi to perform arbitrary allocation */
+							return IR_REG_I64_PAIR(reg, reg);
+						} else
+#endif
+						return reg;
+					}
 				}
 			}
 			use_pos = use_pos->next;
@@ -2725,11 +2907,33 @@ static ir_reg ir_try_allocate_preferred_reg(ir_ctx *ctx, ir_live_interval *ival,
 		use_pos = ival->use_pos;
 		while (use_pos) {
 			if (use_pos->hint_ref > 0) {
-				reg = ctx->live_intervals[ctx->vregs[use_pos->hint_ref]]->reg;
-				if (reg >= 0 && IR_REGSET_IN(available, reg)) {
-					if (ival->end <= freeUntilPos[reg]) {
-						/* register available for the whole interval */
-						return reg;
+				ir_live_interval *hint_ival = ctx->live_intervals[ctx->vregs[use_pos->hint_ref]];
+
+				reg = hint_ival->reg;
+				if (reg >= 0) {
+#if IR_X86_I64
+					if ((hint_ival->flags & IR_LIVE_INTERVAL_TWO_REGS)
+					 && (ival->flags & IR_LIVE_INTERVAL_TWO_REGS)) {
+							ir_reg reg_hi = hint_ival->reg_hi;
+
+							if (IR_REGSET_IN(available, reg) && IR_REGSET_IN(available, reg_hi)) {
+								if (ival->end <= freeUntilPos[reg] && ival->end <= freeUntilPos[reg_hi]) {
+									return IR_REG_I64_PAIR(reg, reg_hi);
+								}
+						}
+					} else
+#endif
+					if (IR_REGSET_IN(available, reg)) {
+						if (ival->end <= freeUntilPos[reg]) {
+							/* register available for the whole interval */
+#if IR_X86_I64
+							if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+								/* Use the same reg for reg_hi to perform arbitrary allocation */
+								return IR_REG_I64_PAIR(reg, reg);
+							} else
+#endif
+							return reg;
+						}
 					}
 				}
 			}
@@ -2748,13 +2952,53 @@ static ir_reg ir_get_preferred_reg(ir_ctx *ctx, ir_live_interval *ival, ir_regse
 	use_pos = ival->use_pos;
 	while (use_pos) {
 		reg = use_pos->hint;
-		if (reg >= 0 && IR_REGSET_IN(available, reg)) {
-			return reg;
-		} else if (use_pos->hint_ref > 0) {
-			reg = ctx->live_intervals[ctx->vregs[use_pos->hint_ref]]->reg;
-			if (reg >= 0 && IR_REGSET_IN(available, reg)) {
+		if (reg >= 0) {
+#if IR_X86_I64
+			if (use_pos->flags & IR_HINT_TWO_REGS) {
+				ir_reg reg_hi = IR_REG_I64_HI(reg);
+				ir_reg reg_lo = IR_REG_I64_LO(reg);
+
+				IR_ASSERT(ival->flags & IR_LIVE_INTERVAL_TWO_REGS);
+				if (IR_REGSET_IN(available, reg_lo) && IR_REGSET_IN(available, reg_hi)) {
+					return reg;
+				}
+			} else
+#endif
+			if (IR_REGSET_IN(available, reg)) {
+#if IR_X86_I64
+				if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+					/* Use the same reg for reg_hi to perform arbitrary allocation */
+					return IR_REG_I64_PAIR(reg, reg);
+				} else
+#endif
 				return reg;
 			}
+		} else if (use_pos->hint_ref > 0) {
+			ir_live_interval *hint_ival = ctx->live_intervals[ctx->vregs[use_pos->hint_ref]];
+
+			reg = hint_ival->reg;
+			if (reg >= 0) {
+#if IR_X86_I64
+				if ((hint_ival->flags & IR_LIVE_INTERVAL_TWO_REGS)
+				 && (ival->flags & IR_LIVE_INTERVAL_TWO_REGS)) {
+					ir_reg reg_hi = hint_ival->reg_hi;
+
+					IR_ASSERT(reg_hi >= 0);
+					if (IR_REGSET_IN(available, reg) && IR_REGSET_IN(available, reg_hi)) {
+						return IR_REG_I64_PAIR(reg, reg_hi);
+					}
+				} else
+#endif
+				if (IR_REGSET_IN(available, reg)) {
+#if IR_X86_I64
+					if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+						/* Use the same reg for reg_hi to perform arbitrary allocation */
+						return IR_REG_I64_PAIR(reg, reg);
+					} else
+#endif
+					return reg;
+				}
+			}
 		}
 		use_pos = use_pos->next;
 	}
@@ -2860,7 +3104,7 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 	ir_live_interval *other;
 	ir_regset available, overlapped, scratch;

-	if (IR_IS_TYPE_FP(ival->type)) {
+	if (IR_IS_TYPE_FP(ival->type) || IR_IS_TYPE_VECTOR(ival->type)) {
 		available = IR_REGSET_FP;
 		/* set freeUntilPos of all physical registers to maxInt */
 		for (i = IR_REG_FP_FIRST; i <= IR_REG_FP_LAST; i++) {
@@ -2898,6 +3142,13 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 		} else {
 			IR_REGSET_EXCL(available, reg);
 		}
+#if IR_X86_I64
+		if (other->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+			reg = other->reg_hi;
+			IR_ASSERT(reg >= 0 && reg < IR_REG_NUM);
+			IR_REGSET_EXCL(available, reg);
+		 }
+#endif
 		other = other->list_next;
 	}

@@ -2930,6 +3181,18 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 						freeUntilPos[reg] = next;
 					}
 				}
+#if IR_X86_I64
+				if (other->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+					reg = other->reg_hi;
+					IR_ASSERT(reg >= 0 && reg < IR_REG_NUM);
+					if (IR_REGSET_IN(available, reg)) {
+						IR_REGSET_INCL(overlapped, reg);
+						if (next < freeUntilPos[reg]) {
+							freeUntilPos[reg] = next;
+						}
+					}
+				 }
+#endif
 			}
 		}
 		other = other->list_next;
@@ -2942,13 +3205,42 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 			/* Try to use hint */
 			reg = ir_try_allocate_preferred_reg(ctx, ival, available, freeUntilPos);
 			if (reg != IR_REG_NONE) {
-				ival->reg = reg;
-				IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (hint available without spilling)");
-				if (*unhandled && ival->end > (*unhandled)->range.start) {
-					ival->list_next = *active;
-					*active = ival;
+#if IR_X86_I64
+				if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+					ir_reg reg_hi = IR_REG_I64_HI(reg);
+					reg = IR_REG_I64_LO(reg);
+					if (reg_hi == reg) {
+						IR_REGSET_EXCL(available, reg);
+						if (available == IR_REGSET_EMPTY) {
+							return IR_REG_NONE;
+						}
+						reg_hi = IR_REGSET_FIRST(available);
+						if (reg > reg_hi) {
+							int tmp = reg;
+							reg = reg_hi;
+							reg_hi = tmp;
+						}
+					}
+					IR_ASSERT(reg != reg_hi);
+					ival->reg = reg;
+					ival->reg_hi = reg_hi;
+					IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
+					if (*unhandled && ival->end > (*unhandled)->range.start) {
+						ival->list_next = *active;
+						*active = ival;
+					}
+					return reg;
+				} else
+#endif
+				{
+					ival->reg = reg;
+					IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (hint available without spilling)");
+					if (*unhandled && ival->end > (*unhandled)->range.start) {
+						ival->list_next = *active;
+						*active = ival;
+					}
+					return reg;
 				}
-				return reg;
 			}
 		}

@@ -2956,13 +3248,31 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 			/* Try to reuse the register previously allocated for splited interval */
 			reg = ctx->live_intervals[ival->vreg]->reg;
 			if (reg >= 0 && IR_REGSET_IN(available, reg)) {
-				ival->reg = reg;
-				IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
-				if (*unhandled && ival->end > (*unhandled)->range.start) {
-					ival->list_next = *active;
-					*active = ival;
+#if IR_X86_I64
+				if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+					int8_t reg_hi = ctx->live_intervals[ival->vreg]->reg_hi;
+					if (reg_hi >= 0 && IR_REGSET_IN(available, reg_hi)) {
+						IR_ASSERT(reg < reg_hi);
+						ival->reg = reg;
+						ival->reg_hi = reg_hi;
+						IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
+						if (*unhandled && ival->end > (*unhandled)->range.start) {
+							ival->list_next = *active;
+							*active = ival;
+						}
+						return reg;
+					}
+				} else
+#endif
+				{
+					ival->reg = reg;
+					IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
+					if (*unhandled && ival->end > (*unhandled)->range.start) {
+						ival->list_next = *active;
+						*active = ival;
+					}
+					return reg;
 				}
-				return reg;
 			}
 		}

@@ -2980,9 +3290,23 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 						reg = ir_get_first_reg_hint(ctx, other, non_conflicting);

 						if (reg >= 0) {
-							IR_REGSET_EXCL(non_conflicting, reg);
-							if (non_conflicting == IR_REGSET_EMPTY) {
-								break;
+#if IR_X86_I64
+							if (other->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+								ir_reg reg_hi = IR_REG_I64_HI(reg);
+								ir_reg reg_lo = IR_REG_I64_LO(reg);
+
+								IR_REGSET_EXCL(non_conflicting, reg_hi);
+								IR_REGSET_EXCL(non_conflicting, reg_lo);
+								if (non_conflicting == IR_REGSET_EMPTY) {
+									break;
+								}
+							} else
+#endif
+							{
+								IR_REGSET_EXCL(non_conflicting, reg);
+								if (non_conflicting == IR_REGSET_EMPTY) {
+									break;
+								}
 							}
 						}
 					}
@@ -2999,15 +3323,47 @@ static ir_reg ir_try_allocate_free_reg(ir_ctx *ctx, ir_live_interval *ival, ir_l
 		} else {
 			reg = IR_REGSET_FIRST(available);
 		}
-		ival->reg = reg;
-		IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
-		if (*unhandled && ival->end > (*unhandled)->range.start) {
-			ival->list_next = *active;
-			*active = ival;
+#if IR_X86_I64
+		if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+			IR_REGSET_EXCL(available, reg);
+			if (available != IR_REGSET_EMPTY) {
+				ir_reg reg_hi = IR_REGSET_FIRST(available);
+
+				if (reg > reg_hi) {
+					int tmp = reg;
+					reg = reg_hi;
+					reg_hi = tmp;
+				}
+
+				IR_ASSERT(reg != reg_hi);
+				ival->reg = reg;
+				ival->reg_hi = reg_hi;
+				IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
+				if (*unhandled && ival->end > (*unhandled)->range.start) {
+					ival->list_next = *active;
+					*active = ival;
+				}
+				return reg;
+			}
+		} else
+#endif
+		{
+			ival->reg = reg;
+			IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (available without spilling)");
+			if (*unhandled && ival->end > (*unhandled)->range.start) {
+				ival->list_next = *active;
+				*active = ival;
+			}
+			return reg;
 		}
-		return reg;
 	}

+#if IR_X86_I64
+	if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+		return IR_REG_NONE;
+	}
+#endif
+
 	/* reg = register with highest freeUntilPos */
 	reg = IR_REG_NONE;
 	pos = 0;
@@ -3062,7 +3418,10 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 	ir_live_pos blockPos[IR_REG_NUM];
 	int score, best_score, scores[IR_REG_NUM];
 	int i, reg;
-	ir_live_pos pos, next_use_pos;
+#if IR_X86_I64
+	int reg_hi = IR_REG_NONE;
+#endif
+	ir_live_pos pos, next_use_pos, block_pos;
 	ir_live_interval *other, *prev;
 	ir_use_pos *use_pos;
 	ir_regset available, tmp_regset;
@@ -3083,7 +3442,7 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 		next_use_pos = ival->range.end;
 	}

-	if (IR_IS_TYPE_FP(ival->type)) {
+	if (IR_IS_TYPE_FP(ival->type) || IR_IS_TYPE_VECTOR(ival->type)) {
 		available = IR_REGSET_FP;
 		/* set nextUsePos of all physical registers to maxInt */
 		for (i = IR_REG_FP_FIRST; i <= IR_REG_FP_LAST; i++) {
@@ -3139,11 +3498,26 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 					IR_USE_MUST_BE_IN_REG | IR_USE_SHOULD_BE_IN_REG);
 				if (pos < nextUsePos[reg]) {
 					nextUsePos[reg] = pos;
-						/* Prefer splitting interval that was already splitted before */
+					/* Prefer splitting interval that was already splitted before */
 					scores[reg] = (other->flags & IR_LIVE_INTERVAL_SPLIT_CHILD) ? 1 : 0;
 				}
 			}
 		}
+#if IR_X86_I64
+		if (other->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+			reg = other->reg_hi;
+			IR_ASSERT(reg >= 0 && reg < IR_REG_NUM);
+			if (IR_REGSET_IN(available, reg)) {
+				pos = ir_first_use_pos_after(other, ival->range.start,
+					IR_USE_MUST_BE_IN_REG | IR_USE_SHOULD_BE_IN_REG);
+				if (pos < nextUsePos[reg]) {
+					nextUsePos[reg] = pos;
+					/* Prefer splitting interval that was already splitted before */
+					scores[reg] = (other->flags & IR_LIVE_INTERVAL_SPLIT_CHILD) ? 1 : 0;
+				}
+			}
+		 }
+#endif
 		other = other->list_next;
 	}

@@ -3191,6 +3565,35 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 				}
 			}
 		}
+#if IR_X86_I64
+		if (other->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+			reg = other->reg_hi;
+			IR_ASSERT(reg >= 0 && reg < IR_REG_NUM);
+			if (IR_REGSET_IN(available, reg)) {
+				ir_live_pos overlap = ir_ivals_overlap(&ival->range, other->current_range);
+
+				if (overlap) {
+					if (other->flags & (IR_LIVE_INTERVAL_FIXED|IR_LIVE_INTERVAL_TEMP)) {
+						if (overlap < nextUsePos[reg]) {
+							nextUsePos[reg] = overlap;
+							scores[reg] = 0;
+						}
+						if (overlap < blockPos[reg]) {
+							blockPos[reg] = overlap;
+						}
+					} else {
+						pos = ir_first_use_pos_after(other, ival->range.start,
+							IR_USE_MUST_BE_IN_REG | IR_USE_SHOULD_BE_IN_REG);
+						if (pos < nextUsePos[reg]) {
+							nextUsePos[reg] = pos;
+							/* Prefer splitting interval that was already splitted before */
+							scores[reg] = (other->flags & IR_LIVE_INTERVAL_SPLIT_CHILD) ? 1 : 0;
+						}
+					}
+				}
+			}
+		 }
+#endif
 		other = other->list_next;
 	}

@@ -3198,10 +3601,22 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 	reg = IR_REG_NONE;
 	if (ival->flags & (IR_LIVE_INTERVAL_HAS_HINT_REGS|IR_LIVE_INTERVAL_HAS_HINT_REFS)) {
 		reg = ir_get_preferred_reg(ctx, ival, available);
+#if IR_X86_I64
+		if (reg != IR_REG_NONE && ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+			reg_hi = IR_REG_I64_HI(reg);
+			reg = IR_REG_I64_LO(reg);
+			if (reg == reg_hi) {
+				reg_hi = IR_REG_NONE;
+			}
+		}
+#endif
 	}
 	if (reg == IR_REG_NONE) {
 select_register:
 		reg = IR_REGSET_FIRST(available);
+#if IR_X86_I64
+		reg_hi = IR_REG_NONE;
+#endif
 	}

 	/* reg = register with highest nextUsePos */
@@ -3220,6 +3635,48 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 		}
 	} IR_REGSET_FOREACH_END();

+	block_pos = blockPos[reg];
+
+#if IR_X86_I64
+	if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+		ir_live_pos pos_hi;
+
+		tmp_regset = available;
+		IR_REGSET_EXCL(tmp_regset, reg);
+
+		if (IR_REGSET_IS_EMPTY(tmp_regset)) {
+			fprintf(stderr, "LSRA Internal Error: Unsolvable conflict. Allocation is not possible\n");
+			IR_ASSERT(0);
+			exit(-1);
+		}
+
+		if (reg_hi == IR_REG_NONE || !IR_REGSET_IN(tmp_regset, reg_hi)) {
+			reg_hi = IR_REGSET_FIRST(tmp_regset);
+		}
+		pos_hi = nextUsePos[reg_hi];
+		best_score = (scores[reg_hi] << 28) + nextUsePos[reg_hi];
+		IR_REGSET_EXCL(tmp_regset, reg_hi);
+		IR_REGSET_FOREACH(tmp_regset, i) {
+			if (nextUsePos[i] > pos_hi) {
+				pos_hi = nextUsePos[i];
+			}
+			score = (scores[i] << 28) + nextUsePos[i];
+			if (score > best_score) {
+				reg_hi = i;
+				best_score = score;
+			}
+		} IR_REGSET_FOREACH_END();
+
+		pos = IR_MIN(pos, pos_hi);
+		block_pos = IR_MIN(block_pos, blockPos[reg_hi]);
+		if (reg > reg_hi) {
+			int tmp = reg;
+			reg = reg_hi;
+			reg_hi = tmp;
+		}
+	}
+#endif
+
 	/* if first usage of current is after nextUsePos[reg] then */
 	if (next_use_pos > pos && !(ival->flags & IR_LIVE_INTERVAL_TEMP)) {
 		/* all other intervals are used before current, so it is best to spill current itself */
@@ -3245,23 +3702,26 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 		}
 	}

-	if (ival->end > blockPos[reg]) {
+	if (ival->end > block_pos) {
 		/* spilling make a register free only for the first part of current */
 		IR_LOG_LSRA("    ---- Conflict with others", ival, " (spilling make a register free only for the first part)");
 		/* split current at optimal position before block_pos[reg] */
-		ir_live_pos split_pos = ir_last_use_pos_before(ival,  blockPos[reg] + 1,
+		ir_live_pos split_pos = ir_last_use_pos_before(ival,  block_pos + 1,
 			IR_USE_MUST_BE_IN_REG | IR_USE_SHOULD_BE_IN_REG);
 		if (split_pos == 0) {
-			split_pos = ir_first_use_pos_after(ival, blockPos[reg],
+			split_pos = ir_first_use_pos_after(ival, block_pos,
 				IR_USE_MUST_BE_IN_REG | IR_USE_SHOULD_BE_IN_REG) - 1;
 			other = ir_split_interval_at(ctx, ival, split_pos);
 			ir_add_to_unhandled(unhandled, other);
 			IR_LOG_LSRA("      ---- Queue", other, "");
 			return IR_REG_NONE;
 		}
-		if (split_pos >= blockPos[reg]) {
+		if (split_pos >= block_pos) {
 try_next_available_register:
 			IR_REGSET_EXCL(available, reg);
+#if IR_X86_I64
+			if (reg_hi != IR_REG_NONE) IR_REGSET_EXCL(available, reg_hi);
+#endif
 			if (IR_REGSET_IS_EMPTY(available)) {
 				fprintf(stderr, "LSRA Internal Error: Unsolvable conflict. Allocation is not possible\n");
 				IR_ASSERT(0);
@@ -3270,7 +3730,7 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 			IR_LOG_LSRA("      ---- Restart", ival, "");
 			goto select_register;
 		}
-		split_pos = ir_find_optimal_split_position(ctx, ival, split_pos, blockPos[reg], 1);
+		split_pos = ir_find_optimal_split_position(ctx, ival, split_pos, block_pos, 1);
 		other = ir_split_interval_at(ctx, ival, split_pos);
 		ir_add_to_unhandled(unhandled, other);
 		IR_LOG_LSRA("      ---- Queue", other, "");
@@ -3282,7 +3742,12 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 	while (other) {
 		ir_live_pos split_pos;

+#if IR_X86_I64
+		if (reg == other->reg || reg == other->reg_hi
+		 || (reg_hi != IR_REG_NONE && (reg_hi == other->reg || reg_hi == other->reg_hi))) {
+#else
 		if (reg == other->reg) {
+#endif
 			/* split active interval for reg at position */
 			ir_live_pos overlap = ir_ivals_overlap(&ival->range, other->current_range);

@@ -3314,6 +3779,9 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 				if (split_pos > child->range.start && split_pos < child->end) {
 					if (child == other) {
 						other->reg = IR_REG_NONE;
+#if IR_X86_I64
+						other->reg_hi = IR_REG_NONE;
+#endif
 						if (prev) {
 							prev->list_next = other->list_next;
 						} else {
@@ -3336,7 +3804,12 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 					goto try_next_available_register;
 				}
 			}
+#if IR_X86_I64
+			other = other->list_next;
+			continue;
+#else
 			break;
+#endif
 		}
 		prev = other;
 		other = other->list_next;
@@ -3346,7 +3819,12 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 	other = *inactive;
 	while (other) {
 		/* freeUntilPos[it.reg] = next intersection of it with current */
+#if IR_X86_I64
+		if (reg == other->reg || reg == other->reg_hi
+		 || (reg_hi != IR_REG_NONE && (reg_hi == other->reg || reg_hi == other->reg_hi))) {
+#else
 		if (reg == other->reg) {
+#endif
 			ir_live_pos overlap = ir_ivals_overlap(&ival->range, other->current_range);

 			if (overlap) {
@@ -3367,6 +3845,9 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li

 	/* current.reg = reg */
 	ival->reg = reg;
+#if IR_X86_I64
+	ival->reg_hi = reg_hi;
+#endif
 	IR_LOG_LSRA_ASSIGN("    ---- Assign", ival, " (after splitting others)");

 	if (*unhandled && ival->end > (*unhandled)->range.start) {
@@ -3376,46 +3857,6 @@ static ir_reg ir_allocate_blocked_reg(ir_ctx *ctx, ir_live_interval *ival, ir_li
 	return reg;
 }

-static int ir_fix_dessa_tmps(ir_ctx *ctx, uint8_t type, ir_ref from, ir_ref to, void *data)
-{
-	ir_block *bb = data;
-	ir_tmp_reg tmp_reg;
-
-	if (to == 0) {
-		if (IR_IS_TYPE_INT(type)) {
-			tmp_reg.num = 0;
-			tmp_reg.type = type;
-			tmp_reg.start = IR_USE_SUB_REF;
-			tmp_reg.end = IR_SAVE_SUB_REF;
-		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
-			tmp_reg.num = 1;
-			tmp_reg.type = type;
-			tmp_reg.start = IR_USE_SUB_REF;
-			tmp_reg.end = IR_SAVE_SUB_REF;
-		}
-	} else if (from != 0) {
-		if (IR_IS_TYPE_INT(type)) {
-			tmp_reg.num = 0;
-			tmp_reg.type = type;
-			tmp_reg.start = IR_USE_SUB_REF;
-			tmp_reg.end = IR_SAVE_SUB_REF;
-		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
-			tmp_reg.num = 1;
-			tmp_reg.type = type;
-			tmp_reg.start = IR_USE_SUB_REF;
-			tmp_reg.end = IR_SAVE_SUB_REF;
-		}
-	} else {
-		return 1;
-	}
-	if (!ir_has_tmp(ctx, bb->end, tmp_reg.num)) {
-		ir_add_tmp(ctx, bb->end, bb->end, tmp_reg.num, tmp_reg);
-	}
-	return 1;
-}
-
 static bool ir_ival_spill_for_fuse_load(ir_ctx *ctx, ir_live_interval *ival)
 {
 	ir_use_pos *use_pos = ival->use_pos;
@@ -3486,10 +3927,25 @@ static int ir_linear_scan(ir_ctx *ctx, ir_ref vars)

 	if (ctx->flags2 & IR_LR_HAVE_DESSA_MOVES) {
 		/* Add fixed intervals for temporary registers used for DESSA moves */
-		for (b = 1, bb = &ctx->cfg_blocks[1]; b <= ctx->cfg_blocks_count; b++, bb++) {
+		for (b = ctx->cfg_blocks_count, bb = &ctx->cfg_blocks[b]; b > 0; b--, bb--) {
 			IR_ASSERT(!(bb->flags & IR_BB_UNREACHABLE));
 			if (bb->flags & IR_BB_DESSA_MOVES) {
-				ir_gen_dessa_moves(ctx, b, ir_fix_dessa_tmps, bb);
+				ir_tmp_reg tmp_reg;
+
+				if (bb->flags & IR_BB_DESSA_TMP_INT) {
+					tmp_reg.num = 0;
+					tmp_reg.type = IR_U32; // ???
+					tmp_reg.start = IR_USE_SUB_REF;
+					tmp_reg.end = IR_SAVE_SUB_REF;
+					ir_add_tmp(ctx, bb->end, bb->end, tmp_reg.num, tmp_reg);
+				}
+				if (bb->flags & IR_BB_DESSA_TMP_FP) {
+					tmp_reg.num = 1;
+					tmp_reg.type = IR_DOUBLE; // ???
+					tmp_reg.start = IR_USE_SUB_REF;
+					tmp_reg.end = IR_SAVE_SUB_REF;
+					ir_add_tmp(ctx, bb->end, bb->end, tmp_reg.num, tmp_reg);
+				}
 			}
 		}
 	}
@@ -3698,10 +4154,8 @@ static int ir_linear_scan(ir_ctx *ctx, ir_ref vars)
 				}
 			}
 		}
-
 		if (unhandled) {
-			uint8_t size;
-			ir_live_interval *handled[9] = {NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL};
+			ir_live_interval *handled[7] = {NULL, NULL, NULL, NULL, NULL, NULL, NULL};
 			ir_live_interval *old;

 			((ir_reg_alloc_data*)(ctx->data))->handled = handled;
@@ -3723,9 +4177,12 @@ static int ir_linear_scan(ir_ctx *ctx, ir_ref vars)
 						} else {
 							active = other->list_next;
 						}
-						size = ir_type_size[other->type];
-						IR_ASSERT(size == 1 || size == 2 || size == 4 || size == 8);
-						old = handled[size];
+
+						uint8_t n, size = ir_get_type_size(other->type);
+
+						IR_ASSERT(size == 1 || size == 2 || size == 4 || size == 8 || size == 16 || size == 32 || size == 64);
+						n = ir_ntz(size);
+						old = handled[n];
 						while (old) {
 							if (old->stack_spill_pos == other->stack_spill_pos) {
 								break;
@@ -3733,8 +4190,8 @@ static int ir_linear_scan(ir_ctx *ctx, ir_ref vars)
 							old = old->list_next;
 						}
 						if (!old) {
-							other->list_next = handled[size];
-							handled[size] = other;
+							other->list_next = handled[n];
+							handled[n] = other;
 						}
 					} else {
 						prev = other;
@@ -3747,9 +4204,11 @@ static int ir_linear_scan(ir_ctx *ctx, ir_ref vars)
 					ival->list_next = active;
 					active = ival;
 				} else {
-					size = ir_type_size[ival->type];
-					IR_ASSERT(size == 1 || size == 2 || size == 4 || size == 8);
-					old = handled[size];
+					uint32_t n, size = ir_get_type_size(ival->type);
+
+					IR_ASSERT(size == 1 || size == 2 || size == 4 || size == 8 || size == 16 || size == 32 || size == 64);
+					n = ir_ntz(size);
+					old = handled[n];
 					while (old) {
 						if (old->stack_spill_pos == ival->stack_spill_pos) {
 							break;
@@ -3757,8 +4216,8 @@ static int ir_linear_scan(ir_ctx *ctx, ir_ref vars)
 						old = old->list_next;
 					}
 					if (!old) {
-						ival->list_next = handled[size];
-						handled[size] = ival;
+						ival->list_next = handled[n];
+						handled[n] = ival;
 					}
 				}
 			}
@@ -3867,6 +4326,13 @@ static void assign_regs(ir_ctx *ctx)
 					if (ival->reg != IR_REG_NONE) {
 						reg = ival->reg;
 						IR_REGSET_INCL(used_regs, reg);
+#if IR_X86_I64
+						if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+							IR_ASSERT(ival->reg_hi != IR_REG_NONE);
+							IR_REGSET_INCL(used_regs, ival->reg_hi);
+							reg = IR_REG_I64_PAIR(reg, ival->reg_hi);
+						}
+#endif
 						use_pos = ival->use_pos;
 						while (use_pos) {
 							ref = (use_pos->hint_ref < 0) ? -use_pos->hint_ref : IR_LIVE_POS_TO_REF(use_pos->pos);
@@ -3887,10 +4353,17 @@ static void assign_regs(ir_ctx *ctx)
 				if (!(ival->flags & IR_LIVE_INTERVAL_SPILLED)) {
 					do {
 						if (ival->reg != IR_REG_NONE) {
-							IR_REGSET_INCL(used_regs, ival->reg);
+							reg = ival->reg;
+							IR_REGSET_INCL(used_regs, reg);
+#if IR_X86_I64
+							if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+								IR_ASSERT(ival->reg_hi != IR_REG_NONE);
+								IR_REGSET_INCL(used_regs, ival->reg_hi);
+								reg = IR_REG_I64_PAIR(reg, ival->reg_hi);
+							}
+#endif
 							use_pos = ival->use_pos;
 							while (use_pos) {
-								reg = ival->reg;
 								ref = IR_LIVE_POS_TO_REF(use_pos->pos);
 								if (use_pos->hint_ref < 0) {
 									ref = -use_pos->hint_ref;
@@ -3906,12 +4379,21 @@ static void assign_regs(ir_ctx *ctx)
 					do {
 						if (ival->reg != IR_REG_NONE) {
 							ir_ref prev_use_ref = IR_UNUSED;
+							int8_t reg0;

 							ir_bitset_clear(available, ir_bitset_len(ctx->cfg_blocks_count + 1));
-							IR_REGSET_INCL(used_regs, ival->reg);
+							reg0 = ival->reg;
+							IR_REGSET_INCL(used_regs, reg0);
+#if IR_X86_I64
+							if (ival->flags & IR_LIVE_INTERVAL_TWO_REGS) {
+								IR_ASSERT(ival->reg_hi != IR_REG_NONE);
+								IR_REGSET_INCL(used_regs, ival->reg_hi);
+								reg0 = IR_REG_I64_PAIR(reg0, ival->reg_hi);
+							}
+#endif
 							use_pos = ival->use_pos;
 							while (use_pos) {
-								reg = ival->reg;
+								reg = reg0;
 								ref = IR_LIVE_POS_TO_REF(use_pos->pos);
 								// TODO: Insert spill loads and stores in optimal positions (resolution)
 								if (use_pos->op_num == 0) {
@@ -3935,6 +4417,13 @@ static void assign_regs(ir_ctx *ctx)
 									 && (ival->flags & IR_LIVE_INTERVAL_MEM_PARAM)) {
 										/* Stack PARAM var is passed through memory */
 										reg = IR_REG_NONE;
+#if defined(IR_TARGET_X86) || defined(IR_TARGET_X64)
+										if (use_pos->next
+										 && ctx->ir_base[IR_LIVE_POS_TO_REF(use_pos->next->pos)].op == IR_VSTORE) {
+											/* skip VSTORE (VAR is going to be remapped to PARAM on x86) */
+											use_pos = use_pos->next;
+										}
+#endif
 									} else {
 										uint32_t use_b = ctx->cfg_map[ref];

@@ -3952,7 +4441,6 @@ static void assign_regs(ir_ctx *ctx)
 									if ((!prev_use_ref || ctx->cfg_map[prev_use_ref] != ctx->cfg_map[ref])
 									 && needs_spill_reload(ctx, ival, ctx->cfg_map[ref], available)) {
 										if (!(use_pos->flags & IR_USE_MUST_BE_IN_REG)
-										 && use_pos->hint != reg
 //										 && ctx->ir_base[ref].op != IR_CALL
 //										 && ctx->ir_base[ref].op != IR_TAILCALL) {
 										 && ctx->ir_base[ref].op != IR_SNAPSHOT
@@ -4057,14 +4545,13 @@ static void assign_regs(ir_ctx *ctx)
 					if (IR_IS_CONST_REF(ops[ival->tmp_op_num])) {
 						/* constant rematerialization */
 						reg |= IR_REG_SPILL_LOAD;
-					} else if (ctx->ir_base[ops[ival->tmp_op_num]].op == IR_ALLOCA
-							|| ctx->ir_base[ops[ival->tmp_op_num]].op == IR_VADDR) {
+					} else if (ctx->rules[ops[ival->tmp_op_num]] == (IR_SKIPPED|IR_FUSED|IR_SIMPLE|IR_ALLOCA)) {
 						/* local address rematerialization */
 						reg |= IR_REG_SPILL_LOAD;
 					}
 				}
 			}
-			ir_set_alocated_reg(ctx, ival->tmp_ref, ival->tmp_op_num, reg);
+			ir_set_alocated_tmp_reg(ctx, ival->tmp_ref, ival->tmp_op_num, reg);
 			ival = ival->next;
 		} while (ival);
 	}
diff --git a/ext/opcache/jit/ir/ir_save.c b/ext/opcache/jit/ir/ir_save.c
index 8b3f3b5c6b5..1286f550385 100644
--- a/ext/opcache/jit/ir/ir_save.c
+++ b/ext/opcache/jit/ir/ir_save.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (IR saver)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -50,15 +50,25 @@ static void ir_print_call_conv(uint32_t flags, FILE *f)
 	}
 }

+void ir_print_type_cname(ir_type type, FILE *f)
+{
+	if (IR_IS_TYPE_VECTOR(type)) {
+		fprintf(f, "<%s*%d>", ir_type_cname[IR_VECTOR_BASE_TYPE(type)], IR_VECTOR_LENGTH(type));
+	} else {
+		fprintf(f, "%s", ir_type_cname[type]);
+	}
+}
+
 void ir_print_proto_ex(uint8_t flags, ir_type ret_type, uint32_t params_count, const uint8_t *param_types, FILE *f)
 {
 	uint32_t j;

 	fprintf(f, "(");
 	if (params_count > 0) {
-		fprintf(f, "%s", ir_type_cname[param_types[0]]);
+		ir_print_type_cname(param_types[0], f);
 		for (j = 1; j < params_count; j++) {
-			fprintf(f, ", %s", ir_type_cname[param_types[j]]);
+			fprintf(f, ", ");
+			ir_print_type_cname(param_types[j], f);
 		}
 		if (flags & IR_VARARG_FUNC) {
 			fprintf(f, ", ...");
@@ -66,7 +76,8 @@ void ir_print_proto_ex(uint8_t flags, ir_type ret_type, uint32_t params_count, c
 	} else if (flags & IR_VARARG_FUNC) {
 		fprintf(f, "...");
 	}
-	fprintf(f, "): %s", ir_type_cname[ret_type]);
+	fprintf(f, "): ");
+	ir_print_type_cname(ret_type, f);
 	ir_print_call_conv(flags, f);
 	if (flags & IR_CONST_FUNC) {
 		fprintf(f, " __const");
@@ -86,10 +97,11 @@ void ir_print_func_proto(const ir_ctx *ctx, const char *name, bool prefix, FILE
 	if (ctx->ir_base[2].op == IR_PARAM) {
 		ir_insn *insn = &ctx->ir_base[2];

-		fprintf(f, "%s", ir_type_cname[insn->type]);
+		ir_print_type_cname(insn->type, f);
 		insn++;
 		while (insn->op == IR_PARAM) {
-			fprintf(f, ", %s", ir_type_cname[insn->type]);
+			fprintf(f, ", ");
+			ir_print_type_cname(insn->type, f);
 			insn++;;
 		}
 		if (ctx->flags & IR_VARARG_FUNC) {
@@ -98,7 +110,8 @@ void ir_print_func_proto(const ir_ctx *ctx, const char *name, bool prefix, FILE
 	} else if (ctx->flags & IR_VARARG_FUNC) {
 		fprintf(f, "...");
 	}
-	fprintf(f, "): %s", ir_type_cname[ctx->ret_type != (ir_type)-1 ? ctx->ret_type : IR_VOID]);
+	fprintf(f, "): ");
+	ir_print_type_cname(ctx->ret_type != (ir_type)-1 ? ctx->ret_type : IR_VOID, f);
 	ir_print_call_conv(ctx->flags, f);
 	if (ctx->flags & IR_CONST_FUNC) {
 		fprintf(f, " __const");
@@ -138,16 +151,14 @@ static void ir_save_dessa_moves(const ir_ctx *ctx, int b, ir_block *bb, FILE *f)
 				int8_t *regs = ctx->regs[use_ref];
 				int8_t reg = regs[k];
 				if (reg != IR_REG_NONE) {
-					fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[input].type),
-						(reg & (IR_REG_SPILL_LOAD|IR_REG_SPILL_SPECIAL)) ? ":load" : "");
+					ir_dump_reg(ctx, reg, input, 0, f);
 				}
 			}
 			fprintf(f, " -> d_%d {R%d}", use_ref, ctx->vregs[use_ref]);
 			if (ctx->regs) {
 				int8_t reg = ctx->regs[use_ref][0];
 				if (reg != IR_REG_NONE) {
-					fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[use_ref].type),
-						(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? ":store" : "");
+					ir_dump_reg(ctx, reg, use_ref, 1, f);
 				}
 			}
 			fprintf(f, "\n");
@@ -163,29 +174,65 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 	bool first;

 	fprintf(f, "{\n");
-	for (i = IR_UNUSED + 1, insn = ctx->ir_base - i; i < ctx->consts_count; i++, insn--) {
-		fprintf(f, "\t%s c_%d = ", ir_type_cname[insn->type], i);
-		if (insn->op == IR_FUNC) {
-			fprintf(f, "func %s%s",
-				(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
-				ir_get_str(ctx, insn->val.name));
-			ir_print_proto(ctx, insn->proto, f);
-		} else if (insn->op == IR_SYM) {
-			fprintf(f, "sym(%s%s)",
-				(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
-				ir_get_str(ctx, insn->val.name));
-		} else if (insn->op == IR_LABEL) {
-			fprintf(f, "label(%s%s)",
-				(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
-				ir_get_str(ctx, insn->val.name));
-		} else if (insn->op == IR_FUNC_ADDR) {
-			fprintf(f, "func *");
-			ir_print_const(ctx, insn, f, true);
-			ir_print_proto(ctx, insn->proto, f);
-		} else {
-			ir_print_const(ctx, insn, f, true);
+	/* Separate behavior to keep tests compatibility. TODO: remove the old behavior */
+	if (ctx->flags2 & IR_HAS_LONG_CONSTANTS) {
+		for (i = 1 - ctx->consts_count, insn = ctx->ir_base + i; i < IR_UNUSED; i++, insn++) {
+			fprintf(f, "\t");
+			ir_print_type_cname(insn->type, f);
+			fprintf(f, " c_%d = ", -i);
+			if (insn->op == IR_FUNC) {
+				fprintf(f, "func %s%s",
+					(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
+					ir_get_str(ctx, insn->val.name));
+				ir_print_proto(ctx, insn->proto, f);
+			} else if (insn->op == IR_SYM) {
+				fprintf(f, "sym(%s%s)",
+					(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
+					ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_LABEL) {
+				fprintf(f, "label(%s%s)",
+					(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
+					ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_FUNC_ADDR) {
+				fprintf(f, "func *");
+				ir_print_const(ctx, insn, f, true);
+				ir_print_proto(ctx, insn->proto, f);
+			} else {
+				ir_print_const(ctx, insn, f, true);
+			}
+			fprintf(f, ";\n");
+			if (insn->op == IR_LONG_CONST) {
+				i += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+				insn += IR_ALIGNED_SIZE(insn->long_const_size, sizeof(ir_insn)) / sizeof(ir_insn);
+			}
+		}
+	} else {
+		for (i = IR_UNUSED + 1, insn = ctx->ir_base - i; i < ctx->consts_count; i++, insn--) {
+			fprintf(f, "\t");
+			ir_print_type_cname(insn->type, f);
+			fprintf(f, " c_%d = ", i);
+			if (insn->op == IR_FUNC) {
+				fprintf(f, "func %s%s",
+					(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
+					ir_get_str(ctx, insn->val.name));
+				ir_print_proto(ctx, insn->proto, f);
+			} else if (insn->op == IR_SYM) {
+				fprintf(f, "sym(%s%s)",
+					(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
+					ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_LABEL) {
+				fprintf(f, "label(%s%s)",
+					(save_flags & IR_SAVE_SAFE_NAMES) ? "@" : "",
+					ir_get_str(ctx, insn->val.name));
+			} else if (insn->op == IR_FUNC_ADDR) {
+				fprintf(f, "func *");
+				ir_print_const(ctx, insn, f, true);
+				ir_print_proto(ctx, insn->proto, f);
+			} else {
+				ir_print_const(ctx, insn, f, true);
+			}
+			fprintf(f, ";\n");
 		}
-		fprintf(f, ";\n");
 	}

 	for (i = IR_UNUSED + 1, insn = ctx->ir_base + i; i < ctx->insns_count;) {
@@ -220,6 +267,9 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 			if (bb->flags & IR_BB_IRREDUCIBLE_LOOP) {
 				fprintf(f, ", IRREDUCIBLE");
 			}
+			if (bb->flags & IR_BB_IRREDUCIBLE_ENTRY) {
+				fprintf(f, ", IRREDUCIBLE_ENTRY");
+			}
 			if (bb->predecessors_count) {
 				uint32_t i;

@@ -245,7 +295,9 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 			if (!(flags & IR_OP_FLAG_MEM) || insn->type == IR_VOID) {
 				fprintf(f, "\tl_%d = ", i);
 			} else {
-				fprintf(f, "\t%s d_%d", ir_type_cname[insn->type], i);
+				fprintf(f, "\t");
+				ir_print_type_cname(insn->type, f);
+				fprintf(f, " d_%d", i);
 				if (save_flags & IR_SAVE_REGS) {
 					if (ctx->vregs && ctx->vregs[i]) {
 						fprintf(f, " {R%d}", ctx->vregs[i]);
@@ -253,8 +305,7 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 					if (ctx->regs) {
 						int8_t reg = ctx->regs[i][0];
 						if (reg != IR_REG_NONE) {
-							fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), insn->type),
-								(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? ":store" : "");
+							ir_dump_reg(ctx, reg, i, 1, f);
 						}
 					}
 				}
@@ -263,7 +314,8 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 		} else {
 			fprintf(f, "\t");
 			if (flags & IR_OP_FLAG_DATA) {
-				fprintf(f, "%s d_%d", ir_type_cname[insn->type], i);
+				ir_print_type_cname(insn->type, f);
+				fprintf(f, " d_%d", i);
 				if (save_flags & IR_SAVE_REGS) {
 					if (ctx->vregs && ctx->vregs[i]) {
 						fprintf(f, " {R%d}", ctx->vregs[i]);
@@ -271,8 +323,7 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 					if (ctx->regs) {
 						int8_t reg = ctx->regs[i][0];
 						if (reg != IR_REG_NONE) {
-							fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), insn->type),
-								(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? ":store" : "");
+							ir_dump_reg(ctx, reg, i, 1, f);
 						}
 					}
 				}
@@ -311,8 +362,7 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 								int8_t *regs = ctx->regs[i];
 								int8_t reg = regs[j];
 								if (reg != IR_REG_NONE) {
-									fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[ref].type),
-										(reg & (IR_REG_SPILL_LOAD|IR_REG_SPILL_SPECIAL)) ? ":load" : "");
+									ir_dump_reg(ctx, reg, ref, 0, f);
 								}
 							}
 						}
@@ -416,6 +466,11 @@ void ir_save(const ir_ctx *ctx, uint32_t save_flags, FILE *f)
 			if (rule & IR_SIMPLE) {
 				fprintf(f, ":SIMPLE");
 			}
+#if IR_X86_I64
+			if (rule & IR_TWO_REGS) {
+				fprintf(f, ":TWO_REGS");
+			}
+#endif
 			fprintf(f, ");");
 		}

diff --git a/ext/opcache/jit/ir/ir_sccp.c b/ext/opcache/jit/ir/ir_sccp.c
index f2b8616e2af..5d37f42734d 100644
--- a/ext/opcache/jit/ir/ir_sccp.c
+++ b/ext/opcache/jit/ir/ir_sccp.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (SCCP - Sparse Conditional Constant Propagation)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  *
  * The SCCP algorithm is based on M. N. Wegman and F. K. Zadeck publication
@@ -109,6 +109,8 @@ IR_ALWAYS_INLINE ir_ref ir_sccp_identity(const ir_ctx *ctx, const ir_sccp_val *_
 			IR_ASSERT(a > 0);
 		} while (_values[a].op == IR_COPY);
 		IR_ASSERT(_values[a].op == IR_BOTTOM);
+	} else if (a > 0 && _values[a].op == IR_LONG_CONST) {
+		a = _values[a].val.i32;
 	}
 	return a;
 }
@@ -226,8 +228,34 @@ IR_ALWAYS_INLINE void ir_sccp_make_bottom_ex(const ir_ctx *ctx, ir_sccp_val *_va
 # define IR_MAKE_BOTTOM_EX(ref) IR_MAKE_BOTTOM(ref)
 #endif

+IR_ALWAYS_INLINE bool ir_sccp_meet_long_const(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitqueue *worklist, ir_ref ref, ir_ref const_ref, const ir_insn *const_insn)
+{
+	if (_values[ref].op == IR_TOP) {
+		/* TOP meet NEW_CONST => NEW_CONST */
+		_values[ref].optx = const_insn->opt;
+		_values[ref].val.i64 = const_ref;
+		return 1;
+	} else if (_values[ref].opt == const_insn->opt) {
+		/* OLD_CONST meet NEW_CONST => (OLD_CONST == NEW_CONST) ? OLD_CONST : BOTTOM */
+		if (_values[ref].val.i32 == const_ref) {
+			return 0;
+		}
+	}
+
+	IR_MAKE_BOTTOM_EX(ref);
+	return 1;
+}
+
 IR_ALWAYS_INLINE bool ir_sccp_meet_const(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitqueue *worklist, ir_ref ref, const ir_insn *val_insn)
 {
+#if IR_COMBO_COPY_PROPAGATION
+	IR_ASSERT(IR_IS_TYPE_SCALAR(val_insn->type));
+#else
+	if (!IR_IS_TYPE_SCALAR(val_insn->type)) {
+		IR_MAKE_BOTTOM_EX(ref);
+		return 1;
+	}
+#endif
 	IR_ASSERT(IR_IS_CONST_OP(val_insn->op) || IR_IS_SYM_CONST(val_insn->op));

 	if (_values[ref].op == IR_TOP) {
@@ -318,6 +346,9 @@ static ir_ref ir_sccp_fold(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitqueue
 			copy = ctx->fold_insn.op1;
 			if (IR_IS_CONST_REF(copy)) {
 				insn = &ctx->ir_base[copy];
+				if (insn->op == IR_LONG_CONST) {
+					return ir_sccp_meet_long_const(ctx, _values, worklist, ref, copy, insn);
+				}
 			} else {
 				insn = &_values[copy].insn;
 				if (!IR_IS_CONST_OP(insn->op) && !IR_IS_SYM_CONST(insn->op)) {
@@ -367,6 +398,12 @@ static bool ir_sccp_analyze_phi(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitq
 		input = *p;
 		if (IR_IS_CONST_REF(input)) {
 			v = &ctx->ir_base[input];
+#if IR_COMBO_COPY_PROPAGATION
+			if (v->op == IR_LONG_CONST) {
+				new_copy = input;
+				goto next;
+			}
+#endif
 		} else if (input == i) {
 			continue;
 		} else {
@@ -384,6 +421,9 @@ static bool ir_sccp_analyze_phi(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitq
 				}
 				new_copy = input;
 				goto next;
+			} else if (v->op == IR_LONG_CONST) {
+				new_copy = v->val.i32;
+				goto next;
 #endif
 			} else if (v->op == IR_BOTTOM) {
 #if IR_COMBO_COPY_PROPAGATION
@@ -417,6 +457,9 @@ static bool ir_sccp_analyze_phi(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitq
 		if (IR_IS_CONST_REF(input)) {
 #if IR_COMBO_COPY_PROPAGATION
 			if (new_copy) {
+				if (new_copy == input) {
+					continue;
+				}
 				goto make_bottom;
 			}
 #endif
@@ -436,6 +479,11 @@ static bool ir_sccp_analyze_phi(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitq
 					continue;
 				}
 				goto make_bottom;
+			} else if (v->op == IR_LONG_CONST) {
+				if (v->val.i32 == new_copy) {
+					continue;
+				}
+				goto make_bottom;
 #endif
 			} else if (v->op == IR_BOTTOM) {
 #if IR_COMBO_COPY_PROPAGATION
@@ -453,6 +501,10 @@ static bool ir_sccp_analyze_phi(const ir_ctx *ctx, ir_sccp_val *_values, ir_bitq

 #if IR_COMBO_COPY_PROPAGATION
 	if (new_copy) {
+		if (IR_IS_CONST_REF(new_copy)) {
+			IR_ASSERT(ctx->ir_base[new_copy].op == IR_LONG_CONST);
+			return ir_sccp_meet_long_const(ctx, _values, worklist, i, new_copy, &ctx->ir_base[new_copy]);
+		}
 		IR_ASSERT(!IR_IS_CONST_REF(new_copy));
 		IR_ASSERT(!IR_IS_CONST_OP(_values[new_copy].op) && !IR_IS_SYM_CONST(_values[new_copy].op));
 		return ir_sccp_meet_copy(ctx, _values, worklist, i, new_copy);
@@ -835,6 +887,8 @@ static IR_NEVER_INLINE void ir_sccp_analyze(const ir_ctx *ctx, ir_sccp_val *_val
 #if IR_COMBO_COPY_PROPAGATION
 			} else if (_values[i].op == IR_COPY) {
 				fprintf(stderr, "%d. COPY(%d)\n", i, _values[i].copy);
+			} else if (_values[i].op == IR_LONG_CONST) {
+				fprintf(stderr, "%d. LONG_CONST(%d)\n", i, _values[i].val.i32);
 #endif
 			} else if (IR_IS_TOP(i)) {
 				if (ctx->ir_base[i].op != IR_TOP) {
@@ -1084,7 +1138,7 @@ static bool ir_sccp_remove_unfeasible_merge_inputs(ir_ctx *ctx, ir_ref ref, ir_i
 					/* remove PHI */
 #if 0
 					use_insn->op1 = IR_UNUSED;
-					ir_iter_remove_insn(ctx, use, worklist);
+					ir_iter_remove_insn(ctx, use);
 #else
 					IR_ASSERT(0);
 #endif
@@ -1138,6 +1192,8 @@ static IR_NEVER_INLINE void ir_sccp_transform(ir_ctx *ctx, const ir_sccp_val *_v
 #if IR_COMBO_COPY_PROPAGATION
 		} else if (value->op == IR_COPY) {
 			ir_sccp_replace_insn(ctx, _values, i, ir_sccp_identity(ctx, _values, value->copy), iter_worklist);
+		} else if (value->op == IR_LONG_CONST) {
+			ir_sccp_replace_insn(ctx, _values, i, value->val.i32, iter_worklist);
 #endif
 		} else if (value->op == IR_TOP) {
 			/* remove unreachable instruction */
@@ -1225,7 +1281,7 @@ static void ir_iter_add_related_uses(const ir_ctx *ctx, ir_ref ref, ir_bitqueue
 	}
 }

-static void ir_iter_remove_insn(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
+static void ir_iter_remove_insn(ir_ctx *ctx, ir_ref ref)
 {
 	ir_ref j, n, *p;
 	ir_insn *insn;
@@ -1241,10 +1297,10 @@ static void ir_iter_remove_insn(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
 			ir_use_list_remove_one(ctx, input, ref);
 			if (ir_is_dead(ctx, input)) {
 				/* schedule DCE */
-				ir_bitqueue_add(worklist, input);
+				ir_bitqueue_add(ctx->iter_worklist, input);
 			} else if (ctx->ir_base[input].op == IR_PHI && ctx->use_lists[input].count == 1) {
 				/* try to optimize PHI into ABS/MIN/MAX/COND */
-				ir_bitqueue_add(worklist, ctx->ir_base[input].op1);
+				ir_bitqueue_add(ctx->iter_worklist, ctx->ir_base[input].op1);
 			}
 		}
 	}
@@ -1294,7 +1350,7 @@ void ir_iter_replace(ir_ctx *ctx, ir_ref ref, ir_ref new_ref, ir_bitqueue *workl
 	}
 }

-static void ir_iter_replace_insn(ir_ctx *ctx, ir_ref ref, ir_ref new_ref, ir_bitqueue *worklist)
+static void ir_iter_replace_insn(ir_ctx *ctx, ir_ref ref, ir_ref new_ref)
 {
 	ir_ref j, n, *p;
 	ir_insn *insn;
@@ -1309,20 +1365,20 @@ static void ir_iter_replace_insn(ir_ctx *ctx, ir_ref ref, ir_ref new_ref, ir_bit
 			ir_use_list_remove_one(ctx, input, ref);
 			if (ir_is_dead(ctx, input)) {
 				/* schedule DCE */
-				ir_bitqueue_add(worklist, input);
+				ir_bitqueue_add(ctx->iter_worklist, input);
 			} else if (ctx->ir_base[input].op == IR_PHI && ctx->use_lists[input].count == 1) {
 				/* try to optimize PHI into ABS/MIN/MAX/COND */
-				ir_bitqueue_add(worklist, ctx->ir_base[input].op1);
+				ir_bitqueue_add(ctx->iter_worklist, ctx->ir_base[input].op1);
 			}
 		}
 	}

-	ir_iter_replace(ctx, ref, new_ref, worklist);
+	ir_iter_replace(ctx, ref, new_ref, ctx->iter_worklist);

 	CLEAR_USES(ref);
 }

-void ir_iter_update_op(ir_ctx *ctx, ir_ref ref, uint32_t idx, ir_ref new_val, ir_bitqueue *worklist)
+static void ir_iter_update_op(ir_ctx *ctx, ir_ref ref, uint32_t idx, ir_ref new_val)
 {
 	ir_insn *insn = &ctx->ir_base[ref];
 	ir_ref old_val = ir_insn_op(insn, idx);
@@ -1336,7 +1392,7 @@ void ir_iter_update_op(ir_ctx *ctx, ir_ref ref, uint32_t idx, ir_ref new_val, ir
 		ir_use_list_remove_one(ctx, old_val, ref);
 		if (ir_is_dead(ctx, old_val)) {
 			/* schedule DCE */
-			ir_bitqueue_add(worklist, old_val);
+			ir_bitqueue_add(ctx->iter_worklist, old_val);
 		}
 	}
 }
@@ -1361,7 +1417,7 @@ static ir_ref ir_iter_find_cse1(const ir_ctx *ctx, uint32_t optx, ir_ref op1)
 	return IR_UNUSED;
 }

-static ir_ref ir_iter_find_cse(const ir_ctx *ctx, ir_ref ref, uint32_t opt, ir_ref op1, ir_ref op2, ir_ref op3, ir_bitqueue *worklist)
+static ir_ref ir_iter_find_cse(const ir_ctx *ctx, ir_ref ref, uint32_t opt, ir_ref op1, ir_ref op2, ir_ref op3)
 {
 	uint32_t n = IR_INPUT_EDGES_COUNT(ir_op_flags[opt & IR_OPT_OP_MASK]);
 	const ir_use_list *use_list = NULL;
@@ -1387,7 +1443,7 @@ static ir_ref ir_iter_find_cse(const ir_ctx *ctx, ir_ref ref, uint32_t opt, ir_r
 						if (use < ref) {
 							return use;
 						} else {
-							ir_bitqueue_add(worklist, use);
+							ir_bitqueue_add(ctx->iter_worklist, use);
 						}
 					}
 				}
@@ -1409,7 +1465,7 @@ static ir_ref ir_iter_find_cse(const ir_ctx *ctx, ir_ref ref, uint32_t opt, ir_r
 						if (use < ref) {
 							return use;
 						} else {
-							ir_bitqueue_add(worklist, use);
+							ir_bitqueue_add(ctx->iter_worklist, use);
 						}
 					}
 				}
@@ -1436,7 +1492,7 @@ static ir_ref ir_iter_find_cse(const ir_ctx *ctx, ir_ref ref, uint32_t opt, ir_r
 						if (use < ref) {
 							return use;
 						} else {
-							ir_bitqueue_add(worklist, use);
+							ir_bitqueue_add(ctx->iter_worklist, use);
 						}
 					}
 				}
@@ -1446,7 +1502,7 @@ static ir_ref ir_iter_find_cse(const ir_ctx *ctx, ir_ref ref, uint32_t opt, ir_r
 	return IR_UNUSED;
 }

-static void ir_iter_fold(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
+static void ir_iter_fold(ir_ctx *ctx, ir_ref ref)
 {
 	uint32_t opt;
 	ir_ref op1, op2, op3, copy;
@@ -1473,9 +1529,9 @@ static void ir_iter_fold(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
 			goto restart;
 		case IR_FOLD_DO_CSE:
 			copy = ir_iter_find_cse(ctx, ref, ctx->fold_insn.opt,
-				ctx->fold_insn.op1, ctx->fold_insn.op2, ctx->fold_insn.op3, worklist);
+				ctx->fold_insn.op1, ctx->fold_insn.op2, ctx->fold_insn.op3);
 			if (copy) {
-				ir_iter_replace_insn(ctx, ref, copy, worklist);
+				ir_iter_replace_insn(ctx, ref, copy);
 				break;
 			}
 			IR_FALLTHROUGH;
@@ -1492,6 +1548,7 @@ static void ir_iter_fold(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
 				if (insn->op1 != ctx->fold_insn.op1) {
 					if (insn->op1 > 0) {
 						ir_use_list_remove_one(ctx, insn->op1, ref);
+						if (ctx->use_lists[insn->op1].count == 0) ir_bitqueue_add(ctx->iter_worklist, insn->op1);
 					}
 					if (ctx->fold_insn.op1 > 0) {
 						ir_use_list_add(ctx, ctx->fold_insn.op1, ref);
@@ -1500,6 +1557,7 @@ static void ir_iter_fold(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
 				if (insn->op2 != ctx->fold_insn.op2) {
 					if (insn->op2 > 0) {
 						ir_use_list_remove_one(ctx, insn->op2, ref);
+						if (ctx->use_lists[insn->op2].count == 0) ir_bitqueue_add(ctx->iter_worklist, insn->op2);
 					}
 					if (ctx->fold_insn.op2 > 0) {
 						ir_use_list_add(ctx, ctx->fold_insn.op2, ref);
@@ -1508,6 +1566,7 @@ static void ir_iter_fold(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
 				if (insn->op3 != ctx->fold_insn.op3) {
 					if (insn->op3 > 0) {
 						ir_use_list_remove_one(ctx, insn->op3, ref);
+						if (ctx->use_lists[insn->op3].count == 0) ir_bitqueue_add(ctx->iter_worklist, insn->op3);
 					}
 					if (ctx->fold_insn.op3 > 0) {
 						ir_use_list_add(ctx, ctx->fold_insn.op3, ref);
@@ -1517,16 +1576,16 @@ static void ir_iter_fold(ir_ctx *ctx, ir_ref ref, ir_bitqueue *worklist)
 				insn->op2 = ctx->fold_insn.op2;
 				insn->op3 = ctx->fold_insn.op3;

-				ir_iter_add_uses(ctx, ref, worklist);
+				ir_iter_add_uses(ctx, ref, ctx->iter_worklist);
 			}
 			break;
 		case IR_FOLD_DO_COPY:
 			op1 = ctx->fold_insn.op1;
-			ir_iter_replace_insn(ctx, ref, op1, worklist);
+			ir_iter_replace_insn(ctx, ref, op1);
 			break;
 		case IR_FOLD_DO_CONST:
 			op1 = ir_const(ctx, ctx->fold_insn.val, ctx->fold_insn.type);
-			ir_iter_replace_insn(ctx, ref, op1, worklist);
+			ir_iter_replace_insn(ctx, ref, op1);
 			break;
 		default:
 			IR_ASSERT(0);
@@ -1600,16 +1659,17 @@ static bool ir_may_promote_f2d(const ir_ctx *ctx, ir_ref ref)
 	return 0;
 }

-static ir_ref ir_promote_d2f(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *worklist)
+static ir_ref ir_promote_d2f(ir_ctx *ctx, ir_ref ref, ir_ref use)
 {
 	ir_insn *insn = &ctx->ir_base[ref];
 	uint32_t count;
+	ir_ref op;

 	IR_ASSERT(insn->type == IR_DOUBLE);
 	if (IR_IS_CONST_REF(ref)) {
 		return ir_const_float(ctx, (float)insn->val.d);
 	} else {
-		ir_bitqueue_add(worklist, ref);
+		ir_bitqueue_add(ctx->iter_worklist, ref);
 		switch (insn->op) {
 			case IR_FP2FP:
 				count = ctx->use_lists[ref].count;
@@ -1639,7 +1699,9 @@ static ir_ref ir_promote_d2f(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *w
 //				return ref;
 			case IR_NEG:
 			case IR_ABS:
-				insn->op1 = ir_promote_d2f(ctx, insn->op1, ref, worklist);
+				op = ir_promote_d2f(ctx, insn->op1, ref);
+				insn = &ctx->ir_base[ref];
+				insn->op1 = op;
 				insn->type = IR_FLOAT;
 				return ref;
 			case IR_ADD:
@@ -1649,10 +1711,17 @@ static ir_ref ir_promote_d2f(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *w
 			case IR_MIN:
 			case IR_MAX:
 				if (insn->op1 == insn->op2) {
-					insn->op2 = insn->op1 = ir_promote_d2f(ctx, insn->op1, ref, worklist);
+					op = ir_promote_d2f(ctx, insn->op1, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op2 = insn->op1 = op;
 				} else {
-					insn->op1 = ir_promote_d2f(ctx, insn->op1, ref, worklist);
-					insn->op2 = ir_promote_d2f(ctx, insn->op2, ref, worklist);
+					ir_ref op1 = insn->op1;
+					ir_ref op2 = insn->op2;
+					op1 = ir_promote_d2f(ctx, op1, ref);
+					op2 = ir_promote_d2f(ctx, op2, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op1 = op1;
+					insn->op2 = op2;
 				}
 				insn->type = IR_FLOAT;
 				return ref;
@@ -1664,17 +1733,18 @@ static ir_ref ir_promote_d2f(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *w
 	return ref;
 }

-static ir_ref ir_promote_f2d(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *worklist)
+static ir_ref ir_promote_f2d(ir_ctx *ctx, ir_ref ref, ir_ref use)
 {
 	ir_insn *insn = &ctx->ir_base[ref];
 	uint32_t count;
 	ir_ref old_ref;
+	ir_ref op;

 	IR_ASSERT(insn->type == IR_FLOAT);
 	if (IR_IS_CONST_REF(ref)) {
 		return ir_const_double(ctx, (double)insn->val.f);
 	} else {
-		ir_bitqueue_add(worklist, ref);
+		ir_bitqueue_add(ctx->iter_worklist, ref);
 		switch (insn->op) {
 			case IR_FP2FP:
 				count = ctx->use_lists[ref].count;
@@ -1713,7 +1783,9 @@ static ir_ref ir_promote_f2d(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *w
 				return ref;
 			case IR_NEG:
 			case IR_ABS:
-				insn->op1 = ir_promote_f2d(ctx, insn->op1, ref, worklist);
+				op = ir_promote_f2d(ctx, insn->op1, ref);
+				insn = &ctx->ir_base[ref];
+				insn->op1 = op;
 				insn->type = IR_DOUBLE;
 				return ref;
 			case IR_ADD:
@@ -1723,10 +1795,17 @@ static ir_ref ir_promote_f2d(ir_ctx *ctx, ir_ref ref, ir_ref use, ir_bitqueue *w
 			case IR_MIN:
 			case IR_MAX:
 				if (insn->op1 == insn->op2) {
-					insn->op2 = insn->op1 = ir_promote_f2d(ctx, insn->op1, ref, worklist);
+					op = ir_promote_f2d(ctx, insn->op1, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op2 = insn->op1 = op;
 				} else {
-					insn->op1 = ir_promote_f2d(ctx, insn->op1, ref, worklist);
-					insn->op2 = ir_promote_f2d(ctx, insn->op2, ref, worklist);
+					ir_ref op1 = insn->op1;
+					ir_ref op2 = insn->op2;
+					op1 = ir_promote_f2d(ctx, op1, ref);
+					op2 = ir_promote_f2d(ctx, op2, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op1 = op1;
+					insn->op2 = op2;
 				}
 				insn->type = IR_DOUBLE;
 				return ref;
@@ -1765,10 +1844,15 @@ static bool ir_may_promote_trunc(const ir_ctx *ctx, ir_type type, ir_ref ref)
 			case IR_OR:
 			case IR_AND:
 			case IR_XOR:
-			case IR_SHL:
 				return ctx->use_lists[ref].count == 1 &&
 					ir_may_promote_trunc(ctx, type, insn->op1) &&
 					ir_may_promote_trunc(ctx, type, insn->op2);
+			case IR_SHL:
+				return ctx->use_lists[ref].count == 1 &&
+					ir_may_promote_trunc(ctx, type, insn->op1) &&
+					IR_IS_CONST_REF(insn->op2) &&
+					(IR_IS_TYPE_UNSIGNED(ctx->ir_base[insn->op2].type) || ctx->ir_base[insn->op2].val.i64 >= 0) &&
+					ctx->ir_base[insn->op2].val.u64 < ir_type_size[type] * 8;
 //			case IR_SHR:
 //			case IR_SAR:
 //			case IR_DIV:
@@ -1777,6 +1861,8 @@ static bool ir_may_promote_trunc(const ir_ctx *ctx, ir_type type, ir_ref ref)
 //				TODO: ???
 			case IR_COND:
 				return ctx->use_lists[ref].count == 1 &&
+					insn->op1 != insn->op2 &&
+					insn->op1 != insn->op3 &&
 					ir_may_promote_trunc(ctx, type, insn->op2) &&
 					ir_may_promote_trunc(ctx, type, insn->op3);
 			case IR_PHI:
@@ -1809,11 +1895,11 @@ static bool ir_may_promote_trunc(const ir_ctx *ctx, ir_type type, ir_ref ref)
 	return 0;
 }

-static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use, ir_bitqueue *worklist)
+static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use)
 {
 	ir_insn *insn = &ctx->ir_base[ref];
 	uint32_t count;
-	ir_ref *p, n, input;
+	ir_ref n, input, op;

 	if (IR_IS_CONST_REF(ref)) {
 		ir_val val;
@@ -1831,13 +1917,15 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 		}
 		return ir_const(ctx, val, type);
 	} else {
-		ir_bitqueue_add(worklist, ref);
+		ir_bitqueue_add(ctx->iter_worklist, ref);
 		switch (insn->op) {
 			case IR_ZEXT:
 			case IR_SEXT:
 			case IR_TRUNC:
 				if (ctx->ir_base[insn->op1].type != type) {
 					ir_type src_type = ctx->ir_base[insn->op1].type;
+
+					IR_ASSERT(IR_IS_TYPE_INT(src_type) && IR_IS_TYPE_INT(type));
 					if (ir_type_size[src_type] == ir_type_size[type]) {
 						insn->op = IR_BITCAST;
 					} else if (ir_type_size[src_type] > ir_type_size[type]) {
@@ -1848,7 +1936,7 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 						}
 					}
 					insn->type = type;
-					ir_iter_add_uses(ctx, ref, worklist);
+					ir_iter_add_uses(ctx, ref, ctx->iter_worklist);
 					return ref;
 				}

@@ -1877,7 +1965,9 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 			case IR_NEG:
 			case IR_ABS:
 			case IR_NOT:
-				insn->op1 = ir_promote_i2i(ctx, type, insn->op1, ref, worklist);
+				op = ir_promote_i2i(ctx, type, insn->op1, ref);
+				insn = &ctx->ir_base[ref];
+				insn->op1 = op;
 				insn->type = type;
 				return ref;
 			case IR_ADD:
@@ -1890,10 +1980,17 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 			case IR_XOR:
 			case IR_SHL:
 				if (insn->op1 == insn->op2) {
-					insn->op2 = insn->op1 = ir_promote_i2i(ctx, type, insn->op1, ref, worklist);
+					op = ir_promote_i2i(ctx, type, insn->op1, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op2 = insn->op1 = op;
 				} else {
-					insn->op1 = ir_promote_i2i(ctx, type, insn->op1, ref, worklist);
-					insn->op2 = ir_promote_i2i(ctx, type, insn->op2, ref, worklist);
+					ir_ref op1 = insn->op1;
+					ir_ref op2 = insn->op2;
+					op1 = ir_promote_i2i(ctx, type, op1, ref);
+					op2 = ir_promote_i2i(ctx, type, op2, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op1 = op1;
+					insn->op2 = op2;
 				}
 				insn->type = type;
 				return ref;
@@ -1905,10 +2002,17 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 //				TODO: ???
 			case IR_COND:
 				if (insn->op2 == insn->op3) {
-					insn->op3 = insn->op2 = ir_promote_i2i(ctx, type, insn->op2, ref, worklist);
+					op = ir_promote_i2i(ctx, type, insn->op2, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op3 = insn->op2 = op;
 				} else {
-					insn->op2 = ir_promote_i2i(ctx, type, insn->op2, ref, worklist);
-					insn->op3 = ir_promote_i2i(ctx, type, insn->op3, ref, worklist);
+					ir_ref op2 = insn->op2;
+					ir_ref op3 = insn->op3;
+					op2 = ir_promote_i2i(ctx, type, op2, ref);
+					op3 = ir_promote_i2i(ctx, type, op3, ref);
+					insn = &ctx->ir_base[ref];
+					insn->op2 = op2;
+					insn->op3 = op3;
 				}
 				insn->type = type;
 				if (IR_IS_TYPE_SIGNED(type)) {
@@ -1917,14 +2021,14 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 						if (cond->op1 == insn->op2 && cond->op2 == insn->op3) {
 							insn->op = (cond->op == IR_LT || cond->op == IR_LE) ? IR_MIN : IR_MAX;
 							ir_use_list_remove_one(ctx, insn->op1, ref);
-							ir_bitqueue_add(worklist, insn->op1);
+							ir_bitqueue_add(ctx->iter_worklist, insn->op1);
 							insn->op1 = insn->op2;
 							insn->op2 = insn->op3;
 							insn->op3 = IR_UNUSED;
 						} else if (cond->op1 == insn->op3 && cond->op2 == insn->op1) {
 							insn->op = (cond->op == IR_LT || cond->op == IR_LE) ? IR_MAX : IR_MIN;
 							ir_use_list_remove_one(ctx, insn->op1, ref);
-							ir_bitqueue_add(worklist, insn->op1);
+							ir_bitqueue_add(ctx->iter_worklist, insn->op1);
 							insn->op1 = insn->op2;
 							insn->op2 = insn->op3;
 							insn->op3 = IR_UNUSED;
@@ -1937,14 +2041,14 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 						if (cond->op1 == insn->op2 && cond->op2 == insn->op3) {
 							insn->op = (cond->op == IR_ULT || cond->op == IR_ULE) ? IR_MIN : IR_MAX;
 							ir_use_list_remove_one(ctx, insn->op1, ref);
-							ir_bitqueue_add(worklist, insn->op1);
+							ir_bitqueue_add(ctx->iter_worklist, insn->op1);
 							insn->op1 = insn->op2;
 							insn->op2 = insn->op3;
 							insn->op3 = IR_UNUSED;
 						} else if (cond->op1 == insn->op3 && cond->op2 == insn->op1) {
 							insn->op = (cond->op == IR_ULT || cond->op == IR_ULE) ? IR_MAX : IR_MIN;
 							ir_use_list_remove_one(ctx, insn->op1, ref);
-							ir_bitqueue_add(worklist, insn->op1);
+							ir_bitqueue_add(ctx->iter_worklist, insn->op1);
 							insn->op1 = insn->op2;
 							insn->op2 = insn->op3;
 							insn->op3 = IR_UNUSED;
@@ -1953,12 +2057,14 @@ static ir_ref ir_promote_i2i(ir_ctx *ctx, ir_type type, ir_ref ref, ir_ref use,
 				}
 				return ref;
 			case IR_PHI:
-				for (p = insn->ops + 2, n = insn->inputs_count - 1; n > 0; p++, n--) {
-					input = *p;
+				for (count = 2, n = insn->inputs_count - 1; n > 0; count++, n--) {
+					input = ir_get_op(ctx, ref, count);
 					if (input != ref) {
-						*p = ir_promote_i2i(ctx, type, input, ref, worklist);
+						op = ir_promote_i2i(ctx, type, input, ref);
+						ir_set_op(ctx, ref, count, op);
 					}
 				}
+				insn = &ctx->ir_base[ref];
 				insn->type = type;
 				return ref;
 			default:
@@ -2006,7 +2112,7 @@ static ir_ref ir_ext_const(ir_ctx *ctx, const ir_insn *val_insn, ir_op op, ir_ty
 	return ir_const(ctx, new_val, type);
 }

-static ir_ref ir_ext_ref(ir_ctx *ctx, ir_ref var_ref, ir_ref src_ref, ir_op op, ir_type type, ir_bitqueue *worklist)
+static ir_ref ir_ext_ref(ir_ctx *ctx, ir_ref var_ref, ir_ref src_ref, ir_op op, ir_type type)
 {
 	uint32_t optx = IR_OPTX(op, type, 1);
 	ir_ref ref;
@@ -2018,7 +2124,7 @@ static ir_ref ir_ext_ref(ir_ctx *ctx, ir_ref var_ref, ir_ref src_ref, ir_op op,
 			if (!IR_IS_CONST_REF(src_ref)) {
 				ir_use_list_remove_one(ctx, src_ref, var_ref);
 			}
-			ir_bitqueue_add(worklist, ref);
+			ir_bitqueue_add(ctx->iter_worklist, ref);
 			return ref;
 		}
 	}
@@ -2028,8 +2134,8 @@ static ir_ref ir_ext_ref(ir_ctx *ctx, ir_ref var_ref, ir_ref src_ref, ir_op op,
 	if (!IR_IS_CONST_REF(src_ref)) {
 		ir_use_list_replace_one(ctx, src_ref, var_ref, ref);
 	}
-	ir_bitqueue_grow(worklist, ref + 1);
-	ir_bitqueue_add(worklist, ref);
+	ir_bitqueue_grow(ctx->iter_worklist, ref + 1);
+	ir_bitqueue_add(ctx->iter_worklist, ref);
 	return ref;
 }

@@ -2114,7 +2220,7 @@ static bool ir_is_cheaper_ext(const ir_ctx *ctx, ir_ref ref, ir_ref loop, ir_ref
 	}
 }

-static bool ir_try_promote_induction_var_ext(ir_ctx *ctx, ir_ref ext_ref, ir_ref phi_ref, ir_ref op_ref, ir_bitqueue *worklist)
+static bool ir_try_promote_induction_var_ext(ir_ctx *ctx, ir_ref ext_ref, ir_ref phi_ref, ir_ref op_ref)
 {
 	ir_op op = ctx->ir_base[ext_ref].op;
 	ir_type type = ctx->ir_base[ext_ref].type;
@@ -2223,22 +2329,22 @@ static bool ir_try_promote_induction_var_ext(ir_ctx *ctx, ir_ref ext_ref, ir_ref
 				 && !IR_IS_SYM_CONST(ctx->ir_base[use_insn->op1].op)) {
 					ctx->ir_base[use].op1 = ir_ext_const(ctx, &ctx->ir_base[use_insn->op1], op, type);
 				} else {
-					ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op1, op, type, worklist);
+					ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op1, op, type);
 					use_insn = &ctx->ir_base[use];
 					use_insn->op1 = tmp;
 				}
-				ir_bitqueue_add(worklist, use);
+				ir_bitqueue_add(ctx->iter_worklist, use);
 			}
 			if (use_insn->op2 != phi_ref) {
 				if (IR_IS_CONST_REF(use_insn->op2)
 				 && !IR_IS_SYM_CONST(ctx->ir_base[use_insn->op2].op)) {
 					ctx->ir_base[use].op2 = ir_ext_const(ctx, &ctx->ir_base[use_insn->op2], op, type);
 				} else {
-					ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op2, op, type, worklist);
+					ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op2, op, type);
 					use_insn = &ctx->ir_base[use];
 					use_insn->op2 = tmp;
 				}
-				ir_bitqueue_add(worklist, use);
+				ir_bitqueue_add(ctx->iter_worklist, use);
 			}
 		}
 	}
@@ -2264,31 +2370,31 @@ static bool ir_try_promote_induction_var_ext(ir_ctx *ctx, ir_ref ext_ref, ir_ref
 					 && !IR_IS_SYM_CONST(ctx->ir_base[use_insn->op1].op)) {
 						ctx->ir_base[use].op1 = ir_ext_const(ctx, &ctx->ir_base[use_insn->op1], op, type);
 					} else {
-						ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op1, op, type, worklist);
+						ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op1, op, type);
 						use_insn = &ctx->ir_base[use];
 						use_insn->op1 = tmp;
 					}
-					ir_bitqueue_add(worklist, use);
+					ir_bitqueue_add(ctx->iter_worklist, use);
 				}
 				if (use_insn->op2 != op_ref) {
 					if (IR_IS_CONST_REF(use_insn->op2)
 					 && !IR_IS_SYM_CONST(ctx->ir_base[use_insn->op2].op)) {
 						ctx->ir_base[use].op2 = ir_ext_const(ctx, &ctx->ir_base[use_insn->op2], op, type);
 					} else {
-						ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op2, op, type, worklist);
+						ir_ref tmp = ir_ext_ref(ctx, use, use_insn->op2, op, type);
 						use_insn = &ctx->ir_base[use];
 						use_insn->op2 = tmp;
 					}
-					ir_bitqueue_add(worklist, use);
+					ir_bitqueue_add(ctx->iter_worklist, use);
 				}
 			}
 		}
 	}

-	ir_iter_replace_insn(ctx, ext_ref, ctx->ir_base[ext_ref].op1, worklist);
+	ir_iter_replace_insn(ctx, ext_ref, ctx->ir_base[ext_ref].op1);

 	if (ext_ref_2) {
-		ir_iter_replace_insn(ctx, ext_ref_2, ctx->ir_base[ext_ref_2].op1, worklist);
+		ir_iter_replace_insn(ctx, ext_ref_2, ctx->ir_base[ext_ref_2].op1);
 	}

 	ctx->ir_base[op_ref].type = type;
@@ -2299,14 +2405,14 @@ static bool ir_try_promote_induction_var_ext(ir_ctx *ctx, ir_ref ext_ref, ir_ref
 	 && !IR_IS_SYM_CONST(ctx->ir_base[phi_insn->op2].op)) {
 		ctx->ir_base[phi_ref].op2 = ir_ext_const(ctx, &ctx->ir_base[phi_insn->op2], op, type);
 	} else {
-		ir_ref tmp = ir_ext_ref(ctx, phi_ref, phi_insn->op2, op, type, worklist);
+		ir_ref tmp = ir_ext_ref(ctx, phi_ref, phi_insn->op2, op, type);
 		ctx->ir_base[phi_ref].op2 = tmp;
 	}

 	return 1;
 }

-static bool ir_try_promote_ext(ir_ctx *ctx, ir_ref ext_ref, ir_insn *insn, ir_bitqueue *worklist)
+static bool ir_try_promote_ext(ir_ctx *ctx, ir_ref ext_ref, ir_insn *insn)
 {
 	ir_ref ref = insn->op1;

@@ -2321,11 +2427,11 @@ static bool ir_try_promote_ext(ir_ctx *ctx, ir_ref ext_ref, ir_insn *insn, ir_bi
 		if (op_insn->op == IR_ADD || op_insn->op == IR_SUB || op_insn->op == IR_MUL) {
 			if (op_insn->op1 == ref) {
 				if (ir_is_loop_invariant(ctx, op_insn->op2, insn->op1)) {
-					return ir_try_promote_induction_var_ext(ctx, ext_ref, ref, op_ref, worklist);
+					return ir_try_promote_induction_var_ext(ctx, ext_ref, ref, op_ref);
 				}
 			} else if (op_insn->op2 == ref) {
 				if (ir_is_loop_invariant(ctx, op_insn->op1, insn->op1)) {
-					return ir_try_promote_induction_var_ext(ctx, ext_ref, ref, op_ref, worklist);
+					return ir_try_promote_induction_var_ext(ctx, ext_ref, ref, op_ref);
 				}
 			}
 		}
@@ -2336,14 +2442,14 @@ static bool ir_try_promote_ext(ir_ctx *ctx, ir_ref ext_ref, ir_insn *insn, ir_bi
 		 && ctx->ir_base[insn->op1].op3 == ref
 		 && ctx->ir_base[ctx->ir_base[insn->op1].op1].op == IR_LOOP_BEGIN
 		 && ir_is_loop_invariant(ctx, insn->op2, ctx->ir_base[insn->op1].op1)) {
-			return ir_try_promote_induction_var_ext(ctx, ext_ref, insn->op1, ref, worklist);
+			return ir_try_promote_induction_var_ext(ctx, ext_ref, insn->op1, ref);
 		} else if (!IR_IS_CONST_REF(insn->op2)
 		 && ctx->ir_base[insn->op2].op == IR_PHI
 		 && ctx->ir_base[insn->op2].inputs_count == 3 /* (2 values) */
 		 && ctx->ir_base[insn->op2].op3 == ref
 		 && ctx->ir_base[ctx->ir_base[insn->op2].op1].op == IR_LOOP_BEGIN
 		 && ir_is_loop_invariant(ctx, insn->op1, ctx->ir_base[insn->op2].op1)) {
-			return ir_try_promote_induction_var_ext(ctx, ext_ref, insn->op2, ref, worklist);
+			return ir_try_promote_induction_var_ext(ctx, ext_ref, insn->op2, ref);
 		}
 	}

@@ -2368,7 +2474,7 @@ static void ir_get_true_false_refs(const ir_ctx *ctx, ir_ref if_ref, ir_ref *if_
 	}
 }

-static void ir_merge_blocks(ir_ctx *ctx, ir_ref end, ir_ref begin, ir_bitqueue *worklist)
+static void ir_merge_blocks(ir_ctx *ctx, ir_ref end, ir_ref begin)
 {
 	ir_ref prev, next;
 	ir_use_list *use_list;
@@ -2393,7 +2499,7 @@ static void ir_merge_blocks(ir_ctx *ctx, ir_ref end, ir_ref begin, ir_bitqueue *
 	ir_use_list_replace_one(ctx, prev, end, next);

 	if (ctx->ir_base[prev].op == IR_BEGIN || ctx->ir_base[prev].op == IR_MERGE) {
-		ir_bitqueue_add(worklist, prev);
+		ir_bitqueue_add(ctx->iter_worklist, prev);
 	}
 }

@@ -2413,7 +2519,7 @@ static void ir_remove_unused_vars(ir_ctx *ctx, ir_ref start, ir_ref end)
 	}
 }

-static bool ir_try_remove_empty_diamond(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue *worklist)
+static bool ir_try_remove_empty_diamond(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	if (insn->inputs_count == 2) {
 		ir_ref end1_ref = insn->op1, end2_ref = insn->op2;
@@ -2474,7 +2580,7 @@ static bool ir_try_remove_empty_diamond(ir_ctx *ctx, ir_ref ref, ir_insn *insn,
 		if (!IR_IS_CONST_REF(root->op2)) {
 			ir_use_list_remove_one(ctx, root->op2, root_ref);
 			if (ir_is_dead(ctx, root->op2)) {
-				ir_bitqueue_add(worklist, root->op2);
+				ir_bitqueue_add(ctx->iter_worklist, root->op2);
 			}
 		}

@@ -2486,7 +2592,7 @@ static bool ir_try_remove_empty_diamond(ir_ctx *ctx, ir_ref ref, ir_insn *insn,
 		MAKE_NOP(insn);   CLEAR_USES(ref);

 		if (ctx->ir_base[next->op1].op == IR_BEGIN || ctx->ir_base[next->op1].op == IR_MERGE) {
-			ir_bitqueue_add(worklist, next->op1);
+			ir_bitqueue_add(ctx->iter_worklist, next->op1);
 		}

 		return 1;
@@ -2532,7 +2638,7 @@ static bool ir_try_remove_empty_diamond(ir_ctx *ctx, ir_ref ref, ir_insn *insn,
 		if (!IR_IS_CONST_REF(root->op2)) {
 			ir_use_list_remove_one(ctx, root->op2, root_ref);
 			if (ir_is_dead(ctx, root->op2)) {
-				ir_bitqueue_add(worklist, root->op2);
+				ir_bitqueue_add(ctx->iter_worklist, root->op2);
 			}
 		}

@@ -2551,7 +2657,7 @@ static bool ir_try_remove_empty_diamond(ir_ctx *ctx, ir_ref ref, ir_insn *insn,
 		MAKE_NOP(insn);   CLEAR_USES(ref);

 		if (ctx->ir_base[next->op1].op == IR_BEGIN || ctx->ir_base[next->op1].op == IR_MERGE) {
-			ir_bitqueue_add(worklist, next->op1);
+			ir_bitqueue_add(ctx->iter_worklist, next->op1);
 		}

 		return 1;
@@ -2611,7 +2717,7 @@ static bool ir_fix_min_max_const(ir_ctx *ctx, ir_insn *cond, ir_ref ref)
 	return 0;
 }

-static bool ir_optimize_phi(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge, ir_ref ref, ir_insn *insn, ir_bitqueue *worklist)
+static bool ir_optimize_phi(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge, ir_ref ref, ir_insn *insn)
 {
 	IR_ASSERT(insn->inputs_count == 3);
 	IR_ASSERT(ctx->use_lists[merge_ref].count == 2);
@@ -2732,7 +2838,7 @@ static bool ir_optimize_phi(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge, ir_re
 					MAKE_NOP(merge);  CLEAR_USES(merge_ref);

 					if (ctx->ir_base[next->op1].op == IR_BEGIN || ctx->ir_base[next->op1].op == IR_MERGE) {
-						ir_bitqueue_add(worklist, next->op1);
+						ir_bitqueue_add(ctx->iter_worklist, next->op1);
 					}

 					return 1;
@@ -2823,7 +2929,7 @@ static bool ir_optimize_phi(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge, ir_re
 					MAKE_NOP(&ctx->ir_base[neg_ref]); CLEAR_USES(neg_ref);

 					if (ctx->ir_base[next->op1].op == IR_BEGIN || ctx->ir_base[next->op1].op == IR_MERGE) {
-						ir_bitqueue_add(worklist, next->op1);
+						ir_bitqueue_add(ctx->iter_worklist, next->op1);
 					}

 					return 1;
@@ -2890,9 +2996,9 @@ static bool ir_optimize_phi(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge, ir_re
 					MAKE_NOP(end2);   CLEAR_USES(end2_ref);
 					MAKE_NOP(merge);  CLEAR_USES(merge_ref);

-					ir_bitqueue_add(worklist, ref);
+					ir_bitqueue_add(ctx->iter_worklist, ref);
 					if (ctx->ir_base[next->op1].op == IR_BEGIN || ctx->ir_base[next->op1].op == IR_MERGE) {
-						ir_bitqueue_add(worklist, next->op1);
+						ir_bitqueue_add(ctx->iter_worklist, next->op1);
 					}

 					return 1;
@@ -3010,7 +3116,7 @@ static bool ir_cmp_is_true(ir_op op, const ir_insn *op1, const ir_insn *op2)
 	}
 }

-static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue *worklist)
+static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	ir_ref cond_ref = insn->op2;
 	ir_insn *cond = &ctx->ir_base[cond_ref];
@@ -3084,8 +3190,8 @@ static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue
 						if_true->op1 = end2_ref;
 						if_true->op2 = IR_UNUSED;

-						ir_bitqueue_add(worklist, if_false_ref);
-						ir_bitqueue_add(worklist, if_true_ref);
+						ir_bitqueue_add(ctx->iter_worklist, if_false_ref);
+						ir_bitqueue_add(ctx->iter_worklist, if_true_ref);

 						return 1;
 					} else {
@@ -3124,7 +3230,7 @@ static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue

 						ctx->flags2 &= ~IR_CFG_REACHABLE;

-						ir_bitqueue_add(worklist, if_false_ref);
+						ir_bitqueue_add(ctx->iter_worklist, if_false_ref);

 						return 1;
 					}
@@ -3158,7 +3264,7 @@ static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue

 				end2->optx = IR_OPTX(IR_IF, IR_VOID, 2);
 				end2->op2 = cond->op3;
-				ir_bitqueue_add(worklist, end2_ref);
+				ir_bitqueue_add(ctx->iter_worklist, end2_ref);

 				merge->optx = IR_OPTX(op, IR_VOID, 1);
 				merge->op1 = end2_ref;
@@ -3177,9 +3283,9 @@ static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue
 				if_false->op1 = end1_ref;
 				if_false->op2 = ref;

-				ir_bitqueue_add(worklist, if_false_ref);
+				ir_bitqueue_add(ctx->iter_worklist, if_false_ref);
 				if (ctx->ir_base[end2->op1].op == IR_BEGIN || ctx->ir_base[end2->op1].op == IR_MERGE) {
-					ir_bitqueue_add(worklist, end2->op1);
+					ir_bitqueue_add(ctx->iter_worklist, end2->op1);
 				}

 				return 1;
@@ -3190,7 +3296,7 @@ static bool ir_try_split_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue
 	return 0;
 }

-static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue *worklist)
+static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	ir_ref cond_ref = insn->op2;
 	ir_insn *cond = &ctx->ir_base[cond_ref];
@@ -3276,8 +3382,8 @@ static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu
 							if_true->op1 = end2_ref;
 							if_true->op2 = IR_UNUSED;

-							ir_bitqueue_add(worklist, if_false_ref);
-							ir_bitqueue_add(worklist, if_true_ref);
+							ir_bitqueue_add(ctx->iter_worklist, if_false_ref);
+							ir_bitqueue_add(ctx->iter_worklist, if_true_ref);

 							return 1;
 						} else {
@@ -3320,7 +3426,7 @@ static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu

 							ctx->flags2 &= ~IR_CFG_REACHABLE;

-							ir_bitqueue_add(worklist, if_false_ref);
+							ir_bitqueue_add(ctx->iter_worklist, if_false_ref);

 							return 1;
 						}
@@ -3357,7 +3463,7 @@ static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu

 						end2->optx = IR_OPTX(IR_IF, IR_VOID, 2);
 						end2->op2 = insn->op2;
-						ir_bitqueue_add(worklist, end2_ref);
+						ir_bitqueue_add(ctx->iter_worklist, end2_ref);

 						merge->optx = IR_OPTX(op, IR_VOID, 1);
 						merge->op1 = end2_ref;
@@ -3377,9 +3483,9 @@ static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu
 						if_false->op1 = end1_ref;
 						if_false->op2 = ref;

-						ir_bitqueue_add(worklist, if_false_ref);
+						ir_bitqueue_add(ctx->iter_worklist, if_false_ref);
 						if (ctx->ir_base[end2->op1].op == IR_BEGIN || ctx->ir_base[end2->op1].op == IR_MERGE) {
-							ir_bitqueue_add(worklist, end2->op1);
+							ir_bitqueue_add(ctx->iter_worklist, end2->op1);
 						}

 						return 1;
@@ -3392,12 +3498,12 @@ static bool ir_try_split_if_cmp(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu
 	return 0;
 }

-static void ir_iter_optimize_merge(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge, ir_bitqueue *worklist)
+static void ir_iter_optimize_merge(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge)
 {
 	ir_use_list *use_list = &ctx->use_lists[merge_ref];

 	if (use_list->count == 1) {
-		ir_try_remove_empty_diamond(ctx, merge_ref, merge, worklist);
+		ir_try_remove_empty_diamond(ctx, merge_ref, merge);
 	} else if (use_list->count == 2) {
 		if (merge->inputs_count == 2) {
 			ir_ref phi_ref = ctx->use_edges[use_list->refs];
@@ -3414,7 +3520,7 @@ static void ir_iter_optimize_merge(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge
 			if (phi->op == IR_PHI && next->op != IR_PHI) {
 				if (next->op == IR_IF && next->op1 == merge_ref && ctx->use_lists[phi_ref].count == 1) {
 					if (next->op2 == phi_ref) {
-						if (ir_try_split_if(ctx, next_ref, next, worklist)) {
+						if (ir_try_split_if(ctx, next_ref, next)) {
 							return;
 						}
 					} else {
@@ -3425,13 +3531,15 @@ static void ir_iter_optimize_merge(ir_ctx *ctx, ir_ref merge_ref, ir_insn *merge
 						 && IR_IS_CONST_REF(cmp->op2)
 						 && !IR_IS_SYM_CONST(ctx->ir_base[cmp->op2].op)
 						 && ctx->use_lists[next->op2].count == 1) {
-							if (ir_try_split_if_cmp(ctx, next_ref, next, worklist)) {
+							if (ir_try_split_if_cmp(ctx, next_ref, next)) {
 								return;
 							}
 						}
 					}
 				}
-				ir_optimize_phi(ctx, merge_ref, merge, phi_ref, phi, worklist);
+				if (IR_IS_TYPE_SCALAR(phi->type)) {
+					ir_optimize_phi(ctx, merge_ref, merge, phi_ref, phi);
+				}
 			}
 		}
 	}
@@ -3453,7 +3561,7 @@ static ir_ref ir_find_ext_use(const ir_ctx *ctx, ir_ref ref)
 	return IR_UNUSED;
 }

-static void ir_iter_optimize_induction_var(ir_ctx *ctx, ir_ref phi_ref, ir_ref op_ref, ir_bitqueue *worklist)
+static void ir_iter_optimize_induction_var(ir_ctx *ctx, ir_ref phi_ref, ir_ref op_ref)
 {
 	ir_ref ext_ref;

@@ -3462,11 +3570,11 @@ static void ir_iter_optimize_induction_var(ir_ctx *ctx, ir_ref phi_ref, ir_ref o
 		ext_ref = ir_find_ext_use(ctx, op_ref);
 	}
 	if (ext_ref) {
-		ir_try_promote_induction_var_ext(ctx, ext_ref, phi_ref, op_ref, worklist);
+		ir_try_promote_induction_var_ext(ctx, ext_ref, phi_ref, op_ref);
 	}
 }

-static void ir_iter_optimize_loop(ir_ctx *ctx, ir_ref loop_ref, ir_insn *loop, ir_bitqueue *worklist)
+static void ir_iter_optimize_loop(ir_ctx *ctx, ir_ref loop_ref, ir_insn *loop)
 {
 	ir_ref n;

@@ -3487,11 +3595,11 @@ static void ir_iter_optimize_loop(ir_ctx *ctx, ir_ref loop_ref, ir_insn *loop, i
 			if (op_insn->op == IR_ADD || op_insn->op == IR_SUB || op_insn->op == IR_MUL) {
 				if (op_insn->op1 == use) {
 					if (ir_is_loop_invariant(ctx, op_insn->op2, loop_ref)) {
-						ir_iter_optimize_induction_var(ctx, use, op_ref, worklist);
+						ir_iter_optimize_induction_var(ctx, use, op_ref);
 					}
 				} else if (op_insn->op2 == use) {
 					if (ir_is_loop_invariant(ctx, op_insn->op1, loop_ref)) {
-						ir_iter_optimize_induction_var(ctx, use, op_ref, worklist);
+						ir_iter_optimize_induction_var(ctx, use, op_ref);
 					}
 				}
 		    }
@@ -3503,7 +3611,7 @@ static ir_ref ir_iter_optimize_condition(ir_ctx *ctx, ir_ref control, ir_ref con
 {
 	ir_insn *condition_insn = &ctx->ir_base[condition];

-	while ((condition_insn->op == IR_BITCAST
+	while (((condition_insn->op == IR_BITCAST && IR_IS_TYPE_SCALAR(ctx->ir_base[condition_insn->op1].type))
 	  || condition_insn->op == IR_ZEXT
 	  || condition_insn->op == IR_SEXT)
 	 && ctx->use_lists[condition].count == 1) {
@@ -3511,6 +3619,8 @@ static ir_ref ir_iter_optimize_condition(ir_ctx *ctx, ir_ref control, ir_ref con
 		condition_insn = &ctx->ir_base[condition];
 	}

+	IR_ASSERT(IR_IS_TYPE_SCALAR(condition_insn->type));
+
 	if (condition_insn->opt == IR_OPT(IR_NOT, IR_BOOL)) {
 		*swap = 1;
 		condition = condition_insn->op1;
@@ -3542,9 +3652,106 @@ static ir_ref ir_iter_optimize_condition(ir_ctx *ctx, ir_ref control, ir_ref con
 		if (!IR_IS_SYM_CONST(val_insn->op) && val_insn->val.u64 == 1) {
 			return IR_TRUE;
 		}
+	} else if (condition_insn->op == IR_SHR
+			&& ctx->use_lists[condition].count == 1
+			&& IR_IS_CONST_REF(condition_insn->op2)
+			&& !IR_IS_SYM_CONST(ctx->ir_base[condition_insn->op2].op)) {
+		uint64_t c1, c2 = ctx->ir_base[condition_insn->op2].val.u64;
+		ir_insn *op1_insn = &ctx->ir_base[condition_insn->op1];
+		ir_val val = {0};
+
+		if (op1_insn->op == IR_SHL
+		 && ctx->use_lists[condition_insn->op1].count == 1
+		 && IR_IS_CONST_REF(op1_insn->op2)
+		 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)) {
+			/* IF(SHR(SHL(X, C1), C2)) => IF(AND(X, ((-1) << C2) >> C1) */
+			c1 = ctx->ir_base[op1_insn->op2].val.u64;
+			if (op1_insn->op1 > 0) {
+				ir_use_list_replace_one(ctx, op1_insn->op1, condition_insn->op1, condition);
+			}
+			CLEAR_USES(condition_insn->op1);
+			condition_insn->op1 = op1_insn->op1;
+			MAKE_NOP(op1_insn);
+		} else if (op1_insn->op == IR_ADD
+		 && ctx->use_lists[condition_insn->op1].count == 1
+		 && op1_insn->op1 == op1_insn->op2) {
+			/* IF(SHR(ADD(X, X), C2)) => IF(AND(X, ((-1) << C2) >> 1) */
+			c1 = 1;
+			if (op1_insn->op1 > 0) {
+				ir_use_list_replace_one(ctx, op1_insn->op1, condition_insn->op1, condition);
+				ir_use_list_remove_one(ctx, op1_insn->op1, condition_insn->op1);
+			}
+			CLEAR_USES(condition_insn->op1);
+			condition_insn->op1 = op1_insn->op1;
+			MAKE_NOP(op1_insn);
+		} else if (op1_insn->op == IR_MUL
+		 && ctx->use_lists[condition_insn->op1].count == 1
+		 && IR_IS_CONST_REF(op1_insn->op2)
+		 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)
+		 && IR_IS_POWER_OF_TWO(ctx->ir_base[op1_insn->op2].val.u64)) {
+			/* IF(SHR(MUL(X, C1), C2)) => IF(AND(X, ((-1) << C2) >> log2(C1)) */
+			c1 = IR_LOG2(ctx->ir_base[op1_insn->op2].val.u64);
+			if (op1_insn->op1 > 0) {
+				ir_use_list_replace_one(ctx, op1_insn->op1, condition_insn->op1, condition);
+			}
+			CLEAR_USES(condition_insn->op1);
+			condition_insn->op1 = op1_insn->op1;
+			MAKE_NOP(op1_insn);
+		} else {
+			/* IF(SHR(X, C2)) => IF(AND(X, (-1) << C2) */
+			c1 = 0;
+		}
+
+		switch (ir_type_size[condition_insn->type]) {
+			case 1:
+				val.u64 = (uint8_t)((uint8_t)((uint8_t)(-1) << ((uint8_t)c2 & 7)) >> ((uint8_t)c1 & 7));
+				break;
+			case 2:
+				val.u64 = (uint16_t)((uint16_t)((uint16_t)(-1) << ((uint16_t)c2 & 0xf)) >> ((uint16_t)c1 & 0xf));
+				break;
+			case 4:
+				val.u64 = (uint32_t)((uint32_t)((uint32_t)(-1) << ((uint32_t)c2 & 0x1f)) >> ((uint32_t)c1 & 0x1f));
+				break;
+			case 8:
+				val.u64 = (((uint64_t)(-1) << (c2 & 0x3f)) >> (c1 & 0x3f));
+				break;
+			default:
+				IR_ASSERT(0);
+		}
+
+		ir_ref c = ir_const(ctx, val, condition_insn->type);
+		condition_insn = &ctx->ir_base[condition];
+		condition_insn->op = IR_AND;
+		condition_insn->op2 = c;
+	}
+
+	if (condition_insn->op == IR_AND && IR_IS_CONST_REF(condition_insn->op2)) {
+		ir_insn *val_insn = &ctx->ir_base[condition_insn->op2];
+		ir_insn *op1_insn = &ctx->ir_base[condition_insn->op1];
+
+		if (!IR_IS_SYM_CONST(val_insn->op)
+		 && ctx->use_lists[condition].count == 1
+		 && ctx->use_lists[condition_insn->op1].count == 1
+		 && (op1_insn->op == IR_ZEXT || op1_insn->op == IR_SEXT)
+		 && val_insn->val.u64 <= (((uint64_t)-1ULL) >> (64 - ir_type_size[ctx->ir_base[op1_insn->op1].type] * 8))) {
+			/* IF(AND(ZEXT(X), C)) => IF(AND(X, C)) */
+			if (op1_insn->op1 > 0) {
+				ir_use_list_replace_one(ctx, op1_insn->op1, condition_insn->op1, condition);
+			}
+			ir_ref op1_ref = condition_insn->op1;
+			CLEAR_USES(condition_insn->op1);
+			condition_insn->type = ctx->ir_base[op1_insn->op1].type;
+			condition_insn->op1 = op1_insn->op1;
+			ir_ref c = ir_const(ctx, val_insn->val, condition_insn->type);
+			condition_insn = &ctx->ir_base[condition];
+			condition_insn->op2 = c;
+			ir_bitqueue_add(ctx->iter_worklist, condition);
+			MAKE_NOP(&ctx->ir_base[op1_ref]);
+			return condition;
+		}
 	}

-	while ((condition_insn->op == IR_BITCAST
+	while (((condition_insn->op == IR_BITCAST && IR_IS_TYPE_SCALAR(ctx->ir_base[condition_insn->op1].type))
 	  || condition_insn->op == IR_ZEXT
 	  || condition_insn->op == IR_SEXT)
 	 && ctx->use_lists[condition].count == 1) {
@@ -3563,11 +3770,13 @@ static ir_ref ir_iter_optimize_condition(ir_ctx *ctx, ir_ref control, ir_ref con
 	return condition;
 }

-static void ir_iter_optimize_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue *worklist)
+static void ir_iter_optimize_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	bool swap = 0;
 	ir_ref condition = ir_iter_optimize_condition(ctx, insn->op1, insn->op2, &swap);

+	insn = &ctx->ir_base[ref];
+
 	if (swap) {
 		ir_use_list *use_list = &ctx->use_lists[ref];
 		ir_ref *p, use;
@@ -3602,7 +3811,7 @@ static void ir_iter_optimize_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu
 		insn->optx = IR_OPTX(IR_END, IR_VOID, 1);
 		if (!IR_IS_CONST_REF(insn->op2)) {
 			ir_use_list_remove_one(ctx, insn->op2, ref);
-			ir_bitqueue_add(worklist, insn->op2);
+			ir_bitqueue_add(ctx->iter_worklist, insn->op2);
 		}
 		insn->op2 = IR_UNUSED;

@@ -3616,23 +3825,25 @@ static void ir_iter_optimize_if(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqu
 		if (ir_ref_is_true(ctx, condition)) {
 			if_false->op1 = IR_UNUSED;
 			ir_use_list_remove_one(ctx, ref, if_false_ref);
-			ir_bitqueue_add(worklist, if_true_ref);
+			ir_bitqueue_add(ctx->iter_worklist, if_true_ref);
 		} else {
 			if_true->op1 = IR_UNUSED;
 			ir_use_list_remove_one(ctx, ref, if_true_ref);
-			ir_bitqueue_add(worklist, if_false_ref);
+			ir_bitqueue_add(ctx->iter_worklist, if_false_ref);
 		}
 		ctx->flags2 &= ~IR_CFG_REACHABLE;
 	} else if (insn->op2 != condition) {
-		ir_iter_update_op(ctx, ref, 2, condition, worklist);
+		ir_iter_update_op(ctx, ref, 2, condition);
 	}
 }

-static void ir_iter_optimize_guard(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bitqueue *worklist)
+static void ir_iter_optimize_guard(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	bool swap = 0;
 	ir_ref condition = ir_iter_optimize_condition(ctx, insn->op1, insn->op2, &swap);

+	insn = &ctx->ir_base[ref];
+
 	if (swap) {
 		if (insn->op == IR_GUARD) {
 			insn->op = IR_GUARD_NOT;
@@ -3655,7 +3866,7 @@ static void ir_iter_optimize_guard(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bi
 					ir_use_list_remove_one(ctx, snapshot, ref);
 					ir_use_list_remove_one(ctx, ref, next);
 					ir_use_list_replace_one(ctx, prev, snapshot, next);
-					ir_iter_remove_insn(ctx, snapshot, worklist);
+					ir_iter_remove_insn(ctx, snapshot);
 				} else {
 					ir_use_list_remove_one(ctx, ref, next);
 					ir_use_list_replace_one(ctx, prev, ref, next);
@@ -3667,7 +3878,7 @@ static void ir_iter_optimize_guard(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bi
 					ir_use_list_remove_one(ctx, insn->op2, ref);
 					if (ir_is_dead(ctx, insn->op2)) {
 						/* schedule DCE */
-						ir_bitqueue_add(worklist, insn->op2);
+						ir_bitqueue_add(ctx->iter_worklist, insn->op2);
 					}
 				}

@@ -3675,7 +3886,7 @@ static void ir_iter_optimize_guard(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bi
 					ir_use_list_remove_one(ctx, insn->op3, ref);
 					if (ir_is_dead(ctx, insn->op3)) {
 						/* schedule DCE */
-						ir_bitqueue_add(worklist, insn->op3);
+						ir_bitqueue_add(ctx->iter_worklist, insn->op3);
 					}
 				}

@@ -3694,7 +3905,7 @@ static void ir_iter_optimize_guard(ir_ctx *ctx, ir_ref ref, ir_insn *insn, ir_bi
 	}

 	if (insn->op2 != condition) {
-		ir_iter_update_op(ctx, ref, 2, condition, worklist);
+		ir_iter_update_op(ctx, ref, 2, condition);
 	}
 }

@@ -3703,30 +3914,33 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 	ir_ref i, val;
 	ir_insn *insn;

-	while ((i = ir_bitqueue_pop(worklist)) >= 0) {
+	ctx->iter_worklist = worklist;
+	while ((i = ir_bitqueue_pop(ctx->iter_worklist)) >= 0) {
 		insn = &ctx->ir_base[i];
 		if (IR_IS_FOLDABLE_OP(insn->op)) {
 			if (ctx->use_lists[i].count == 0) {
 				if (insn->op == IR_PHI) {
-					ir_bitqueue_add(worklist, insn->op1);
+					ir_bitqueue_add(ctx->iter_worklist, insn->op1);
 				}
-				ir_iter_remove_insn(ctx, i, worklist);
+				ir_iter_remove_insn(ctx, i);
 			} else {
 				insn = &ctx->ir_base[i];
 				switch (insn->op) {
 					case IR_FP2FP:
 						if (insn->type == IR_FLOAT) {
 							if (ir_may_promote_d2f(ctx, insn->op1)) {
-								ir_ref ref = ir_promote_d2f(ctx, insn->op1, i, worklist);
+								ir_ref ref = ir_promote_d2f(ctx, insn->op1, i);
+								insn = &ctx->ir_base[i];
 								insn->op1 = ref;
-								ir_iter_replace_insn(ctx, i, ref, worklist);
+								ir_iter_replace_insn(ctx, i, ref);
 								break;
 							}
-						} else {
+						} else if (insn->type == IR_DOUBLE) {
 							if (ir_may_promote_f2d(ctx, insn->op1)) {
-								ir_ref ref = ir_promote_f2d(ctx, insn->op1, i, worklist);
+								ir_ref ref = ir_promote_f2d(ctx, insn->op1, i);
+								insn = &ctx->ir_base[i];
 								insn->op1 = ref;
-								ir_iter_replace_insn(ctx, i, ref, worklist);
+								ir_iter_replace_insn(ctx, i, ref);
 								break;
 							}
 						}
@@ -3734,25 +3948,30 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 					case IR_FP2INT:
 						if (ctx->ir_base[insn->op1].type == IR_DOUBLE) {
 							if (ir_may_promote_d2f(ctx, insn->op1)) {
-								insn->op1 = ir_promote_d2f(ctx, insn->op1, i, worklist);
+								ir_ref ref = ir_promote_d2f(ctx, insn->op1, i);
+								insn = &ctx->ir_base[i];
+								insn->op1 = ref;
 							}
-						} else {
+						} else if (ctx->ir_base[insn->op1].type == IR_FLOAT) {
 							if (ir_may_promote_f2d(ctx, insn->op1)) {
-								insn->op1 = ir_promote_f2d(ctx, insn->op1, i, worklist);
+								ir_ref ref = ir_promote_f2d(ctx, insn->op1, i);
+								insn = &ctx->ir_base[i];
+								insn->op1 = ref;
 							}
 						}
 						goto folding;
 					case IR_TRUNC:
 						if (ir_may_promote_trunc(ctx, insn->type, insn->op1)) {
-							ir_ref ref = ir_promote_i2i(ctx, insn->type, insn->op1, i, worklist);
+							ir_ref ref = ir_promote_i2i(ctx, insn->type, insn->op1, i);
+							insn = &ctx->ir_base[i];
 							insn->op1 = ref;
-							ir_iter_replace_insn(ctx, i, ref, worklist);
+							ir_iter_replace_insn(ctx, i, ref);
 							break;
 						}
 						goto folding;
 					case IR_SEXT:
 					case IR_ZEXT:
-						if (ir_try_promote_ext(ctx, i, insn, worklist)) {
+						if (ir_try_promote_ext(ctx, i, insn)) {
 							break;
 						}
 						goto folding;
@@ -3760,7 +3979,7 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 						break;
 					default:
 folding:
-						ir_iter_fold(ctx, i, worklist);
+						ir_iter_fold(ctx, i);
 						break;
 				}
 			}
@@ -3772,12 +3991,12 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 				 && !insn->op2 /* no computed goto label */
 				 && ctx->use_lists[i].count == 1
 				 && ctx->ir_base[insn->op1].op == IR_END) {
-					ir_merge_blocks(ctx, insn->op1, i, worklist);
+					ir_merge_blocks(ctx, insn->op1, i);
 				}
 			} else if (insn->op == IR_MERGE) {
-				ir_iter_optimize_merge(ctx, i, insn, worklist);
+				ir_iter_optimize_merge(ctx, i, insn);
 			} else if (insn->op == IR_LOOP_BEGIN) {
-				ir_iter_optimize_loop(ctx, i, insn, worklist);
+				ir_iter_optimize_loop(ctx, i, insn);
 			}
 		} else if (ir_is_dead_load(ctx, i)) {
 			ir_ref next;
@@ -3789,7 +4008,7 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 			ctx->ir_base[next].op1 = insn->op1;
 			ir_use_list_replace_one(ctx, insn->op1, i, next);
 			insn->op1 = IR_UNUSED;
-			ir_iter_remove_insn(ctx, i, worklist);
+			ir_iter_remove_insn(ctx, i);
 		} else if (insn->op == IR_LOAD) {
 			val = ir_find_aliasing_load(ctx, insn->op1, insn->type, insn->op2);
 			if (val) {
@@ -3806,19 +4025,19 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)

 				val_insn = &ctx->ir_base[val];
 				if (val_insn->type == insn->type) {
-					ir_iter_replace_insn(ctx, i, val, worklist);
+					ir_iter_replace_insn(ctx, i, val);
 				} else {
 					if (!IR_IS_CONST_REF(insn->op2)) {
 						ir_use_list_remove_one(ctx, insn->op2, i);
 						if (ir_is_dead(ctx, insn->op2)) {
 							/* schedule DCE */
-							ir_bitqueue_add(worklist, insn->op2);
+							ir_bitqueue_add(ctx->iter_worklist, insn->op2);
 						}
 					}
 					if (!IR_IS_CONST_REF(val)) {
 						ir_use_list_add(ctx, val, i);
 					}
-					if (ir_type_size[val_insn->type] == ir_type_size[insn->type]) {
+					if (ir_get_type_size(val_insn->type) == ir_get_type_size(insn->type)) {
 						/* load forwarding with bitcast (L2L) */
 						insn->optx = IR_OPTX(IR_BITCAST, insn->type, 1);
 					} else {
@@ -3827,8 +4046,8 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 					}
 					insn->op1 = val;
 					insn->op2 = IR_UNUSED;
-					ir_bitqueue_add(worklist, i);
-					ir_iter_add_uses(ctx, i, worklist);
+					ir_bitqueue_add(ctx->iter_worklist, i);
+					ir_iter_add_uses(ctx, i, ctx->iter_worklist);
 				}
 			}
 		} else if (insn->op == IR_STORE) {
@@ -3841,14 +4060,14 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 				val = insn->op3;
 				val_insn = &ctx->ir_base[val];
 				if (val_insn->op == IR_BITCAST
-				 && ir_type_size[val_insn->type] == ir_type_size[ctx->ir_base[val_insn->op1].type]) {
+				 && ir_get_type_size(val_insn->type) == ir_get_type_size(ctx->ir_base[val_insn->op1].type)) {
 					insn->op3 = val_insn->op1;
 					ir_use_list_remove_one(ctx, val, i);
 					if (ctx->use_lists[val].count == 0) {
 						if (!IR_IS_CONST_REF(val_insn->op1)) {
 							ir_use_list_replace_one(ctx, val_insn->op1, val, i);
 						}
-						ir_iter_remove_insn(ctx, val, worklist);
+						ir_iter_remove_insn(ctx, val);
 					} else {
 						if (!IR_IS_CONST_REF(val_insn->op1)) {
 							ir_use_list_add(ctx, val_insn->op1, i);
@@ -3868,11 +4087,12 @@ void ir_iter_opt(ir_ctx *ctx, ir_bitqueue *worklist)
 				goto remove_bitcast;
 			}
 		} else if (insn->op == IR_IF) {
-			ir_iter_optimize_if(ctx, i, insn, worklist);
+			ir_iter_optimize_if(ctx, i, insn);
 		} else if (insn->op == IR_GUARD || insn->op == IR_GUARD_NOT) {
-			ir_iter_optimize_guard(ctx, i, insn, worklist);
+			ir_iter_optimize_guard(ctx, i, insn);
 		}
 	}
+	ctx->iter_worklist = NULL;
 }

 void ir_iter_cleanup(ir_ctx *ctx)
@@ -3886,10 +4106,11 @@ void ir_iter_cleanup(ir_ctx *ctx)
 	ir_bitqueue_init(&iter_worklist, ctx->insns_count);

 	/* Remove unused nodes */
+	ctx->iter_worklist = &iter_worklist;
 	for (i = IR_UNUSED + 1, insn = ctx->ir_base + i; i < ctx->insns_count;) {
 		if (IR_IS_FOLDABLE_OP(insn->op)) {
 			if (insn->op != IR_NOP && ctx->use_lists[i].count == 0) {
-				ir_iter_remove_insn(ctx, i, &iter_worklist);
+				ir_iter_remove_insn(ctx, i);
 			}
 		} else if (insn->op == IR_IF || insn->op == IR_MERGE) {
 			ir_bitqueue_add(&cfg_worklist, i);
@@ -3904,15 +4125,16 @@ void ir_iter_cleanup(ir_ctx *ctx)
 		insn = &ctx->ir_base[i];
 		if (IR_IS_FOLDABLE_OP(insn->op)) {
 			if (ctx->use_lists[i].count == 0) {
-				ir_iter_remove_insn(ctx, i, &iter_worklist);
+				ir_iter_remove_insn(ctx, i);
 			}
 		}
 	}

+	ctx->iter_worklist = NULL;
+	ir_bitqueue_free(&iter_worklist);
+
 	/* Cleanup Control Flow */
 	ir_iter_opt(ctx, &cfg_worklist);
-
-	ir_bitqueue_free(&iter_worklist);
 	ir_bitqueue_free(&cfg_worklist);
 }

diff --git a/ext/opcache/jit/ir/ir_strtab.c b/ext/opcache/jit/ir/ir_strtab.c
index 476bdccef5d..b93860d0ca1 100644
--- a/ext/opcache/jit/ir/ir_strtab.c
+++ b/ext/opcache/jit/ir/ir_strtab.c
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (String table)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -13,7 +13,7 @@ typedef struct _ir_strtab_bucket {
 	uint32_t    len;
 	const char *str;
 	uint32_t    next;
-	ir_ref      val;
+	ir_str      val;
 } ir_strtab_bucket;

 static uint32_t ir_str_hash(const char *str, size_t len)
@@ -112,7 +112,7 @@ void ir_strtab_init(ir_strtab *strtab, uint32_t size, uint32_t buf_size)
 	}
 }

-ir_ref ir_strtab_find(const ir_strtab *strtab, const char *str, uint32_t len)
+ir_str ir_strtab_find(const ir_strtab *strtab, const char *str, uint32_t len)
 {
 	uint32_t h = ir_str_hash(str, len);
 	const char *data = (const char*)strtab->data;
@@ -131,7 +131,7 @@ ir_ref ir_strtab_find(const ir_strtab *strtab, const char *str, uint32_t len)
 	return 0;
 }

-ir_ref ir_strtab_lookup(ir_strtab *strtab, const char *str, uint32_t len, ir_ref val)
+ir_str ir_strtab_lookup(ir_strtab *strtab, const char *str, uint32_t len, ir_str val)
 {
 	uint32_t h = ir_str_hash(str, len);
 	char *data = (char*)strtab->data;
@@ -180,7 +180,7 @@ ir_ref ir_strtab_lookup(ir_strtab *strtab, const char *str, uint32_t len, ir_ref
 	return val;
 }

-ir_ref ir_strtab_update(ir_strtab *strtab, const char *str, uint32_t len, ir_ref val)
+ir_str ir_strtab_update(ir_strtab *strtab, const char *str, uint32_t len, ir_str val)
 {
 	uint32_t h = ir_str_hash(str, len);
 	char *data = (char*)strtab->data;
@@ -199,13 +199,13 @@ ir_ref ir_strtab_update(ir_strtab *strtab, const char *str, uint32_t len, ir_ref
 	return 0;
 }

-const char *ir_strtab_str(const ir_strtab *strtab, ir_ref idx)
+const char *ir_strtab_str(const ir_strtab *strtab, ir_str idx)
 {
 	IR_ASSERT(idx >= 0 && (uint32_t)idx < strtab->count);
 	return ((const ir_strtab_bucket*)strtab->data)[idx].str;
 }

-const char *ir_strtab_strl(const ir_strtab *strtab, ir_ref idx, size_t *len)
+const char *ir_strtab_strl(const ir_strtab *strtab, ir_str idx, size_t *len)
 {
 	const ir_strtab_bucket *b = ((const ir_strtab_bucket*)strtab->data) + idx;
 	IR_ASSERT(idx >= 0 && (uint32_t)idx < strtab->count);
diff --git a/ext/opcache/jit/ir/ir_x86.dasc b/ext/opcache/jit/ir/ir_x86.dasc
index f5efb66698d..faa14c0cfd1 100644
--- a/ext/opcache/jit/ir/ir_x86.dasc
+++ b/ext/opcache/jit/ir/ir_x86.dasc
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (x86/x86_64 native code generator based on DynAsm)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -20,11 +20,15 @@
 #ifdef IR_DEBUG
 typedef struct _ir_mem {uint64_t v;} ir_mem;

+# define IR_MEM_NONE                (ir_mem){0}
 # define IR_MEM_VAL(loc)            ((loc).v)
+# define IR_MEM_ADD(mem, offset)    IR_MEM(IR_MEM_BASE(mem), IR_MEM_OFFSET(mem) + (offset), IR_MEM_INDEX(mem), IR_MEM_SCALE(mem))
 #else
 typedef uint64_t ir_mem;

+# define IR_MEM_NONE                0
 # define IR_MEM_VAL(loc)            (loc)
+# define IR_MEM_ADD(mem, offset)    ((mem) + (offset))
 #endif

 #define IR_MEM_OFFSET(loc)          ((int32_t)(IR_MEM_VAL(loc) & 0xffffffff))
@@ -36,6 +40,8 @@ typedef uint64_t ir_mem;
 #define IR_MEM_B(base)            IR_MEM(base, 0, IR_REG_NONE, 1)
 #define IR_MEM_BO(base, offset)   IR_MEM(base, offset, IR_REG_NONE, 1)

+#define IR_MEM_I64_HI(mem)        IR_MEM_ADD(mem, 4)
+
 IR_ALWAYS_INLINE ir_mem IR_MEM(ir_reg base, int32_t offset, ir_reg index, int32_t scale)
 {
 	ir_mem mem;
@@ -543,6 +549,88 @@ IR_ALWAYS_INLINE ir_mem IR_MEM(ir_reg base, int32_t offset, ir_reg index, int32_
 ||	} while (0);
 |.endmacro

+|.macro ASM_TXT_TXT_TMEM_TXT_OP, op, op1, op2, type, op3, op4
+||	do {
+||		int32_t offset = IR_MEM_OFFSET(op3);
+||		int32_t base = IR_MEM_BASE(op3);
+||		int32_t index = IR_MEM_INDEX(op3);
+||		int32_t scale = IR_MEM_SCALE(op3);
+||  	if (index == IR_REG_NONE) {
+||			if (base == IR_REG_NONE) {
+|				op op1, op2, type [offset], op4
+||			} else {
+|				op op1, op2, type [Ra(base)+offset], op4
+||			}
+||		} else if (scale == 8) {
+||			if (base == IR_REG_NONE) {
+|				op op1, op2, type [Ra(index)*8+offset], op4
+||			} else {
+|				op op1, op2, type [Ra(base)+Ra(index)*8+offset], op4
+||			}
+||		} else if (scale == 4) {
+||			if (base == IR_REG_NONE) {
+|				op op1, op2, type [Ra(index)*4+offset], op4
+||			} else {
+|				op op1, op2, type [Ra(base)+Ra(index)*4+offset], op4
+||			}
+||		} else if (scale == 2) {
+||			if (base == IR_REG_NONE) {
+|				op op1, op2, type [Ra(index)*2+offset], op4
+||			} else {
+|				op op1, op2, type [Ra(base)+Ra(index)*2+offset], op4
+||			}
+||		} else {
+||			IR_ASSERT(scale == 1);
+||			if (base == IR_REG_NONE) {
+|				op op1, op2, type [Ra(index)+offset], op4
+||			} else {
+|				op op1, op2, type [Ra(base)+Ra(index)+offset], op4
+||			}
+||		}
+||	} while (0);
+|.endmacro
+
+|.macro ASM_TXT_TMEM_TXT_OP, op, op1, type, op2, op3
+||	do {
+||		int32_t offset = IR_MEM_OFFSET(op2);
+||		int32_t base = IR_MEM_BASE(op2);
+||		int32_t index = IR_MEM_INDEX(op2);
+||		int32_t scale = IR_MEM_SCALE(op2);
+||  	if (index == IR_REG_NONE) {
+||			if (base == IR_REG_NONE) {
+|				op op1, type [offset], op3
+||			} else {
+|				op op1, type [Ra(base)+offset], op3
+||			}
+||		} else if (scale == 8) {
+||			if (base == IR_REG_NONE) {
+|				op op1, type [Ra(index)*8+offset], op3
+||			} else {
+|				op op1, type [Ra(base)+Ra(index)*8+offset], op3
+||			}
+||		} else if (scale == 4) {
+||			if (base == IR_REG_NONE) {
+|				op op1, type [Ra(index)*4+offset], op3
+||			} else {
+|				op op1, type [Ra(base)+Ra(index)*4+offset], op3
+||			}
+||		} else if (scale == 2) {
+||			if (base == IR_REG_NONE) {
+|				op op1, type [Ra(index)*2+offset], op3
+||			} else {
+|				op op1, type [Ra(base)+Ra(index)*2+offset], op3
+||			}
+||		} else {
+||			IR_ASSERT(scale == 1);
+||			if (base == IR_REG_NONE) {
+|				op op1, type [Ra(index)+offset], op3
+||			} else {
+|				op op1, type [Ra(base)+Ra(index)+offset], op3
+||			}
+||		}
+||	} while (0);
+|.endmacro
+
 |.macro ASM_REG_OP, op, type, op1
 ||	switch (ir_type_size[type]) {
 || 		default:
@@ -568,6 +656,14 @@ IR_ALWAYS_INLINE ir_mem IR_MEM(ir_reg base, int32_t offset, ir_reg index, int32_
 |	ASM_EXPAND_OP_MEM ASM_EXPAND_TYPE_MEM, op, type, op1
 |.endmacro

+|.macro ASM_EXPAND_PUSH_TYPE_MEM, op, type, op1
+|	op dword op1
+|.endmacro
+
+|.macro ASM_MEM_PUSH_OP, op, type, op1
+|	ASM_EXPAND_OP_MEM ASM_EXPAND_PUSH_TYPE_MEM, op, type, op1
+|.endmacro
+
 |.macro ASM_REG_REG_OP, op, type, op1, op2
 ||	switch (ir_type_size[type]) {
 || 		default:
@@ -810,6 +906,302 @@ IR_ALWAYS_INLINE ir_mem IR_MEM(ir_reg base, int32_t offset, ir_reg index, int32_
 |	ASM_EXPAND_OP3_MEM ASM_AVX_REG_REG_TXT_OP, op, type, op1, op2, op3
 |.endmacro

+|.macro ASM_SSE_INT_VEC_REG_REG_OP, op, type, op1, op2
+||	if (type == IR_I8 || type == IR_U8) {
+|		op..b xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST)
+||	} else if (type == IR_I16 || type == IR_U16) {
+|		op..w xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST)
+||	} else if (type == IR_I32 || type == IR_U32) {
+|		op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST)
+||	} else if (type == IR_I64 || type == IR_U64) {
+|		op..q xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST)
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector type");
+||	}
+|.endmacro
+
+|.macro ASM_SSE_INT_VEC_REG_TXT_OP, op, type, op1, op2
+||	if (type == IR_I8 || type == IR_U8) {
+|		op..b xmm(op1-IR_REG_FP_FIRST), op2
+||	} else if (type == IR_I16 || type == IR_U16) {
+|		op..w xmm(op1-IR_REG_FP_FIRST), op2
+||	} else if (type == IR_I32 || type == IR_U32) {
+|		op..d xmm(op1-IR_REG_FP_FIRST), op2
+||	} else if (type == IR_I64 || type == IR_U64) {
+|		op..q xmm(op1-IR_REG_FP_FIRST), op2
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector type");
+||	}
+|.endmacro
+
+|.macro ASM_SSE_INT_VEC_REG_MEM_OP, op, type, op1, op2
+||	if (type == IR_I8 || type == IR_U8) {
+|		ASM_TXT_TMEM_OP op..b, xmm(op1-IR_REG_FP_FIRST), oword, op2
+||	} else if (type == IR_I16 || type == IR_U16) {
+|		ASM_TXT_TMEM_OP op..w, xmm(op1-IR_REG_FP_FIRST), oword, op2
+||	} else if (type == IR_I32 || type == IR_U32) {
+|		ASM_TXT_TMEM_OP op..d, xmm(op1-IR_REG_FP_FIRST), oword, op2
+||	} else if (type == IR_I64 || type == IR_U64) {
+|		ASM_TXT_TMEM_OP op..q, xmm(op1-IR_REG_FP_FIRST), oword, op2
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector type");
+||	}
+|.endmacro
+
+|.macro ASM_SSE_FP_VEC_REG_REG_OP, op, type, op1, op2
+|	ASM_SSE2_REG_REG_OP op, type, op1, op2
+|.endmacro
+
+|.macro ASM_SSE_FP_VEC_REG_TXT_OP, op, type, op1, op2
+||	if (type == IR_DOUBLE) {
+|		op..d xmm(op1-IR_REG_FP_FIRST), op2
+||	} else {
+||		IR_ASSERT(type == IR_FLOAT);
+|		op..s xmm(op1-IR_REG_FP_FIRST), op2
+||	}
+|.endmacro
+
+|.macro ASM_SSE_FP_VEC_REG_TXT_TXT_OP, op, type, op1, op2, op3
+||	if (type == IR_DOUBLE) {
+|		op..d xmm(op1-IR_REG_FP_FIRST), op2, op3
+||	} else {
+||		IR_ASSERT(type == IR_FLOAT);
+|		op..s xmm(op1-IR_REG_FP_FIRST), op2, op3
+||	}
+|.endmacro
+
+|.macro ASM_SSE_FP_VEC_REG_MEM_OP, op, type, op1, op2
+||	if (type == IR_DOUBLE) {
+|		ASM_TXT_TMEM_OP op..d, xmm(op1-IR_REG_FP_FIRST), oword, op2
+||	} else {
+||		IR_ASSERT(type == IR_FLOAT);
+|		ASM_TXT_TMEM_OP op..s, xmm(op1-IR_REG_FP_FIRST), oword, op2
+||	}
+|.endmacro
+
+|.macro ASM_SSE_FP_VEC_REG_MEM_TXT_OP, op, type, op1, op2, op3
+||	if (type == IR_DOUBLE) {
+|		ASM_TXT_TMEM_TXT_OP op..d, xmm(op1-IR_REG_FP_FIRST), oword, op2, op3
+||	} else {
+||		IR_ASSERT(type == IR_FLOAT);
+|		ASM_TXT_TMEM_TXT_OP op..s, xmm(op1-IR_REG_FP_FIRST), oword, op2, op3
+||	}
+|.endmacro
+
+|.macro ASM_AVX_INT_VEC_REG_REG_REG_OP, op, type, width, op1, op2, op3
+||	if (width <= 16) {
+||		if (type == IR_I8 || type == IR_U8) {
+|			op..b xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), xmm(op3-IR_REG_FP_FIRST)
+||		} else if (type == IR_I16 || type == IR_U16) {
+|			op..w xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), xmm(op3-IR_REG_FP_FIRST)
+||		} else if (type == IR_I32 || type == IR_U32) {
+|			op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), xmm(op3-IR_REG_FP_FIRST)
+||		} else if (type == IR_I64 || type == IR_U64) {
+|			op..q xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), xmm(op3-IR_REG_FP_FIRST)
+||		} else {
+||			IR_ASSERT(0 && "unsupported vector type");
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_I8 || type == IR_U8) {
+|			op..b ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST)
+||		} else if (type == IR_I16 || type == IR_U16) {
+|			op..w ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST)
+||		} else if (type == IR_I32 || type == IR_U32) {
+|			op..d ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST)
+||		} else if (type == IR_I64 || type == IR_U64) {
+|			op..q ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST)
+||		} else {
+||			IR_ASSERT(0 && "unsupported vector type");
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_INT_VEC_REG_REG_TXT_OP, op, type, width, op1, op2, op3
+||	if (width <= 16) {
+||		if (type == IR_I8 || type == IR_U8) {
+|			op..b xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword op3
+||		} else if (type == IR_I16 || type == IR_U16) {
+|			op..w xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword op3
+||		} else if (type == IR_I32 || type == IR_U32) {
+|			op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword op3
+||		} else if (type == IR_I64 || type == IR_U64) {
+|			op..q xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword op3
+||		} else {
+||			IR_ASSERT(0 && "unsupported vector type");
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_I8 || type == IR_U8) {
+|			op..b ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword op3
+||		} else if (type == IR_I16 || type == IR_U16) {
+|			op..w ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword op3
+||		} else if (type == IR_I32 || type == IR_U32) {
+|			op..d ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword op3
+||		} else if (type == IR_I64 || type == IR_U64) {
+|			op..q ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword op3
+||		} else {
+||			IR_ASSERT(0 && "unsupported vector type");
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_INT_VEC_REG_REG_MEM_OP, op, type, width, op1, op2, op3
+||	if (width <= 16) {
+||		if (type == IR_I8 || type == IR_U8) {
+|			ASM_TXT_TXT_TMEM_OP op..b, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3
+||		} else if (type == IR_I16 || type == IR_U16) {
+|			ASM_TXT_TXT_TMEM_OP op..w, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3
+||		} else if (type == IR_I32 || type == IR_U32) {
+|			ASM_TXT_TXT_TMEM_OP op..d, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3
+||		} else if (type == IR_I64 || type == IR_U64) {
+|			ASM_TXT_TXT_TMEM_OP op..q, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3
+||		} else {
+||			IR_ASSERT(0 && "unsupported vector type");
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_I8 || type == IR_U8) {
+|			ASM_TXT_TXT_TMEM_OP op..b, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3
+||		} else if (type == IR_I16 || type == IR_U16) {
+|			ASM_TXT_TXT_TMEM_OP op..w, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3
+||		} else if (type == IR_I32 || type == IR_U32) {
+|			ASM_TXT_TXT_TMEM_OP op..d, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3
+||		} else if (type == IR_I64 || type == IR_U64) {
+|			ASM_TXT_TXT_TMEM_OP op..q, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3
+||		} else {
+||			IR_ASSERT(0 && "unsupported vector type");
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_FP_VEC_REG_REG_REG_OP, op, type, width, op1, op2, op3
+||	if (width <= 16) {
+|		ASM_AVX_REG_REG_REG_OP op, type, op1, op2, op3
+||	} else if (width == 32) {
+||		if (type == IR_DOUBLE) {
+|			op..d ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST)
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST)
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_FP_VEC_REG_REG_TXT_OP, op, type, width, op1, op2, op3
+||	if (width <= 16) {
+||		if (type == IR_DOUBLE) {
+|			op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), op3
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), op3
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_DOUBLE) {
+|			op..d ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), op3
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), op3
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_FP_VEC_REG_REG_TXT_TXT_OP, op, type, width, op1, op2, op3, op4
+||	if (width <= 16) {
+||		if (type == IR_DOUBLE) {
+|			op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), op3, op4
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), op3, op4
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_DOUBLE) {
+|			op..d ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), op3, op4
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), op3, op4
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_FP_VEC_REG_REG_MEM_OP, op, type, width, op1, op2, op3
+||	if (width <= 16) {
+||		if (type == IR_DOUBLE) {
+|			ASM_TXT_TXT_TMEM_OP op..d, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			ASM_TXT_TXT_TMEM_OP op..s, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_DOUBLE) {
+|			ASM_TXT_TXT_TMEM_OP op..d, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			ASM_TXT_TXT_TMEM_OP op..s, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_AVX_FP_VEC_REG_REG_MEM_TXT_OP, op, type, width, op1, op2, op3, op4
+||	if (width <= 16) {
+||		if (type == IR_DOUBLE) {
+|			ASM_TXT_TXT_TMEM_TXT_OP op..d, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3, op4
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			ASM_TXT_TXT_TMEM_TXT_OP op..s, xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), oword, op3, op4
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_DOUBLE) {
+|			ASM_TXT_TXT_TMEM_TXT_OP op..d, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3, op4
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			ASM_TXT_TXT_TMEM_TXT_OP op..s, ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), yword, op3, op4
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
+|.macro ASM_SSE_FP_VEC_REG_REG_TXT_OP, op, type, op1, op2, op3
+||	if (type == IR_DOUBLE) {
+|		op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), op3
+||	} else {
+||		IR_ASSERT(type == IR_FLOAT);
+|		op..s xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), op3
+||	}
+|.endmacro
+
+|.macro ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP, op, type, width, op1, op2, op3, op4
+||	if (width <= 16) {
+||		if (type == IR_DOUBLE) {
+|			op..d xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), xmm(op3-IR_REG_FP_FIRST), op4
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s xmm(op1-IR_REG_FP_FIRST), xmm(op2-IR_REG_FP_FIRST), xmm(op3-IR_REG_FP_FIRST), op4
+||		}
+||	} else if (width == 32) {
+||		if (type == IR_DOUBLE) {
+|			op..d ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST), op4
+||		} else {
+||			IR_ASSERT(type == IR_FLOAT);
+|			op..s ymm(op1-IR_REG_FP_FIRST), ymm(op2-IR_REG_FP_FIRST), ymm(op3-IR_REG_FP_FIRST), op4
+||		}
+||	} else {
+||		IR_ASSERT(0 && "unsupported vector width");
+||	}
+|.endmacro
+
 |.macro ASM_FP_REG_REG_OP, op, type, op1, op2
 ||	if (ctx->mflags & IR_X86_AVX) {
 |		ASM_SSE2_REG_REG_OP v..op, type, op1, op2
@@ -891,9 +1283,25 @@ typedef struct _ir_backend_data {
 	bool               double_abs_const;
 	bool               float_abs_const;
 	bool               double_zero_const;
-	bool               u2d_const;
-	bool               u2f_const;
+	bool               u2d_const;      /* 0, 0x41e00000 */
+	bool               ull2d_const;    /* 0, 0x43e00000 */
+	bool               ull2f_const;    /*    0x5f000000 */
+	bool               ull2fp_const;   /*    0x5f800000 */
 	bool               resolved_label_syms;
+#ifdef _WIN64
+	bool               chkstk_addr;
+#endif
+#ifdef IR_SIMD
+	bool               v_u32_to_d;     /* 0x43300000, 0 */
+	bool               v_u64_to_d;
+	bool               v_i64_to_d;
+	bool               v_u32_to_f;
+	bool               v_d_to_u32;
+	bool               v_f_to_u32;
+	bool               v_u32_to_u16;
+	bool               v_u32_to_u8;
+	bool               v_u16_to_u8;
+#endif
 } ir_backend_data;

 typedef struct _ir_x86_64_sysv_va_list {
@@ -951,7 +1359,7 @@ const char *ir_reg_name(int8_t reg, ir_type type)
 	if (type == IR_VOID) {
 		type = (reg < IR_REG_FP_FIRST) ? IR_ADDR : IR_DOUBLE;
 	}
-	if (IR_IS_TYPE_FP(type) || ir_type_size[type] == 8) {
+	if (!IR_IS_TYPE_INT(type) || ir_type_size[type] == 8) {
 		return _ir_reg_name[reg];
 	} else if (ir_type_size[type] == 4) {
 		return _ir_reg_name32[reg];
@@ -963,6 +1371,23 @@ const char *ir_reg_name(int8_t reg, ir_type type)
 	}
 }

+void ir_dump_reg(const ir_ctx *ctx, int8_t reg, ir_ref ref, bool store, FILE *f)
+{
+	if (reg != IR_REG_NONE) {
+#if IR_X86_I64
+		if (ctx->rules
+		 && (ctx->rules[ref] & IR_TWO_REGS)) {
+			fprintf(f, " {%%%s,%%%s%s}",
+				_ir_reg_name32[IR_REG_I64_LO(reg)],
+				_ir_reg_name32[IR_REG_I64_HI(reg)],
+				(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? (store ? ":store" : ":load") : "");
+		} else
+#endif
+		fprintf(f, " {%%%s%s}", ir_reg_name(IR_REG_NUM(reg), ctx->ir_base[ref].type),
+			(reg & (IR_REG_SPILL_STORE|IR_REG_SPILL_SPECIAL)) ? (store ? ":store" : ":load") : "");
+	}
+}
+
 /* Calling Conventions */
 #ifdef IR_TARGET_X64

@@ -1010,12 +1435,18 @@ const ir_call_conv_dsc ir_call_conv_x86_64_ms = {
 	32,          /* shadow_store_size       */
 	4,           /* int_param_regs_count    */
 	4,           /* fp_param_regs_count     */
+	4,           /* vector_param_regs_count */
 	IR_REG_RAX,  /* int_ret_reg             */
+	IR_REG_NONE, /* int_ret2_reg            */
 	IR_REG_XMM0, /* fp_ret_reg              */
+	IR_REG_NONE, /* fp_ret2_reg             */
+	IR_REG_XMM0, /* vector_ret_reg          */
+	IR_REG_NONE, /* vector_ret2_reg         */
 	IR_REG_NONE, /* fp_varargs_reg          */
 	IR_REG_SCRATH_X86_64_MS,
 	(const int8_t[4]){IR_REG_RCX, IR_REG_RDX, IR_REG_R8, IR_REG_R9},
 	(const int8_t[4]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2, IR_REG_XMM3},
+	(const int8_t[4]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2, IR_REG_XMM3},
 	IR_REGSET(IR_REG_RBX) | IR_REGSET(IR_REG_RBP) | IR_REGSET(IR_REG_RSI) | IR_REGSET(IR_REG_RDI) |
 		IR_REGSET_INTERVAL(IR_REG_R12, IR_REG_R15) | IR_REGSET_INTERVAL(IR_REG_XMM6, IR_REG_XMM15),
 };
@@ -1028,11 +1459,18 @@ const ir_call_conv_dsc ir_call_conv_x86_64_sysv = {
 	0,           /* shadow_store_size       */
 	6,           /* int_param_regs_count    */
 	8,           /* fp_param_regs_count     */
+	8,           /* vector_param_regs_count */
 	IR_REG_RAX,  /* int_ret_reg             */
+	IR_REG_RDX,  /* int_ret2_reg            */
 	IR_REG_XMM0, /* fp_ret_reg              */
+	IR_REG_XMM1, /* fp_ret2_reg             */
+	IR_REG_XMM0, /* vector_ret_reg          */
+	IR_REG_XMM1, /* vector_ret2_reg         */
 	IR_REG_RAX,  /* fp_varargs_reg          */
 	IR_REG_SCRATH_X86_64_SYSV,
 	(const int8_t[6]){IR_REG_RDI, IR_REG_RSI, IR_REG_RDX, IR_REG_RCX, IR_REG_R8, IR_REG_R9},
+	(const int8_t[8]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2, IR_REG_XMM3,
+	                  IR_REG_XMM4, IR_REG_XMM5, IR_REG_XMM6, IR_REG_XMM7},
 	(const int8_t[8]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2, IR_REG_XMM3,
 	                  IR_REG_XMM4, IR_REG_XMM5, IR_REG_XMM6, IR_REG_XMM7},
 	IR_REGSET(IR_REG_RBX) | IR_REGSET(IR_REG_RBP) | IR_REGSET_INTERVAL(IR_REG_R12, IR_REG_R15),
@@ -1047,8 +1485,13 @@ const ir_call_conv_dsc ir_call_conv_x86_64_preserve_none = {
 	0,           /* shadow_store_size       */
 	12,          /* int_param_regs_count    */
 	8,           /* fp_param_regs_count     */
+	8,           /* vector_param_regs_count */
 	IR_REG_RAX,  /* int_ret_reg             */
+	IR_REG_RDX,  /* int_ret2_reg            */
 	IR_REG_XMM0, /* fp_ret_reg              */
+	IR_REG_XMM1, /* fp_ret2_reg             */
+	IR_REG_XMM0, /* vector_ret_reg          */
+	IR_REG_XMM1, /* vector_ret2_reg         */
 	IR_REG_RAX,  /* fp_varargs_reg          */
 	IR_REG_SCRATH_X86_64_PN,
 	(const int8_t[12]){IR_REG_R12, IR_REG_R13, IR_REG_R14, IR_REG_R15,
@@ -1056,6 +1499,8 @@ const ir_call_conv_dsc ir_call_conv_x86_64_preserve_none = {
 	                   IR_REG_R11, IR_REG_RAX},
 	(const int8_t[8]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2, IR_REG_XMM3,
 	                  IR_REG_XMM4, IR_REG_XMM5, IR_REG_XMM6, IR_REG_XMM7},
+	(const int8_t[8]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2, IR_REG_XMM3,
+	                  IR_REG_XMM4, IR_REG_XMM5, IR_REG_XMM6, IR_REG_XMM7},
 	IR_REGSET(IR_REG_RBP),

 };
@@ -1086,12 +1531,18 @@ const ir_call_conv_dsc ir_call_conv_x86_cdecl = {
 	0,           /* shadow_store_size       */
 	0,           /* int_param_regs_count    */
 	0,           /* fp_param_regs_count     */
+	3,           /* vector_param_regs_count */
 	IR_REG_RAX,  /* int_ret_reg             */
+	IR_REG_RDX,  /* int_ret2_reg            */
 	IR_REG_NONE, /* fp_ret_reg              */
+	IR_REG_NONE, /* fp_ret2_reg             */
+	IR_REG_XMM0, /* vector_ret_reg          */
+	IR_REG_NONE, /* vector_ret2_reg         */
 	IR_REG_NONE, /* fp_varargs_reg          */
 	IR_REG_SCRATCH_X86,
 	NULL,
 	NULL,
+	(const int8_t[3]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2},
 	IR_REGSET(IR_REG_RBX) | IR_REGSET(IR_REG_RBP) | IR_REGSET(IR_REG_RSI) | IR_REGSET(IR_REG_RDI),
 };

@@ -1103,12 +1554,18 @@ const ir_call_conv_dsc ir_call_conv_x86_fastcall = {
 	0,           /* shadow_store_size       */
 	2,           /* int_param_regs_count    */
 	0,           /* fp_param_regs_count     */
+	3,           /* vector_param_regs_count */
 	IR_REG_RAX,  /* int_ret_reg             */
+	IR_REG_RDX,  /* int_ret2_reg            */
 	IR_REG_NONE, /* fp_ret_reg              */
+	IR_REG_NONE, /* fp_ret2_reg             */
+	IR_REG_XMM0, /* vector_ret_reg          */
+	IR_REG_NONE, /* vector_ret2_reg         */
 	IR_REG_NONE, /* fp_varargs_reg          */
 	IR_REG_SCRATCH_X86,
 	(const int8_t[4]){IR_REG_RCX, IR_REG_RDX},
 	NULL,
+	(const int8_t[3]){IR_REG_XMM0, IR_REG_XMM1, IR_REG_XMM2},
 	IR_REGSET(IR_REG_RBX) | IR_REGSET(IR_REG_RBP) | IR_REGSET(IR_REG_RSI) | IR_REGSET(IR_REG_RDI),
 };

@@ -1125,6 +1582,8 @@ const ir_call_conv_dsc ir_call_conv_x86_fastcall = {
 	_(TEST_INT)            \
 	_(SETCC_INT)           \
 	_(TESTCC_INT)          \
+	_(TEST_BIT)            \
+	_(TESTCC_BIT)          \
 	_(LEA_OB)              \
 	_(LEA_SI)              \
 	_(LEA_SIB)             \
@@ -1166,13 +1625,16 @@ const ir_call_conv_dsc ir_call_conv_x86_fastcall = {
 	_(CMP_AND_BRANCH_INT)  \
 	_(CMP_AND_BRANCH_FP)   \
 	_(TEST_AND_BRANCH_INT) \
+	_(TEST_AND_BRANCH_BIT) \
 	_(JCC_INT)             \
 	_(COND_TEST_INT)       \
+	_(COND_TEST_BIT)       \
 	_(COND_CMP_INT)        \
 	_(COND_CMP_FP)         \
 	_(GUARD_CMP_INT)       \
 	_(GUARD_CMP_FP)        \
 	_(GUARD_TEST_INT)      \
+	_(GUARD_TEST_BIT)      \
 	_(GUARD_JCC_INT)       \
 	_(GUARD_OVERFLOW)      \
 	_(OVERFLOW_AND_BRANCH) \
@@ -1205,7 +1667,71 @@ const ir_call_conv_dsc ir_call_conv_x86_fastcall = {
 	_(SSE_TRUNC)           \
 	_(SSE_NEARBYINT)       \
 	_(BIT_OP)              \
+	_(AND_ZEXT)            \
 	_(IGOTO_DUP)           \
+	_(TLS_LOAD)            \
+	_(TLS_STORE)           \
+
+#if IR_SIMD
+# define IR_RULES_SIMD(_)  \
+	_(VECTOR_OP)           \
+	_(VECTOR_BINOP_SSE2)   \
+	_(VECTOR_BINOP_AVX)    \
+	_(VECTOR_BINOP_EXPAND) \
+	_(VECTOR_EXT)          \
+	_(VECTOR_TRUNC)        \
+	_(VECTOR_FP2FP)        \
+	_(VECTOR_INT2FP)       \
+	_(VECTOR_FP2INT)       \
+	_(SHUFPD_11)           \
+	_(SHUFPD_22)           \
+	_(SHUFPD_12)           \
+	_(SHUFPD_21)           \
+	_(MOVSD_12)            \
+	_(SHUFPS_11)           \
+	_(SHUFPS_22)           \
+	_(SHUFPS_12)           \
+	_(SHUFPS_21)           \
+	_(SHUFPS_12_0)         \
+	_(SHUFPS_12_1)         \
+	_(SHUFPS_12_2)         \
+	_(SHUFPS_1_21)         \
+	_(SHUFPS_2_12)         \
+	_(BLENDPS_12)          \
+
+#endif
+
+#if IR_X86_I64
+# define IR_RULES_I64(_)   \
+	_(CMP_I64)             \
+	_(BINOP_I64)           \
+	_(MUL_I64)             \
+	_(MUL_OV_I64)          \
+	_(BINOP_HELPER_I64)    \
+	_(OP_I64)              \
+	_(SHIFT_I64)           \
+	_(SHIFT_CONST_I64)     \
+	_(SEXT_I64)            \
+	_(ZEXT_I64)            \
+	_(BITCAST_I64)         \
+	_(FP2INT_I64)          \
+	_(INT2FP_I64)          \
+	_(BIT_COUNT_I64)       \
+	_(BIT_COUNT_HELPER_I64)\
+	_(MIN_MAX_I64)         \
+	_(IF_I64)              \
+	_(CMP_AND_BRANCH_I64)  \
+	_(GUARD_I64)           \
+	_(GUARD_CMP_I64)       \
+	_(COND_I64)            \
+	_(COND_I64_CMP_INT)    \
+	_(COND_I64_CMP_FP)     \
+	_(COND_CMP_I64)        \
+	_(PARAM_I64)           \
+	_(LOAD_I64)            \
+	_(RETURN_I64)          \
+
+#endif

 #define IR_LEA_FIRST IR_LEA_OB
 #define IR_LEA_LAST  IR_LEA_O_SYM
@@ -1217,13 +1743,25 @@ const ir_call_conv_dsc ir_call_conv_x86_fastcall = {
 enum _ir_rule {
 	IR_FIRST_RULE = IR_LAST_OP,
 	IR_RULES(IR_RULE_ENUM)
+#if IR_SIMD
+	IR_RULES_SIMD(IR_RULE_ENUM)
+#endif
+#if IR_X86_I64
+	IR_RULES_I64(IR_RULE_ENUM)
+#endif
 	IR_LAST_RULE
 };

 #define IR_RULE_NAME(name)  #name,
-const char *ir_rule_name[IR_LAST_OP] = {
+const char *ir_rule_name[IR_LAST_RULE] = {
 	NULL,
 	IR_RULES(IR_RULE_NAME)
+#if IR_SIMD
+	IR_RULES_SIMD(IR_RULE_NAME)
+#endif
+#if IR_X86_I64
+	IR_RULES_I64(IR_RULE_NAME)
+#endif
 	NULL
 };

@@ -1252,6 +1790,13 @@ static bool ir_may_fuse_imm(ir_ctx *ctx, const ir_insn *val_insn)
 	}
 }

+#if IR_SIMD
+static bool ir_may_fuse_load_vector(ir_ctx *ctx, ir_type type)
+{
+	return IR_VECTOR_SIZE(type) == 16 || (IR_VECTOR_SIZE(type) == 32 && (ctx->mflags & IR_X86_AVX));
+}
+#endif
+
 /* register allocation */
 static int ir_add_const_tmp_reg(ir_ctx *ctx, ir_ref ref, uint32_t num, int n, ir_target_constraints *constraints)
 {
@@ -1274,6 +1819,9 @@ int ir_get_target_constraints(ir_ctx *ctx, ir_ref ref, ir_target_constraints *co
 	const ir_proto_t *proto;
 	const ir_call_conv_dsc *cc;
 	ir_ref next;
+#if IR_SIMD
+	ir_type type;
+#endif

 	constraints->def_reg = IR_REG_NONE;
 	constraints->hints_count = 0;
@@ -1380,6 +1928,20 @@ op2_const:
 				n++;
 			}
 			break;
+		case IR_TEST_BIT:
+			insn = &ctx->ir_base[ref];
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			if (IR_IS_CONST_REF(insn->op1)) {
+				const ir_insn *val_insn = &ctx->ir_base[insn->op1];
+				constraints->tmp_regs[0] = IR_TMP_REG(1, val_insn->type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			} else if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_ADDR, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			} else if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			}
+			break;
 		case IR_CMP_FP:
 			insn = &ctx->ir_base[ref];
 			if (!(rule & IR_FUSED)) {
@@ -1410,6 +1972,7 @@ op2_const:
 			IR_FALLTHROUGH;
 		case IR_COND_CMP_INT:
 		case IR_COND_TEST_INT:
+		case IR_COND_TEST_BIT:
 			insn = &ctx->ir_base[ref];
 			if (IR_IS_TYPE_INT(insn->type)) {
 				if (IR_IS_CONST_REF(insn->op3) || ir_rule(ctx, insn->op3) == IR_STATIC_ALLOCA) {
@@ -1505,10 +2068,12 @@ op2_const:
 			n++;
 			break;
 		case IR_ARGVAL:
+			flags = IR_OP1_SHOULD_BE_IN_REG;
 			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_RSI, IR_DEF_SUB_REF - IR_SUB_REFS_COUNT, IR_USE_SUB_REF);
 			constraints->tmp_regs[1] = IR_SCRATCH_REG(IR_REG_RDI, IR_DEF_SUB_REF - IR_SUB_REFS_COUNT, IR_USE_SUB_REF);
 			constraints->tmp_regs[2] = IR_SCRATCH_REG(IR_REG_RCX, IR_DEF_SUB_REF - IR_SUB_REFS_COUNT, IR_USE_SUB_REF);
 			n = 3;
+			ctx->flags2 |= IR_HAS_MEMCPY;
 			break;
 		case IR_CALL:
 			insn = &ctx->ir_base[ref];
@@ -1518,9 +2083,14 @@ op2_const:
 				if (IR_IS_TYPE_INT(insn->type)) {
 					constraints->def_reg = cc->int_ret_reg;
 				} else {
-					IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+					IR_ASSERT(IR_IS_TYPE_FP(insn->type) || IR_IS_TYPE_VECTOR(insn->type));
 #ifdef IR_TARGET_X86
 					if (cc->fp_ret_reg == IR_REG_NONE) {
+# if IR_SIMD
+						if (IR_IS_TYPE_VECTOR(insn->type)) {
+							constraints->def_reg = cc->vector_ret_reg;
+						} else
+# endif
 						ctx->flags2 |= IR_HAS_FP_RET_SLOT;
 					} else
 #endif
@@ -1550,6 +2120,11 @@ op2_const:
 				constraints->tmp_regs[n] = IR_SCRATCH_REG(cc->fp_varargs_reg, IR_LOAD_SUB_REF, IR_USE_SUB_REF);
 				n++;
 			}
+			if (IR_IS_CONST_REF(insn->op2)
+			 && ctx->ir_base[insn->op2].op == IR_FUNC
+			 && ctx->func_name == ctx->ir_base[insn->op2].val.name) {
+				ctx->flags2 |= IR_RECURSIVE_TAILCALL;
+			}
 			if (insn->inputs_count > 2) {
 get_arg_hints:
 				constraints->hints[2] = IR_REG_NONE;
@@ -1560,6 +2135,12 @@ get_arg_hints:
 				}
 			}
 			flags = IR_USE_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
+#if IR_X86_I64
+			if (insn->op == IR_CALL && (insn->type == IR_I64 || insn->type == IR_U64)) {
+				flags |= IR_HINT_TWO_REGS;
+				constraints->def_reg = IR_REG_I64_PAIR(cc->int_ret_reg, cc->int_ret2_reg);
+			}
+#endif
 			break;
 		case IR_BINOP_SSE2:
 			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG;
@@ -1626,7 +2207,7 @@ get_arg_hints:
 			insn = &ctx->ir_base[ref];
 			constraints->tmp_regs[0] = IR_TMP_REG(2, insn->type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
 			n = 1;
-			if (ir_type_size[insn->type] == 8) {
+			if (ir_type_size[ctx->ir_base[insn->op1].type] == 8) {
 				constraints->tmp_regs[1] = IR_TMP_REG(3, insn->type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
 				n = 2;
 			}
@@ -1635,6 +2216,7 @@ get_arg_hints:
 		case IR_COPY_FP:
 		case IR_SEXT:
 		case IR_ZEXT:
+		case IR_AND_ZEXT:
 		case IR_TRUNC:
 		case IR_PROTO:
 		case IR_FP2FP:
@@ -1649,7 +2231,25 @@ get_arg_hints:
 			}
 			break;
 		case IR_FP2INT:
-			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (insn->type == IR_U64) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+				if (IR_IS_CONST_REF(insn->op1)) {
+					constraints->tmp_regs[n] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				}
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			} else if (sizeof(void*) == 4 && insn->type == IR_U32) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+				if (IR_IS_CONST_REF(insn->op1)) {
+					constraints->tmp_regs[n] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				}
+				ctx->flags2 |= IR_HAS_FP_RET_SLOT;
+			} else {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
+			}
 			break;
 		case IR_INT2FP:
 			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
@@ -1662,6 +2262,8 @@ get_arg_hints:
 			 && ir_type_size[ctx->ir_base[insn->op1].type] >= sizeof(void*)) {
 				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
 				n++;
+			} else if (ctx->ir_base[insn->op1].type == IR_U32 || ir_type_size[ctx->ir_base[insn->op1].type] < 4) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
 			}
 			break;
 		case IR_ABS_INT:
@@ -1705,6 +2307,13 @@ get_arg_hints:
 		case IR_RETURN_FP:
 			cc = ir_get_call_conv_dsc(ctx->flags);
 #ifdef IR_TARGET_X86
+# if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(ctx->ir_base[ctx->ir_base[ref].op2].type)) {
+				flags = IR_OP2_SHOULD_BE_IN_REG;
+				constraints->hints[2] = cc->vector_ret_reg;
+				constraints->hints_count = 3;
+			} else
+# endif
 			if (cc->fp_ret_reg != IR_REG_NONE)
 #endif
 			{
@@ -1768,6040 +2377,14114 @@ get_arg_hints:
 				n = 1;
 			}
 			break;
-	}
-	constraints->tmps_count = n;
-
-	return flags;
-}
-
-/* instruction selection */
-static uint32_t ir_match_insn(ir_ctx *ctx, ir_ref ref);
-static bool ir_match_try_fuse_load(ir_ctx *ctx, ir_ref ref, ir_ref root);
-
-static void ir_swap_ops(ir_insn *insn)
-{
-	SWAP_REFS(insn->op1, insn->op2);
-}
-
-static bool ir_match_try_revert_lea_to_add(ir_ctx *ctx, ir_ref ref)
-{
-	ir_insn *insn = &ctx->ir_base[ref];
-
-	/* TODO: This optimization makes sense only if the other operand is killed */
-	if (insn->op1 == insn->op2) {
-		/* pass */
-	} else if (ir_match_try_fuse_load(ctx, insn->op2, ref)) {
-		ctx->rules[ref] = IR_BINOP_INT | IR_MAY_SWAP;
-		return 1;
-	} else if (ir_match_try_fuse_load(ctx, insn->op1, ref)) {
-		/* swap for better load fusion */
-		ir_swap_ops(insn);
-		ctx->rules[ref] = IR_BINOP_INT | IR_MAY_SWAP;
-		return 1;
-	}
-	return 0;
-}
-
-static void ir_match_fuse_addr(ir_ctx *ctx, ir_ref addr_ref)
-{
-	if (!IR_IS_CONST_REF(addr_ref)) {
-		uint32_t rule = ctx->rules[addr_ref];
-
-		if (!rule) {
-			ctx->rules[addr_ref] = rule = ir_match_insn(ctx, addr_ref);
-		}
-		if (rule >= IR_LEA_FIRST && rule <= IR_LEA_LAST) {
-			ir_use_list *use_list;
-			ir_ref j;
-
-			if (rule == IR_LEA_IB && ir_match_try_revert_lea_to_add(ctx, addr_ref)) {
-				return;
-			}
-
-			use_list = &ctx->use_lists[addr_ref];
-			j = use_list->count;
-			if (j > 1) {
-				/* check if address is used only in LOAD and STORE */
-				ir_ref *p = &ctx->use_edges[use_list->refs];
-
-				do {
-					ir_insn *insn = &ctx->ir_base[*p];
-					if (insn->op != IR_LOAD
-					 && insn->op != IR_LOAD_v
-					 && ((insn->op != IR_STORE && insn->op != IR_STORE_v) || insn->op3 == addr_ref)) {
-						return;
-					}
-					p++;
-				} while (--j);
+		case IR_TLS_LOAD:
+			insn = &ctx->ir_base[ref];
+			flags = IR_USE_MUST_BE_IN_REG;
+			if (ctx->ir_base[insn->op2].op2 >= 0 /* we need an extra register to access dynamic TLS */
+			 && !IR_IS_TYPE_INT(insn->type)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(3, IR_ADDR, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-			ctx->rules[addr_ref] = IR_FUSED | IR_SIMPLE | rule;
-		}
-	}
-}
-
-static bool ir_match_may_fuse_SI(ir_ctx *ctx, ir_ref ref, ir_ref use)
-{
-	ir_insn *op2_insn, *insn = &ctx->ir_base[use];
-
-	if (insn->op == IR_ADD) {
-		if (insn->op1 == ref) {
-			if (IR_IS_CONST_REF(insn->op2)) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (IR_IS_SYM_CONST(op2_insn->op)) {
-					if (ir_may_fuse_addr(ctx, op2_insn)) {
-						return 1; // LEA_SI_O
-					}
-				} else if (IR_IS_SIGNED_32BIT(op2_insn->val.i64)) {
-					return 1; // LEA_SI_O
+			break;
+		case IR_TLS_STORE:
+			insn = &ctx->ir_base[ref];
+			flags = IR_OP3_MUST_BE_IN_REG;
+			if (IR_IS_CONST_REF(insn->op3)) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
+					n = ir_add_const_tmp_reg(ctx, insn->op3, 3, n, constraints);
+				} else {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, ctx->ir_base[insn->op3].type, IR_LOAD_SUB_REF, IR_USE_SUB_REF);
+					n++;
 				}
-			} else if (insn->op2 != ref) {
-				return 1; // LEA_SI_B or LEA_SI_OB
 			}
-		} else if (insn->op2 == ref && insn->op1 != insn->op2) {
-			return 1; // LEA_B_SI or LEA_OB_SI
-		}
-	}
-	return 0;
-}
-
-static bool ir_match_fuse_addr_all_useges(ir_ctx *ctx, ir_ref ref)
-{
-	uint32_t rule = ctx->rules[ref];
-	ir_use_list *use_list;
-	ir_ref n, *p, use;
-
-	if (rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
-		return 1;
-	} else if (!rule) {
-		ir_insn *insn = &ctx->ir_base[ref];
-
-		IR_ASSERT(IR_IS_TYPE_INT(insn->type) && ir_type_size[insn->type] >= 4);
-		if (insn->op == IR_MUL
-		 && IR_IS_CONST_REF(insn->op2)) {
-			insn = &ctx->ir_base[insn->op2];
-			if (!IR_IS_SYM_CONST(insn->op)
-			 &&	(insn->val.u64 == 2 || insn->val.u64 == 4 || insn->val.u64 == 8)) {
-				ctx->rules[ref] = IR_LEA_SI;
-
-				use_list = &ctx->use_lists[ref];
-				n = use_list->count;
-				IR_ASSERT(n > 1);
-				p = &ctx->use_edges[use_list->refs];
-				for (; n > 0; p++, n--) {
-					use = *p;
-					if (!ir_match_may_fuse_SI(ctx, ref, use)) {
-						return 0;
-					}
-				}
-
-				return 1;
+			if (ctx->ir_base[insn->op2].op2 >= 0) { /* we need an extra register to access dynamic TLS */
+				constraints->tmp_regs[n] = IR_TMP_REG(0, IR_ADDR, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
 			}
-		}
-	}
-
-	return 0;
-}
-
-/* A naive check if there is a STORE or CALL between this LOAD and the fusion root */
-static bool ir_match_has_mem_deps(ir_ctx *ctx, ir_ref ref, ir_ref root)
-{
-	if (ref + 1 != root) {
-		ir_ref pos = ctx->prev_ref[root];
-
-		do {
-			ir_insn *insn = &ctx->ir_base[pos];
-
-			if (insn->op == IR_STORE || insn->op == IR_STORE_v || insn->op == IR_VSTORE || insn->op == IR_VSTORE_v) {
-				// TODO: check if LOAD and STORE addresses may alias
-				return 1;
-			} else if (insn->op == IR_CALL) {
-				return 1;
+			break;
+#if IR_X86_I64
+		case IR_BINOP_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_DEF_REUSES_OP1_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG;
+			break;
+		case IR_SHIFT_I64:
+			flags = IR_DEF_REUSES_OP1_REG | IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG;
+			constraints->hints[1] = IR_REG_NONE;
+			constraints->hints[2] = IR_REG_RCX;
+			constraints->hints_count = 3;
+			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_RCX, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			break;
+		case IR_CMP_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+			insn = &ctx->ir_base[ref];
+			if (insn->op == IR_EQ || insn->op == IR_NE) {
+				constraints->tmp_regs[0] = IR_TMP_REG(3, IR_U32, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+				break;
 			}
-			pos = ctx->prev_ref[pos];
-		} while (ref != pos);
-	}
-	return 0;
-}
-
-/* A naive check if anything that emits code, and so clobbers the flags, is
- * scheduled between the flags setting instruction and the fusion root */
-static bool ir_match_has_flags_deps(ir_ctx *ctx, ir_ref ref, ir_ref root)
-{
-	ir_ref pos = ctx->prev_ref[root];
-
-	while (pos > ref) {
-		if (ctx->ir_base[pos].op != IR_SNAPSHOT) {
-			return 1;
-		}
-		pos = ctx->prev_ref[pos];
-	}
-	return pos != ref;
-}
-
-static void ir_match_fuse_load(ir_ctx *ctx, ir_ref ref, ir_ref root)
-{
-	if (ir_in_same_block(ctx, ref) &&
-	    (ctx->ir_base[ref].op == IR_LOAD || ctx->ir_base[ref].op == IR_LOAD_v ||
-	     ctx->ir_base[ref].op == IR_VLOAD || ctx->ir_base[ref].op == IR_VLOAD_v)) {
-		if (ctx->use_lists[ref].count == 2
-		 && !ir_match_has_mem_deps(ctx, ref, root)) {
-			ir_ref addr_ref = ctx->ir_base[ref].op2;
-			ir_insn *addr_insn = &ctx->ir_base[addr_ref];
-
-			if (IR_IS_CONST_REF(addr_ref)) {
-				if (ir_may_fuse_addr(ctx, addr_insn)) {
-					ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
-					return;
-				}
+			break;
+		case IR_SHIFT_CONST_I64:
+		case IR_OP_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_DEF_REUSES_OP1_REG | IR_OP1_SHOULD_BE_IN_REG;
+			break;
+		case IR_SEXT_I64:
+			flags = IR_USE_SHOULD_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_HINT_TWO_REGS;
+			constraints->def_reg = IR_REG_I64_PAIR(IR_REG_RAX, IR_REG_RDX);
+			constraints->hints[1] = IR_REG_RAX;
+			constraints->hints_count = 2;
+			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_RAX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+			constraints->tmp_regs[1] = IR_SCRATCH_REG(IR_REG_RDX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+			n = 2;
+			break;
+		case IR_ZEXT_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_DEF_REUSES_OP1_REG | IR_OP1_SHOULD_BE_IN_REG;
+			break;
+		case IR_BITCAST_I64:
+			insn = &ctx->ir_base[ref];
+			if (insn->type == IR_DOUBLE) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
 			} else {
-				ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
-				ir_match_fuse_addr(ctx, addr_ref);
-				return;
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
 			}
-		}
-	}
-}
-
-static bool ir_match_try_fuse_load(ir_ctx *ctx, ir_ref ref, ir_ref root)
-{
-	ir_insn *insn = &ctx->ir_base[ref];
-
-	if (ir_in_same_block(ctx, ref)
-	 && (insn->op == IR_LOAD || insn->op == IR_LOAD_v || insn->op == IR_VLOAD || insn->op == IR_VLOAD_v)) {
-		if (ctx->use_lists[ref].count == 2
-		 && !ir_match_has_mem_deps(ctx, ref, root)) {
-			ir_ref addr_ref = ctx->ir_base[ref].op2;
-			ir_insn *addr_insn = &ctx->ir_base[addr_ref];
-
-			if (IR_IS_CONST_REF(addr_ref)) {
-				if (ir_may_fuse_addr(ctx, addr_insn)) {
-					ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
-					return 1;
+			if (!(ctx->mflags & (IR_X86_AVX|IR_X86_SSE41))) {
+				ctx->flags2 |= IR_HAS_FP_RET_SLOT;
+			}
+			break;
+		case IR_FP2INT_I64:
+			insn = &ctx->ir_base[ref];
+			if (insn->type == IR_U64) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+				if (IR_IS_CONST_REF(insn->op1)) {
+					constraints->tmp_regs[n] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
 				}
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
 			} else {
-				ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
-				ir_match_fuse_addr(ctx, addr_ref);
-				return 1;
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
 			}
-		}
-	} else if (insn->op == IR_PARAM) {
-		if (ctx->use_lists[ref].count == 1
-		 && ir_get_param_reg(ctx, ref) == IR_REG_NONE) {
-			return 1;
-		}
-	}
-	return 0;
-}
-
-static void ir_match_fuse_load_commutative_int(ir_ctx *ctx, ir_insn *insn, ir_ref root)
-{
-	if (IR_IS_CONST_REF(insn->op2)
-	 && ir_may_fuse_imm(ctx, &ctx->ir_base[insn->op2])) {
-		return;
-	} else if (ir_match_try_fuse_load(ctx, insn->op2, root)) {
-		return;
-	} else if (ir_match_try_fuse_load(ctx, insn->op1, root)) {
-		ir_swap_ops(insn);
-	}
-}
-
-static void ir_match_fuse_load_commutative_fp(ir_ctx *ctx, ir_insn *insn, ir_ref root)
-{
-	if (!IR_IS_CONST_REF(insn->op2)
-	 && !ir_match_try_fuse_load(ctx, insn->op2, root)
-	 && (IR_IS_CONST_REF(insn->op1) || ir_match_try_fuse_load(ctx, insn->op1, root))) {
-		ir_swap_ops(insn);
-	}
-}
-
-static void ir_match_fuse_load_cmp_int(ir_ctx *ctx, ir_insn *insn, ir_ref root)
-{
-	if (IR_IS_CONST_REF(insn->op2)
-	 && ir_may_fuse_imm(ctx, &ctx->ir_base[insn->op2])) {
-		ir_match_fuse_load(ctx, insn->op1, root);
-	} else if (!ir_match_try_fuse_load(ctx, insn->op2, root)
-	 && ir_match_try_fuse_load(ctx, insn->op1, root)) {
-		ir_swap_ops(insn);
-		if (insn->op != IR_EQ && insn->op != IR_NE) {
-			insn->op ^= 3;
-		}
-	}
-}
-
-static void ir_match_fuse_load_test_int(ir_ctx *ctx, ir_insn *insn, ir_ref root)
-{
-	if (IR_IS_CONST_REF(insn->op2)
-	 && ir_may_fuse_imm(ctx, &ctx->ir_base[insn->op2])) {
-		ir_match_fuse_load(ctx, insn->op1, root);
-	} else if (!ir_match_try_fuse_load(ctx, insn->op2, root)
-	 && ir_match_try_fuse_load(ctx, insn->op1, root)) {
-		ir_swap_ops(insn);
-	}
-}
-
-static void ir_match_fuse_load_cmp_fp(ir_ctx *ctx, ir_insn *insn, ir_ref root)
-{
-	if (insn->op != IR_EQ && insn->op != IR_NE) {
-		if (insn->op == IR_LT || insn->op == IR_LE) {
-			/* swap operands to avoid P flag check */
-			ir_swap_ops(insn);
-			insn->op ^= 3;
-		}
-		ir_match_fuse_load(ctx, insn->op2, root);
-	} else if (IR_IS_CONST_REF(insn->op2) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op2])) {
-		/* pass */
-	} else if (ir_match_try_fuse_load(ctx, insn->op2, root)) {
-		/* pass */
-	} else if ((IR_IS_CONST_REF(insn->op1) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op1])) || ir_match_try_fuse_load(ctx, insn->op1, root)) {
-		ir_swap_ops(insn);
-		if (insn->op != IR_EQ && insn->op != IR_NE
-		 && insn->op != IR_ORDERED && insn->op != IR_UNORDERED) {
-			insn->op ^= 3;
-		}
-	}
-}
-
-static void ir_match_fuse_load_cmp_fp_br(ir_ctx *ctx, ir_insn *insn, ir_ref root)
-{
-	if (insn->op == IR_LT || insn->op == IR_LE || insn->op == IR_UGT || insn->op == IR_UGE) {
-		/* swap operands to avoid P flag check */
-		ir_swap_ops(insn);
-		insn->op ^= 3;
-	}
-	if (IR_IS_CONST_REF(insn->op2) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op2])) {
-		/* pass */
-	} else if (ir_match_try_fuse_load(ctx, insn->op2, root)) {
-		/* pass */
-	} else if ((IR_IS_CONST_REF(insn->op1) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op1])) || ir_match_try_fuse_load(ctx, insn->op1, root)) {
-		ir_swap_ops(insn);
-		if (insn->op != IR_EQ && insn->op != IR_NE
-		 && insn->op != IR_ORDERED && insn->op != IR_UNORDERED) {
-			insn->op ^= 3;
-		}
-	}
-}
-
-#define STR_EQUAL(name, name_len, str) (name_len == strlen(str) && memcmp(name, str, strlen(str)) == 0)
-
-#define IR_IS_FP_FUNC_1(proto, _type)  (proto->params_count == 1 && \
-                                        proto->param_types[0] == _type && \
-                                        proto->ret_type == _type)
-
-static uint32_t ir_match_builtin_call(ir_ctx *ctx, const ir_insn *func)
-{
-	const ir_proto_t *proto = (const ir_proto_t *)ir_get_str(ctx, func->proto);
-
-	if ((proto->flags & IR_CALL_CONV_MASK) == IR_CC_BUILTIN) {
-		size_t name_len;
-		const char *name = ir_get_strl(ctx, func->val.name, &name_len);
-
-		if (STR_EQUAL(name, name_len, "sqrt")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
-				return IR_SSE_SQRT;
+			ctx->flags2 |= IR_HAS_FP_RET_SLOT;
+			break;
+		case IR_INT2FP_I64:
+			flags = 0;
+			ctx->flags2 |= IR_HAS_FP_RET_SLOT;
+			break;
+		case IR_MUL_I64:
+		case IR_MUL_OV_I64:
+			flags = IR_USE_SHOULD_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_HINT_TWO_REGS;
+			constraints->def_reg = IR_REG_I64_PAIR(IR_REG_RAX, IR_REG_RDX);
+			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_RAX, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			constraints->tmp_regs[1] = IR_SCRATCH_REG(IR_REG_RDX, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			constraints->tmp_regs[2] = IR_TMP_REG(3, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n = 3;
+			if ((rule & IR_RULE_MASK) == IR_MUL_OV_I64) {
+				constraints->tmp_regs[3] = IR_TMP_REG(4, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 4;
 			}
-		} else if (STR_EQUAL(name, name_len, "sqrtf")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
-				return IR_SSE_SQRT;
+			insn = &ctx->ir_base[ref];
+			if (insn->op1 != insn->op2) {
+				// TODO: hint for op1 removs inference between op1 and tmp(3), but op1 also used for op2 ???
+				flags |= IR_OP1_HINT_TWO_REGS;
+				constraints->hints[1] = IR_REG_I64_PAIR(IR_REG_RAX, IR_REG_RDX);
+				constraints->hints_count = 2;
 			}
-		} else if (STR_EQUAL(name, name_len, "rint")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
-				return IR_SSE_RINT;
+			break;
+		case IR_BIT_COUNT_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
+			constraints->tmp_regs[0] = IR_TMP_REG(2, IR_U32, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+			n = 1;
+			break;
+		case IR_BINOP_HELPER_I64:
+			flags = IR_USE_SHOULD_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_HINT_TWO_REGS;
+			constraints->def_reg = IR_REG_I64_PAIR(IR_REG_RAX, IR_REG_RDX);
+			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_SCRATCH_X86, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			break;
+		case IR_BIT_COUNT_HELPER_I64:
+			flags = IR_USE_SHOULD_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
+			constraints->def_reg = IR_REG_RAX;
+			constraints->tmp_regs[0] = IR_SCRATCH_REG(IR_REG_SCRATCH_X86, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			break;
+		case IR_IF_I64:
+			flags = IR_OP2_SHOULD_BE_IN_REG;
+			constraints->tmp_regs[0] = IR_TMP_REG(0, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			break;
+		case IR_CMP_AND_BRANCH_I64:
+		case IR_GUARD_CMP_I64:
+			constraints->tmp_regs[0] = IR_TMP_REG(0, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			insn = &ctx->ir_base[ref];
+			insn = &ctx->ir_base[insn->op2];
+			break;
+		case IR_MIN_MAX_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG;
+			constraints->tmp_regs[0] = IR_TMP_REG(3, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			break;
+		case IR_COND_I64:
+			insn = &ctx->ir_base[ref];
+			if (!IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
+				break;
 			}
-		} else if (STR_EQUAL(name, name_len, "rintf")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
-				return IR_SSE_RINT;
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
+			if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				/* 4-th temporary is stored in ctx->tmp_regs */
+				constraints->tmp_regs[0] = IR_TMP_REG(4, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-		} else if (STR_EQUAL(name, name_len, "floor")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
-				return IR_SSE_FLOOR;
+			break;
+		case IR_COND_I64_CMP_INT:
+			insn = &ctx->ir_base[ref];
+			insn = &ctx->ir_base[insn->op1];
+			if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-		} else if (STR_EQUAL(name, name_len, "floorf")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
-				return IR_SSE_FLOOR;
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
+			break;
+		case IR_COND_I64_CMP_FP:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
+			break;
+		case IR_COND_CMP_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG | IR_OP3_SHOULD_BE_IN_REG;
+			constraints->tmp_regs[0] = IR_TMP_REG(1, IR_U32, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+			n = 1;
+			break;
+		case IR_PARAM_I64:
+			flags = 0;
+			break;
+		case IR_LOAD_I64:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (!IR_IS_CONST_REF(insn->op2) && (ir_rule(ctx, insn->op2) & IR_FUSED)) {
+				uint32_t addr_rule = ir_rule(ctx, insn->op2) & IR_RULE_MASK;
+				/* For rules that use Base and Index */
+				if (addr_rule >= IR_LEA_SIB && addr_rule <= IR_LEA_SI_B_O && addr_rule != IR_LEA_SI_O) {
+					/* Base and Index registers used for LOAD inferes with at least one of the result registers */
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+				}
 			}
-		} else if (STR_EQUAL(name, name_len, "ceil")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
-				return IR_SSE_CEIL;
+			break;
+		case IR_RETURN_I64:
+			flags = IR_OP1_SHOULD_BE_IN_REG | IR_OP2_HINT_TWO_REGS;
+			constraints->hints[1] = IR_REG_NONE;
+			constraints->hints[2] = IR_REG_I64_PAIR(IR_REG_RAX, IR_REG_RDX);
+			constraints->hints_count = 3;
+			break;
+#endif
+#if IR_SIMD
+		case IR_SPLAT:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-		} else if (STR_EQUAL(name, name_len, "ceilf")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
-				return IR_SSE_CEIL;
+			if (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(insn->type))
+			 && ir_type_size[IR_VECTOR_BASE_TYPE(insn->type)] == 1
+			 && ((ctx->mflags & (IR_X86_AVX|IR_X86_AVX2)) == IR_X86_AVX || (ctx->mflags & IR_X86_SSSE3))) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
-		} else if (STR_EQUAL(name, name_len, "trunc")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
-				return IR_SSE_TRUNC;
+			break;
+		case IR_EXTRACT:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-		} else if (STR_EQUAL(name, name_len, "truncf")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
-				return IR_SSE_TRUNC;
+			if (!IR_IS_CONST_REF(insn->op2)) {
+				/* extract through stack */
+				ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+			} else if (IR_IS_TYPE_INT(insn->type)) {
+				if ((IR_IS_CONST_REF(insn->op2) && ctx->ir_base[insn->op2].val.u64 >= 16 / ir_type_size[insn->type]) ||
+					(ir_type_size[insn->type] >= 4 && !(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX))) ||
+					(sizeof(void*) == 4 && ir_type_size[insn->type] == 8)) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, IR_DOUBLE, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				}
 			}
-		} else if (STR_EQUAL(name, name_len, "nearbyint")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
-				return IR_SSE_NEARBYINT;
+			break;
+		case IR_REPLACE:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_MUST_BE_IN_REG | IR_OP3_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op3)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(3, ctx->ir_base[insn->op3].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-		} else if (STR_EQUAL(name, name_len, "nearbyintf")) {
-			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
-				return IR_SSE_NEARBYINT;
+			if (!IR_IS_CONST_REF(insn->op2) ||
+					(ir_type_size[IR_VECTOR_BASE_TYPE(insn->type)] == 1 &&
+						!(ctx->mflags & (IR_X86_SSE42|IR_X86_AVX)))) {
+				/* replace through stack */
+				ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+			} else if (IR_VECTOR_SIZE(insn->type) > 16 ||
+				(IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(insn->type)) &&
+					ir_type_size[IR_VECTOR_BASE_TYPE(insn->type)] >= 4 &&
+					!(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX)))) {
+				constraints->tmp_regs[n] = IR_TMP_REG(4, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
-		}
-	}
-
-	return 0;
-}
-
-static bool all_usages_are_fusable(ir_ctx *ctx, ir_ref ref)
-{
-	ir_insn *insn = &ctx->ir_base[ref];
-
-	if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
-		ir_use_list *use_list = &ctx->use_lists[ref];
-		ir_ref n = use_list->count;
-
-		if (n > 0) {
-			ir_ref *p = ctx->use_edges + use_list->refs;
-
-			do {
-				insn = &ctx->ir_base[*p];
-				if (insn->op != IR_IF
-				 && insn->op != IR_GUARD
-				 && insn->op != IR_GUARD_NOT
-				 && (insn->op != IR_COND || insn->op2 == ref || insn->op3 == ref)) {
-					return 0;
-				}
-				p++;
-				n--;
-			} while (n);
-			return 1;
-		}
-	}
-	return 0;
-}
-
-static uint32_t ir_match_insn(ir_ctx *ctx, ir_ref ref)
-{
-	ir_insn *op2_insn;
-	ir_insn *insn = &ctx->ir_base[ref];
-	uint32_t store_rule;
-	ir_op load_op;
-
-	switch (insn->op) {
-		case IR_EQ:
-		case IR_NE:
-		case IR_LT:
-		case IR_GE:
-		case IR_LE:
-		case IR_GT:
-		case IR_ULT:
-		case IR_UGE:
-		case IR_ULE:
-		case IR_UGT:
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
-				if (IR_IS_CONST_REF(insn->op2)
-				 && !IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op)
-				 && ctx->ir_base[insn->op2].val.i64 == 0
-				 && insn->op1 == ref - 1) { /* previous instruction */
-					ir_insn *op1_insn = &ctx->ir_base[insn->op1];
-
-					if (op1_insn->op == IR_AND && ctx->use_lists[insn->op1].count == 1) {
-						/* v = AND(_, _); CMP(v, 0) => SKIP_TEST; TEST */
-						ir_match_fuse_load_test_int(ctx, op1_insn, ref);
-						ctx->rules[insn->op1] = IR_FUSED | IR_TEST_INT;
-						return IR_TESTCC_INT;
-					} else if ((op1_insn->op == IR_OR || op1_insn->op == IR_AND || op1_insn->op == IR_XOR) ||
-							/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
-							((op1_insn->op == IR_ADD || op1_insn->op == IR_SUB) &&
-								(insn->op == IR_EQ || insn->op == IR_NE ||
-									insn->op == IR_LT || insn->op == IR_GE))) {
-						/* v = BINOP(_, _); CMP(v, 0) => BINOP; SETCC */
-						if (ir_op_flags[op1_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
-							ir_match_fuse_load_commutative_int(ctx, op1_insn, ref);
-							ctx->rules[insn->op1] = IR_BINOP_INT | IR_MAY_SWAP;
-						} else {
-							ir_match_fuse_load(ctx, op1_insn->op2, ref);
-							ctx->rules[insn->op1] = IR_BINOP_INT;
-						}
-						return IR_SETCC_INT;
-					}
+			break;
+		case IR_SHUFPD_11:
+		case IR_SHUFPS_11:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_USE_SUB_REF);
+				n = 1;
+			}
+			break;
+		case IR_SHUFPD_12:
+		case IR_MOVSD_12:
+		case IR_SHUFPS_12:
+		case IR_SHUFPS_12_0:
+		case IR_SHUFPS_12_2:
+		case IR_BLENDPS_12:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_USE_SUB_REF);
+				n = 1;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+			}
+			break;
+		case IR_SHUFPD_22:
+		case IR_SHUFPS_22:
+			flags = IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op2)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(2, IR_DOUBLE, IR_LOAD_SUB_REF, IR_USE_SUB_REF);
+				n = 1;
+			}
+			break;
+		case IR_SHUFPD_21:
+		case IR_SHUFPS_21:
+		case IR_SHUFPS_12_1:
+			flags = IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n = 1;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+			}
+			break;
+		case IR_SHUFPS_1_21:
+			flags = IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n = 1;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+			}
+			constraints->tmp_regs[n] = IR_TMP_REG(4, IR_DOUBLE, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+			n++;
+			break;
+		case IR_SHUFPS_2_12:
+			flags = IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n = 1;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+			}
+			constraints->tmp_regs[n] = IR_TMP_REG(4, IR_DOUBLE, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+			n++;
+			break;
+		case IR_SHUFFLE:
+			flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG | IR_OP3_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(1, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			if (IR_IS_CONST_REF(insn->op2) && insn->op1 != insn->op2) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op2].type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			}
+			constraints->tmp_regs[n] = IR_TMP_REG(4, IR_VECTOR_BASE_TYPE(insn->type), IR_USE_SUB_REF, IR_DEF_SUB_REF);
+			n++;
+			if (!IR_IS_CONST_REF(insn->op3)) {
+				if (IR_IS_TYPE_FP(IR_VECTOR_BASE_TYPE(insn->type))) {
+					// TODO: hardcoded index register ???
+					constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RCX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+#if IR_X86_I64
+				} else if (IR_VECTOR_BASE_TYPE(insn->type) == IR_I64 || IR_VECTOR_BASE_TYPE(insn->type) == IR_U64) {
+					constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RCX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+#endif
 				}
-				ir_match_fuse_load_cmp_int(ctx, insn, ref);
-				return IR_CMP_INT;
+			}
+			/* shuffle through stack */
+			ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+			break;
+		case IR_VECTOR_OP:
+			insn = &ctx->ir_base[ref];
+			if (ir_may_fuse_load_vector(ctx, insn->type)) {
+				flags = IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
 			} else {
-				ir_match_fuse_load_cmp_fp(ctx, insn, ref);
-				return IR_CMP_FP;
+				flags = IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			}
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, insn->type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			if (!(ctx->mflags & IR_X86_AVX2)
+			 && insn->op == IR_NEG
+			 && IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(insn->type))
+			 && IR_VECTOR_SIZE(insn->type) == 32) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, insn->type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+				if (ctx->vregs[insn->op1]) {
+					flags &= ~IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+					flags |= IR_DEF_REUSES_OP1_REG;
+					constraints->tmp_regs[n] = IR_TMP_REG(3, insn->type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				}
 			}
 			break;
-		case IR_ORDERED:
-		case IR_UNORDERED:
-			ir_match_fuse_load_cmp_fp(ctx, insn, ref);
-			return IR_CMP_FP;
-		case IR_ADD:
-		case IR_SUB:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_CONST_REF(insn->op1)) {
-						ir_insn *op1_insn = &ctx->ir_base[insn->op1];
-
-						if (insn->op == IR_ADD
-						 && IR_IS_SYM_CONST(op1_insn->op)
-						 && !IR_IS_SYM_CONST(op2_insn->op)
-						 && IR_IS_SIGNED_32BIT((intptr_t)ir_sym_val(ctx, op1_insn) + (intptr_t)op2_insn->val.i64)) {
-							return IR_LEA_SYM_O;
-						} else if (insn->op == IR_ADD
-						 && IR_IS_SYM_CONST(op2_insn->op)
-						 && !IR_IS_SYM_CONST(op1_insn->op)
-						 && IR_IS_SIGNED_32BIT((intptr_t)ir_sym_val(ctx, op2_insn) + (intptr_t)op1_insn->val.i64)) {
-							return IR_LEA_O_SYM;
-						}
-						// const
-						// TODO: add support for sym+offset ???
-					} else if (IR_IS_SYM_CONST(op2_insn->op)) {
-						if (insn->op == IR_ADD && ir_may_fuse_addr(ctx, op2_insn)) {
-							goto lea;
-						}
-						/* pass */
-					} else if (op2_insn->val.i64 == 0) {
-						// return IR_COPY_INT;
-					} else if ((ir_type_size[insn->type] >= 4 && insn->op == IR_ADD && IR_IS_SIGNED_32BIT(op2_insn->val.i64)) ||
-							(ir_type_size[insn->type] >= 4 && insn->op == IR_SUB && IR_IS_SIGNED_NEG_32BIT(op2_insn->val.i64))) {
-lea:
-						if (ctx->use_lists[insn->op1].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op1)) {
-							uint32_t rule = ctx->rules[insn->op1];
+		case IR_VECTOR_BINOP_SSE2:
+			insn = &ctx->ir_base[ref];
+			if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+				type = ctx->ir_base[insn->op1].type;
+			} else {
+				type = insn->type;
+			}

-							if (!rule) {
-								ctx->rules[insn->op1] = rule = ir_match_insn(ctx, insn->op1);
-							}
-							if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
-								/* z = MUL(Y, 2|4|8) ... ADD(z, imm32) => SKIP ... LEA [Y*2|4|8+im32] */
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
-								return IR_LEA_SI_O;
-							} else if (rule == IR_LEA_SIB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SIB)) {
-								/* z = ADD(X, MUL(Y, 2|4|8)) ... ADD(z, imm32) => SKIP ... LEA [X+Y*2|4|8+im32] */
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SIB;
-								return IR_LEA_SIB_O;
-							} else if (rule == IR_LEA_IB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_IB)) {
-								/* z = ADD(X, Y) ... ADD(z, imm32) => SKIP ... LEA [X+Y+im32] */
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_IB;
-								return IR_LEA_IB_O;
-							} else if (rule == IR_LEA_B_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_B_SI)) {
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_B_SI;
-								return IR_LEA_B_SI_O;
-							} else if (rule == IR_LEA_SI_B || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI_B)) {
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SI_B;
-								return IR_LEA_SI_B_O;
-							}
-						}
-						/* ADD(X, imm32) => LEA [X+imm32] */
-						return IR_LEA_OB;
-					} else if (op2_insn->val.i64 == 1 || op2_insn->val.i64 == -1) {
-						if (insn->op == IR_ADD) {
-							if (op2_insn->val.i64 == 1) {
-								/* ADD(_, 1) => INC */
-								return IR_INC;
-						    } else {
-								/* ADD(_, -1) => DEC */
-								return IR_DEC;
-						    }
-						} else {
-							if (op2_insn->val.i64 == 1) {
-								/* SUB(_, 1) => DEC */
-								return IR_DEC;
-						    } else {
-								/* SUB(_, -1) => INC */
-								return IR_INC;
-						    }
-						}
-					}
-				} else if ((ctx->flags & IR_OPT_CODEGEN) && insn->op == IR_ADD && ir_type_size[insn->type] >= 4) {
-					if (insn->op1 != insn->op2) {
-						if (ctx->use_lists[insn->op1].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op1)) {
-							uint32_t rule =ctx->rules[insn->op1];
-							if (!rule) {
-								ctx->rules[insn->op1] = rule = ir_match_insn(ctx, insn->op1);
-							}
-							if (rule == IR_LEA_OB) {
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_OB;
-								if (ctx->use_lists[insn->op2].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op2)) {
-									rule = ctx->rules[insn->op2];
-									if (!rule) {
-										ctx->rules[insn->op2] = rule = ir_match_insn(ctx, insn->op2);
-									}
-									if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
-										/* x = ADD(X, imm32) ... y = MUL(Y, 2|4|8) ... ADD(x, y) => SKIP ... SKIP ... LEA */
-										ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
-										return IR_LEA_OB_SI;
-									}
-								}
-								/* x = ADD(X, imm32) ... ADD(x, Y) => SKIP ... LEA */
-								return IR_LEA_OB_I;
-							} else if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
-								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
-								if (ctx->use_lists[insn->op2].count == 1) {
-									rule = ctx->rules[insn->op2];
-									if (!rule) {
-										ctx->rules[insn->op2] = rule = ir_match_insn(ctx, insn->op2);
-									}
-									if (rule == IR_LEA_OB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_OB)) {
-										/* x = ADD(X, imm32) ... y = MUL(Y, 2|4|8) ... ADD(y, x) => SKIP ... SKIP ... LEA */
-										ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_OB;
-										return IR_LEA_SI_OB;
-									}
-								}
-								/* x = MUL(X, 2|4|8) ... ADD(x, Y) => SKIP ... LEA */
-								return IR_LEA_SI_B;
-							}
-						}
-						if (ctx->use_lists[insn->op2].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op2)) {
-							uint32_t rule = ctx->rules[insn->op2];
-							if (!rule) {
-								ctx->rules[insn->op2] = rule = ir_match_insn(ctx, insn->op2);
-							}
-							if (rule == IR_LEA_OB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_OB)) {
-								ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_OB;
-								/* x = ADD(X, imm32) ... ADD(Y, x) => SKIP ... LEA */
-								return IR_LEA_I_OB;
-							} else if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
-								ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
-								/* x = MUL(X, 2|4|8) ... ADD(Y, x) => SKIP ... LEA */
-								return IR_LEA_B_SI;
-							}
-						}
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			if (IR_VECTOR_SIZE(type) == 16) {
+				if (!IR_IS_CONST_REF(insn->op2) && IR_IS_TYPE_SCALAR(ctx->ir_base[insn->op2].type)) {
+					flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				} else {
+					flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG;
+				}
+				if (IR_IS_CONST_REF(insn->op2) && insn->op2 != insn->op1) {
+					if (insn->op >= IR_LT && insn->op <= IR_UGT) {
+						/* vector comparison may require the second oprand in the register */
+						constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+						n++;
+					} else if (IR_IS_TYPE_FP(IR_VECTOR_BASE_TYPE(type)) && insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+						// TODO: workaroud against DynASM limitation: "rip-relative displacement followed by immediate" ???
+						constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+						n++;
 					}
-					/* ADD(X, Y) => LEA [X + Y] */
-					return IR_LEA_IB;
 				}
-binop_int:
-				if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
-					ir_match_fuse_load_commutative_int(ctx, insn, ref);
-					return IR_BINOP_INT | IR_MAY_SWAP;
-				} else {
-					ir_match_fuse_load(ctx, insn->op2, ref);
-					return IR_BINOP_INT;
+			} else {
+				flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				if (IR_IS_CONST_REF(insn->op2) && insn->op2 != insn->op1) {
+					constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				}
+			}
+			if (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(type))) {
+				if (!(ctx->mflags & IR_X86_SSE42)
+				 && insn->op >= IR_LT && insn->op <= IR_UGT
+				 && ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+					constraints->tmp_regs[n] = IR_TMP_REG(4, type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if (insn->op == IR_NE ||
+						(insn->op == IR_EQ &&
+						 ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8 &&
+						 !(ctx->mflags & IR_X86_SSE41))) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if ((insn->op == IR_LE || insn->op == IR_GE) &&
+						 (ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8 ||
+						  (!(ctx->mflags & IR_X86_SSE41) && ir_type_size[IR_VECTOR_BASE_TYPE(type)] != 2))) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if ((insn->op == IR_ULE || insn->op == IR_UGE) &&
+						 (ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8 ||
+						  (!(ctx->mflags & IR_X86_SSE41) && ir_type_size[IR_VECTOR_BASE_TYPE(type)] != 1))) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if (insn->op == IR_ULT || insn->op == IR_UGT) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if (insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				} else if (!(ctx->mflags & IR_X86_SSE41) && insn->op == IR_MUL && ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 4) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				}
+				if (insn->op == IR_LT || insn->op == IR_GE || insn->op == IR_ULT ||  insn->op == IR_UGE) {
+					flags &= ~IR_DEF_REUSES_OP1_REG;
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				} else if (insn->op == IR_UGT && ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 4) {
+					flags &= ~IR_DEF_REUSES_OP1_REG;
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				} else if ((insn->op == IR_LE || insn->op == IR_ULT || insn->op == IR_ULE) &&
+						ir_type_size[IR_VECTOR_BASE_TYPE(type)] != 8) {
+					flags &= ~IR_DEF_REUSES_OP1_REG;
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				} else if (insn->op >= IR_LT && insn->op <= IR_UGT) {
+					flags |= IR_OP1_MUST_BE_IN_REG;
 				}
 			} else {
-binop_fp:
-				if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
-					ir_match_fuse_load_commutative_fp(ctx, insn, ref);
-					if (ctx->mflags & IR_X86_AVX) {
-						return IR_BINOP_AVX;
-					} else {
-						return IR_BINOP_SSE2 | IR_MAY_SWAP;
-					}
-				} else {
-					ir_match_fuse_load(ctx, insn->op2, ref);
-					if (ctx->mflags & IR_X86_AVX) {
-						return IR_BINOP_AVX;
-					} else {
-						return IR_BINOP_SSE2;
-					}
+				if (insn->op == IR_GT || insn->op == IR_GE || insn->op == IR_ULT || insn->op == IR_ULE) {
+					flags &= ~IR_DEF_REUSES_OP1_REG;
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				} else if (insn->op >= IR_LT && insn->op <= IR_UGT) {
+					flags |= IR_OP1_MUST_BE_IN_REG;
 				}
 			}
 			break;
-		case IR_MUL:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_SYM_CONST(op2_insn->op)) {
-						/* pass */
-					} else if (IR_IS_CONST_REF(insn->op1)) {
-						// const
-					} else if (op2_insn->val.u64 == 0) {
-						// 0
-					} else if (op2_insn->val.u64 == 1) {
-						// return IR_COPY_INT;
-					} else if (ir_type_size[insn->type] >= 4 &&
-							(op2_insn->val.u64 == 2 || op2_insn->val.u64 == 4 || op2_insn->val.u64 == 8)) {
-						/* MUL(X, 2|4|8) => LEA [X*2|4|8] */
-						return IR_LEA_SI;
-					} else if (ir_type_size[insn->type] >= 4 &&
-							(op2_insn->val.u64 == 3 || op2_insn->val.u64 == 5 || op2_insn->val.u64 == 9)) {
-						/* MUL(X, 3|5|9) => LEA [X+X*2|4|8] */
-						return IR_LEA_SIB;
-					} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
-						/* MUL(X, PWR2) => SHL */
-						return IR_MUL_PWR2;
-					} else if (IR_IS_TYPE_SIGNED(insn->type)
-					 && ir_type_size[insn->type] != 1
-					 && IR_IS_SIGNED_32BIT(op2_insn->val.i64)
-					 && !IR_IS_CONST_REF(insn->op1)) {
-						/* MUL(_, imm32) => IMUL */
-						ir_match_fuse_load(ctx, insn->op1, ref);
-						return IR_IMUL3;
-					}
-				}
-				/* Prefer IMUL over MUL because it's more flexible and uses less registers ??? */
-//				if (IR_IS_TYPE_SIGNED(insn->type) && ir_type_size[insn->type] != 1) {
-				if (ir_type_size[insn->type] != 1) {
-					goto binop_int;
-				}
-				ir_match_fuse_load(ctx, insn->op2, ref);
-				return IR_MUL_INT;
+		case IR_VECTOR_BINOP_AVX:
+			insn = &ctx->ir_base[ref];
+			if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+				type = ctx->ir_base[insn->op1].type;
 			} else {
-				goto binop_fp;
+				type = insn->type;
 			}
-			break;
-		case IR_ADD_OV:
-		case IR_SUB_OV:
-			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
-			goto binop_int;
-		case IR_MUL_OV:
-			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
-			if (IR_IS_TYPE_SIGNED(insn->type) && ir_type_size[insn->type] != 1) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (!IR_IS_SYM_CONST(op2_insn->op)
-					 && IR_IS_SIGNED_32BIT(op2_insn->val.i64)
-					 && !IR_IS_CONST_REF(insn->op1)) {
-						/* MUL(_, imm32) => IMUL */
-						ir_match_fuse_load(ctx, insn->op1, ref);
-						return IR_IMUL3;
-					}
-				}
-				goto binop_int;
+
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-			ir_match_fuse_load(ctx, insn->op2, ref);
-			return IR_MUL_INT;
-		case IR_DIV:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_SYM_CONST(op2_insn->op)) {
-						/* pass */
-					} else if (IR_IS_CONST_REF(insn->op1)) {
-						// const
-					} else if (op2_insn->val.u64 == 1) {
-						// return IR_COPY_INT;
-					} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
-						/* DIV(X, PWR2) => SHR */
-						if (IR_IS_TYPE_UNSIGNED(insn->type)) {
-							return IR_DIV_PWR2;
-						} else {
-							return IR_SDIV_PWR2;
-						}
+			if (IR_VECTOR_SIZE(type) == 16 || IR_VECTOR_SIZE(type) == 32) {
+				if (!IR_IS_CONST_REF(insn->op2) && IR_IS_TYPE_SCALAR(ctx->ir_base[insn->op2].type)) {
+					flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				} else {
+					flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_SHOULD_BE_IN_REG;
+				}
+				if (IR_IS_CONST_REF(insn->op2) && insn->op2 != insn->op1) {
+					if (insn->op >= IR_LT && insn->op <= IR_UGT) {
+						/* vector comparison may require the second oprand in the register */
+						constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+						n++;
+					} else if (IR_IS_TYPE_FP(IR_VECTOR_BASE_TYPE(type)) && insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+						// TODO: workaroud against DynASM limitation: "rip-relative displacement followed by immediate" ???
+						constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+						n++;
+					} else if (!(ctx->mflags & IR_X86_AVX2)
+							&& IR_VECTOR_SIZE(type) == 32
+							&& IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op2].type)) {
+						// TODO: workaroud against DynASM limitation: [=>label+16] doesn't work ???
+						constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+						n++;
 					}
 				}
-				ir_match_fuse_load(ctx, insn->op2, ref);
-				return IR_DIV_INT;
 			} else {
-				goto binop_fp;
-			}
-			break;
-		case IR_MOD:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (IR_IS_SYM_CONST(op2_insn->op)) {
-					/* pass */
-				} else if (IR_IS_CONST_REF(insn->op1)) {
-					// const
-				} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
-					/* MOD(X, PWR2) => AND */
-					if (IR_IS_TYPE_UNSIGNED(insn->type)) {
-						return IR_MOD_PWR2;
-					} else {
-						return IR_SMOD_PWR2;
-					}
+				flags = IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG | IR_OP2_MUST_BE_IN_REG;
+				if (IR_IS_CONST_REF(insn->op2) && insn->op2 != insn->op1) {
+					constraints->tmp_regs[n] = IR_TMP_REG(2, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
 				}
 			}
-			ir_match_fuse_load(ctx, insn->op2, ref);
-			return IR_MOD_INT;
-		case IR_BSWAP:
-			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
-			return IR_OP_INT;
-		case IR_NOT:
-			if (insn->type == IR_BOOL) {
-				if (ctx->ir_base[insn->op1].type == IR_BOOL) {
-					return IR_BOOL_NOT;
-				} else {
-					IR_ASSERT(IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)); // TODO: IR_BOOL_NOT_FP
-					return IR_BOOL_NOT_INT;
+			if (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(type))) {
+				if ((insn->op == IR_UGT || insn->op == IR_ULT || insn->op == IR_UGE || insn->op == IR_ULE) &&
+						ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8) {
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+				}
+				if (!(ctx->mflags & IR_X86_AVX2) && IR_VECTOR_SIZE(insn->type) == 32) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, insn->type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+					if (ctx->vregs[insn->op2] || insn->op == IR_NE) {
+						constraints->tmp_regs[n] = IR_TMP_REG(4, insn->type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+						n++;
+					}
+					if (insn->op == IR_UGT || insn->op == IR_ULT || insn->op == IR_UGE || insn->op == IR_ULE) {
+						flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+					}
+				} else if ((insn->op == IR_UGT || insn->op == IR_ULT || insn->op == IR_UGE || insn->op == IR_ULE) &&
+						ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8) {
+					flags |= IR_DEF_CONFLICTS_WITH_INPUT_REGS;
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if ((insn->op == IR_UGT || insn->op == IR_ULT) &&
+						ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 4) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if (insn->op == IR_UGE || insn->op == IR_ULE) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				} else if (insn->op == IR_NE ||
+						insn->op == IR_GE ||
+						insn->op == IR_LE ||
+						insn->op == IR_UGT ||
+						insn->op == IR_ULT ||
+						(insn->op == IR_EQ &&
+						 ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8 &&
+						 !(ctx->mflags & IR_X86_SSE41))) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				} else if (insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, type, IR_LOAD_SUB_REF, IR_DEF_SUB_REF);
+					n++;
 				}
-			} else {
-				IR_ASSERT(IR_IS_TYPE_INT(insn->type));
-				return IR_OP_INT;
 			}
 			break;
-		case IR_NEG:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				return IR_OP_INT;
+		case IR_VECTOR_BINOP_EXPAND:
+			insn = &ctx->ir_base[ref];
+			if (!IR_IS_CONST_REF(insn->op2) && IR_IS_TYPE_SCALAR(ctx->ir_base[insn->op2].type)) {
+				flags = IR_OP2_SHOULD_BE_IN_REG; /* result and first operand may be in memory */
+				constraints->hints[1] = IR_REG_NONE;
+				constraints->hints[2] = IR_REG_RCX;
+				constraints->hints_count = 3;
 			} else {
-				return IR_OP_FP;
+				flags = 0; /* result and operands may be in memory */
 			}
-		case IR_ABS:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				return IR_ABS_INT; // movl %edi, %eax; negl %eax; cmovs %edi, %eax
+			if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+				type = ctx->ir_base[insn->op1].type;
 			} else {
-				return IR_OP_FP;
+				type = insn->type;
 			}
-		case IR_OR:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (IR_IS_SYM_CONST(op2_insn->op)) {
-					/* pass */
-				} else if (IR_IS_CONST_REF(insn->op1)) {
-					// const
-				} else if (op2_insn->val.i64 == 0) {
-					// return IR_COPY_INT;
-				} else if (op2_insn->val.i64 == -1) {
-					// -1
-				} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64) && !IR_IS_SIGNED_32BIT(op2_insn->val.i64)) {
-					/* OR(X, PWR2) => BTS */
-					return IR_BIT_OP;
+#if IR_X86_I64
+			if (IR_VECTOR_BASE_TYPE(type) == IR_I64 || IR_VECTOR_BASE_TYPE(type) == IR_U64) {
+				constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RAX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RDX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RCX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+			} else
+#endif
+			if (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(type))) {
+				constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RAX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				if (insn->op == IR_MUL || insn->op == IR_DIV || insn->op == IR_MOD) {
+					constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RDX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+					if (IR_IS_CONST_REF(insn->op2)) {
+						constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RCX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+						n++;
+					}
+				} else if (insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR) {
+					constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RCX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+					n++;
+				} else if (IR_IS_CONST_REF(insn->op2) && ir_type_size[IR_VECTOR_BASE_TYPE(type)] == 8) {
+					constraints->tmp_regs[n] = IR_SCRATCH_REG(IR_REG_RCX, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+					n++;
 				}
+			} else {
+				constraints->tmp_regs[n] = IR_TMP_REG(IR_VECTOR_BASE_TYPE(type), type, IR_USE_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
-			goto binop_int;
-		case IR_AND:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (IR_IS_SYM_CONST(op2_insn->op)) {
-					/* pass */
-				} else if (IR_IS_CONST_REF(insn->op1)) {
-					// const
-				} else if (op2_insn->val.i64 == 0) {
-					// 0
-				} else if (op2_insn->val.i64 == -1) {
-					// return IR_COPY_INT;
-				} else if (IR_IS_POWER_OF_TWO(~op2_insn->val.u64) && !IR_IS_SIGNED_32BIT(op2_insn->val.i64)) {
-					/* AND(X, ~PWR2) => BTR */
-					return IR_BIT_OP;
-				}
+			/* load/store through stack */
+			ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+			break;
+		case IR_VECTOR_EXT:
+		case IR_VECTOR_FP2FP:
+			insn = &ctx->ir_base[ref];
+			if (ir_may_fuse_load_vector(ctx, ctx->ir_base[insn->op1].type)) {
+				flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_SHOULD_BE_IN_REG;
+			} else {
+				flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
 			}
-			goto binop_int;
-		case IR_XOR:
-			if ((ctx->flags & IR_OPT_CODEGEN) && IR_IS_CONST_REF(insn->op2)) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (IR_IS_SYM_CONST(op2_insn->op)) {
-					/* pass */
-				} else if (IR_IS_CONST_REF(insn->op1)) {
-					// const
-				}
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, insn->type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-			goto binop_int;
-		case IR_SHL:
-			if (IR_IS_CONST_REF(insn->op2)) {
-				if (ctx->flags & IR_OPT_CODEGEN) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_SYM_CONST(op2_insn->op)) {
-						/* pass */
-					} else if (IR_IS_CONST_REF(insn->op1)) {
-						// const
-					} else if (op2_insn->val.u64 == 0) {
-						// return IR_COPY_INT;
-					} else if (ir_type_size[insn->type] >= 4) {
-						if (op2_insn->val.u64 == 1) {
-							// lea [op1*2]
-						} else if (op2_insn->val.u64 == 2) {
-							// lea [op1*4]
-						} else if (op2_insn->val.u64 == 3) {
-							// lea [op1*8]
-						}
-					}
-				}
-				return IR_SHIFT_CONST;
+			if (!(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX))
+			 && (IR_VECTOR_BASE_TYPE(insn->type) == IR_I64 || IR_VECTOR_BASE_TYPE(insn->type) == IR_U64)) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, insn->type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+			} else if (!(ctx->mflags & IR_X86_AVX2) && IR_VECTOR_SIZE(insn->type) == 32) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, insn->type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
-			return IR_SHIFT;
-		case IR_SHR:
-		case IR_SAR:
-		case IR_ROL:
-		case IR_ROR:
-			if (IR_IS_CONST_REF(insn->op2)) {
-				if (ctx->flags & IR_OPT_CODEGEN) {
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (IR_IS_SYM_CONST(op2_insn->op)) {
-						/* pass */
-					} else if (IR_IS_CONST_REF(insn->op1)) {
-						// const
-					} else if (op2_insn->val.u64 == 0) {
-						// return IR_COPY_INT;
-					}
-				}
-				return IR_SHIFT_CONST;
+			break;
+		case IR_VECTOR_TRUNC:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, insn->type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-			return IR_SHIFT;
-		case IR_MIN:
-		case IR_MAX:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				return IR_MIN_MAX_INT | IR_MAY_SWAP;
-			} else {
-				goto binop_fp;
+			if (!(ctx->mflags & IR_X86_AVX2) && IR_VECTOR_SIZE(ctx->ir_base[insn->op1].type) == 32) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, insn->type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
 			break;
-		case IR_COPY:
-			if (IR_IS_TYPE_INT(insn->type)) {
-				return IR_COPY_INT | IR_MAY_REUSE;
-			} else {
-				return IR_COPY_FP | IR_MAY_REUSE;
+		case IR_VECTOR_FP2INT:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, insn->type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
 			}
-			break;
-		case IR_CALL:
-			if (IR_IS_CONST_REF(insn->op2)) {
-				const ir_insn *func = &ctx->ir_base[insn->op2];
-
-				if (func->op == IR_FUNC && func->proto) {
-					uint32_t rule = ir_match_builtin_call(ctx, func);
-
-					if (rule) {
-						return rule;
-					}
+			if (IR_VECTOR_BASE_TYPE(insn->type) == IR_I64 || IR_VECTOR_BASE_TYPE(insn->type) == IR_U64) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+#if defined(IR_TARGET_X64)
+				constraints->tmp_regs[n] = IR_TMP_REG(3, IR_I64, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				if (IR_VECTOR_SIZE(insn->type) == 32) {
+					constraints->tmp_regs[n] = IR_TMP_REG(4, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
 				}
+#elif defined(IR_TARGET_X86)
+				ctx->flags2 |= IR_16B_FRAME_ALIGNMENT;
+#endif
+			} else if (IR_VECTOR_BASE_TYPE(insn->type) == IR_U32) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+				constraints->tmp_regs[n] = IR_TMP_REG(3, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+			} else if (IR_VECTOR_SIZE(ctx->ir_base[insn->op1].type) == 32
+					&& IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op1].type) == IR_FLOAT
+					&& ir_type_size[IR_VECTOR_BASE_TYPE(insn->type)] < 4) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
-			ctx->flags2 |= IR_HAS_CALLS | IR_16B_FRAME_ALIGNMENT;
-			IR_FALLTHROUGH;
-		case IR_TAILCALL:
-		case IR_IJMP:
-			if (!IR_IS_CONST_REF(insn->op2)) {
-				if (ctx->ir_base[insn->op2].op == IR_PROTO) {
-					if (IR_IS_CONST_REF(ctx->ir_base[insn->op2].op1)) {
-						ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_PROTO;
-					} else {
-						ir_match_fuse_load(ctx, ctx->ir_base[insn->op2].op1, ref);
-						if (ctx->rules[ctx->ir_base[insn->op2].op1] & IR_FUSED) {
-							ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_PROTO;
-						}
-				   }
-				} else {
-					ir_match_fuse_load(ctx, insn->op2, ref);
+			break;
+		case IR_VECTOR_INT2FP:
+			flags = IR_DEF_REUSES_OP1_REG | IR_USE_MUST_BE_IN_REG | IR_OP1_MUST_BE_IN_REG;
+			insn = &ctx->ir_base[ref];
+			if (IR_IS_CONST_REF(insn->op1)) {
+				constraints->tmp_regs[0] = IR_TMP_REG(1, insn->type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n = 1;
+			}
+			if (IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op1].type) == IR_I64) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+#if defined(IR_TARGET_X64)
+				constraints->tmp_regs[n] = IR_TMP_REG(3, IR_I64, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+				if (IR_VECTOR_SIZE(ctx->ir_base[insn->op1].type) == 32) {
+					constraints->tmp_regs[n] = IR_TMP_REG(4, IR_DOUBLE, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
 				}
+#elif defined(IR_TARGET_X86)
+				constraints->tmp_regs[n] = IR_TMP_REG(3, ctx->ir_base[insn->op1].type, IR_USE_SUB_REF, IR_DEF_SUB_REF);
+				n++;
+#endif
+			} else if (IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op1].type) == IR_U32) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+				if (!(ctx->mflags & IR_X86_AVX2)
+				 && IR_VECTOR_SIZE(insn->type) == 32
+				 && IR_VECTOR_BASE_TYPE(insn->type) == IR_FLOAT) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				}
+			} else if (IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op1].type) == IR_U64) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
+				if (IR_VECTOR_SIZE(ctx->ir_base[insn->op1].type) == 32) {
+					constraints->tmp_regs[n] = IR_TMP_REG(3, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+					n++;
+				}
+			} else if (!(ctx->mflags & IR_X86_AVX2)
+					&& IR_VECTOR_SIZE(insn->type) == 32
+					&& IR_VECTOR_BASE_TYPE(insn->type) == IR_FLOAT
+					&& ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op1].type)] < 4) {
+				constraints->tmp_regs[n] = IR_TMP_REG(2, ctx->ir_base[insn->op1].type, IR_LOAD_SUB_REF, IR_SAVE_SUB_REF);
+				n++;
 			}
-			return insn->op;
-		case IR_IGOTO:
-			if (ctx->ir_base[insn->op1].op == IR_MERGE || ctx->ir_base[insn->op1].op == IR_LOOP_BEGIN) {
-				ir_insn *merge = &ctx->ir_base[insn->op1];
-				ir_ref *p, n = merge->inputs_count;
+			break;
+#endif
+	}

-				for (p = merge->ops + 1; n > 0; p++, n--) {
-					ir_ref input = *p;
-					IR_ASSERT(ctx->ir_base[input].op == IR_END || ctx->ir_base[input].op == IR_LOOP_END);
-					ctx->rules[input] = IR_IGOTO_DUP;
-				}
+	IR_ASSERT(n <= 4);
+	constraints->tmps_count = n;
+
+	return flags;
+}
+
+/* instruction selection */
+static uint32_t ir_match_insn(ir_ctx *ctx, ir_ref ref);
+static bool ir_match_try_fuse_load(ir_ctx *ctx, ir_ref ref, ir_ref root);
+
+static void ir_swap_ops(ir_insn *insn)
+{
+	SWAP_REFS(insn->op1, insn->op2);
+}
+
+static bool ir_match_try_revert_lea_to_add(ir_ctx *ctx, ir_ref ref)
+{
+	ir_insn *insn = &ctx->ir_base[ref];
+
+	/* TODO: This optimization makes sense only if the other operand is killed */
+	if (insn->op1 == insn->op2) {
+		/* pass */
+	} else if (ir_match_try_fuse_load(ctx, insn->op2, ref)) {
+		ctx->rules[ref] = IR_BINOP_INT | IR_MAY_SWAP;
+		return 1;
+	} else if (ir_match_try_fuse_load(ctx, insn->op1, ref)) {
+		/* swap for better load fusion */
+		ir_swap_ops(insn);
+		ctx->rules[ref] = IR_BINOP_INT | IR_MAY_SWAP;
+		return 1;
+	}
+	return 0;
+}
+
+static void ir_match_fuse_addr(ir_ctx *ctx, ir_ref addr_ref)
+{
+	if (!IR_IS_CONST_REF(addr_ref)) {
+		uint32_t rule = ctx->rules[addr_ref];
+
+		if (!rule) {
+			ctx->rules[addr_ref] = rule = ir_match_insn(ctx, addr_ref);
+		}
+		if (rule >= IR_LEA_FIRST && rule <= IR_LEA_LAST) {
+			ir_use_list *use_list;
+			ir_ref j;
+
+			if (rule == IR_LEA_IB && ir_match_try_revert_lea_to_add(ctx, addr_ref)) {
+				return;
 			}
-			ir_match_fuse_load(ctx, insn->op2, ref);
-			return insn->op;
-		case IR_VAR:
-			return IR_STATIC_ALLOCA;
-		case IR_PARAM:
-			if (ctx->value_params && ctx->value_params[insn->op3 - 1].align) {
-				const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(ctx->flags);
-				if (cc->pass_struct_by_val) {
-					return IR_STATIC_ALLOCA;
-				}
+
+			use_list = &ctx->use_lists[addr_ref];
+			j = use_list->count;
+			if (j > 1) {
+				/* check if address is used only in LOAD and STORE */
+				ir_ref *p = &ctx->use_edges[use_list->refs];
+
+				do {
+					ir_insn *insn = &ctx->ir_base[*p];
+					if (insn->op != IR_LOAD
+					 && insn->op != IR_LOAD_v
+					 && ((insn->op != IR_STORE && insn->op != IR_STORE_v) || insn->op3 == addr_ref)) {
+						return;
+					}
+					p++;
+				} while (--j);
 			}
-			return ctx->use_lists[ref].count > 0 ? IR_PARAM : IR_SKIPPED | IR_PARAM;
-		case IR_ALLOCA:
-			/* alloca() may be used only in functions */
-			if (ctx->flags & IR_FUNCTION) {
-				if (IR_IS_CONST_REF(insn->op2) && ctx->cfg_map[ref] == 1) {
-					ir_insn *val = &ctx->ir_base[insn->op2];
+			ctx->rules[addr_ref] = IR_FUSED | IR_SIMPLE | rule;
+		}
+	}
+}

-					if (!IR_IS_SYM_CONST(val->op)) {
-						return IR_STATIC_ALLOCA;
+static bool ir_match_may_fuse_SI(ir_ctx *ctx, ir_ref ref, ir_ref use)
+{
+	ir_insn *op2_insn, *insn = &ctx->ir_base[use];
+
+	if (insn->op == IR_ADD) {
+		if (insn->op1 == ref) {
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					if (ir_may_fuse_addr(ctx, op2_insn)) {
+						return 1; // LEA_SI_O
 					}
+				} else if (IR_IS_SIGNED_32BIT(op2_insn->val.i64)) {
+					return 1; // LEA_SI_O
 				}
-				ctx->flags |= IR_USE_FRAME_POINTER;
-				ctx->flags2 |= IR_HAS_ALLOCA | IR_16B_FRAME_ALIGNMENT;
+			} else if (insn->op2 != ref) {
+				return 1; // LEA_SI_B or LEA_SI_OB
 			}
-			return IR_ALLOCA;
-		case IR_VSTORE:
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
-				store_rule = IR_VSTORE_INT;
-				load_op = IR_VLOAD;
-store_int:
-				if ((ctx->flags & IR_OPT_CODEGEN)
-				 && ir_in_same_block(ctx, insn->op3)
-				 && (ctx->use_lists[insn->op3].count == 1 ||
-				     (ctx->use_lists[insn->op3].count == 2
-				   && (ctx->ir_base[insn->op3].op == IR_ADD_OV ||
-				       ctx->ir_base[insn->op3].op == IR_SUB_OV)))) {
-					ir_insn *op_insn = &ctx->ir_base[insn->op3];
-					uint32_t rule = ctx->rules[insn->op3];
+		} else if (insn->op2 == ref && insn->op1 != insn->op2) {
+			return 1; // LEA_B_SI or LEA_OB_SI
+		}
+	}
+	return 0;
+}

-					if (!rule) {
-						ctx->rules[insn->op3] = rule = ir_match_insn(ctx, insn->op3);
-					}
-					if (((rule & IR_RULE_MASK) == IR_BINOP_INT && op_insn->op != IR_MUL) || rule == IR_LEA_OB || rule == IR_LEA_IB) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = BINOP(l, _) ... STORE(l, a, v) => SKIP ... SKIP_MEM_BINOP ... MEM_BINOP */
-							ctx->rules[insn->op3] = IR_FUSED | IR_BINOP_INT;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							if (!IR_IS_CONST_REF(op_insn->op2)
-							 && ctx->rules[op_insn->op2] == (IR_FUSED|IR_SIMPLE|IR_LOAD)) {
-								ctx->rules[op_insn->op2] = IR_LOAD_INT;
-							}
-							return IR_MEM_BINOP_INT;
-						} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
-						 && insn->op1 == op_insn->op2
-						 && ctx->ir_base[op_insn->op2].op == load_op
-						 && ctx->ir_base[op_insn->op2].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op2].count == 2) {
-							/* l = LOAD(_, a) ... v = BINOP(_, l) ... STORE(l, a, v) => SKIP ... SKIP_MEM_BINOP ... MEM_BINOP */
-							ir_swap_ops(op_insn);
-							ctx->rules[insn->op3] = IR_FUSED | IR_BINOP_INT;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_BINOP_INT;
-						}
-					} else if (rule == IR_INC) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = INC(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_INC */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_INC;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_INC;
-						}
-					} else if (rule == IR_DEC) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2){
-							/* l = LOAD(_, a) ... v = DEC(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_DEC */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_DEC;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_DEC;
-						}
-					} else if (rule == IR_MUL_PWR2) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = MUL_PWR2(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_MUL_PWR2 */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_MUL_PWR2;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_MUL_PWR2;
-						}
-					} else if (rule == IR_DIV_PWR2) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = DIV_PWR2(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_DIV_PWR2 */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_DIV_PWR2;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_DIV_PWR2;
-						}
-					} else if (rule == IR_MOD_PWR2) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = MOD_PWR2(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_MOD_PWR2 */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_MOD_PWR2;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_MOD_PWR2;
-						}
-					} else if (rule == IR_SHIFT) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = SHIFT(l, _) ... STORE(l, a, v) => SKIP ... SKIP_SHIFT ... MEM_SHIFT */
-							ctx->rules[insn->op3] = IR_FUSED | IR_SHIFT;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_SHIFT;
-						}
-					} else if (rule == IR_SHIFT_CONST) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = SHIFT(l, CONST) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_SHIFT_CONST */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_SHIFT_CONST;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_SHIFT_CONST;
-						}
-					} else if (rule == IR_OP_INT && op_insn->op != IR_BSWAP) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == load_op
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = LOAD(_, a) ... v = OP(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_OP */
-							ctx->rules[insn->op3] = IR_SKIPPED | IR_OP_INT;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
-							return IR_MEM_OP_INT;
-						}
-					} else if (rule == IR_CMP_INT && load_op == IR_LOAD) {
-						/* c = CMP(_, _) ... STORE(c) => SKIP_CMP ... CMP_AND_STORE_INT */
-						ctx->rules[insn->op3] = IR_FUSED | IR_CMP_INT;
-						return IR_CMP_AND_STORE_INT;
+static bool ir_match_fuse_addr_all_useges(ir_ctx *ctx, ir_ref ref)
+{
+	uint32_t rule = ctx->rules[ref];
+	ir_use_list *use_list;
+	ir_ref n, *p, use;
+
+	if (rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
+		return 1;
+	} else if (!rule) {
+		ir_insn *insn = &ctx->ir_base[ref];
+
+		IR_ASSERT(IR_IS_TYPE_INT(insn->type) && ir_type_size[insn->type] >= 4);
+		if (insn->op == IR_MUL
+		 && IR_IS_CONST_REF(insn->op2)) {
+			insn = &ctx->ir_base[insn->op2];
+			if (!IR_IS_SYM_CONST(insn->op)
+			 &&	(insn->val.u64 == 2 || insn->val.u64 == 4 || insn->val.u64 == 8)) {
+				ctx->rules[ref] = IR_LEA_SI;
+
+				use_list = &ctx->use_lists[ref];
+				n = use_list->count;
+				IR_ASSERT(n > 1);
+				p = &ctx->use_edges[use_list->refs];
+				for (; n > 0; p++, n--) {
+					use = *p;
+					if (!ir_match_may_fuse_SI(ctx, ref, use)) {
+						return 0;
 					}
 				}
-				return store_rule;
-			} else {
-				return IR_VSTORE_FP;
-			}
-			break;
-		case IR_VSTORE_v:
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
-				return IR_VSTORE_INT;
-			} else {
-				return IR_VSTORE_FP;
-			}
-			break;
-		case IR_LOAD:
-		case IR_LOAD_v:
-			ir_match_fuse_addr(ctx, insn->op2);
-			if (IR_IS_TYPE_INT(insn->type)) {
-				return IR_LOAD_INT;
-			} else {
-				return IR_LOAD_FP;
+
+				return 1;
 			}
-			break;
-		case IR_STORE:
-			ir_match_fuse_addr(ctx, insn->op2);
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
-				store_rule = IR_STORE_INT;
-				load_op = IR_LOAD;
-				goto store_int;
-			} else {
-				return IR_STORE_FP;
+		}
+	}
+
+	return 0;
+}
+
+/* A naive check if there is a STORE or CALL between this LOAD and the fusion root */
+static bool ir_match_has_mem_deps(ir_ctx *ctx, ir_ref ref, ir_ref root)
+{
+	if (ref + 1 != root) {
+		ir_ref pos = ctx->prev_ref[root];
+
+		do {
+			ir_insn *insn = &ctx->ir_base[pos];
+
+			if (insn->op == IR_STORE || insn->op == IR_STORE_v || insn->op == IR_VSTORE || insn->op == IR_VSTORE_v) {
+				// TODO: check if LOAD and STORE addresses may alias
+				return 1;
+			} else if (insn->op == IR_CALL) {
+				return 1;
 			}
-			break;
-		case IR_STORE_v:
-			ir_match_fuse_addr(ctx, insn->op2);
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
-				return IR_STORE_INT;
+			pos = ctx->prev_ref[pos];
+		} while (ref != pos);
+	}
+	return 0;
+}
+
+/* A naive check if anything that emits code, and so clobbers the flags, is
+ * scheduled between the flags setting instruction and the fusion root */
+static bool ir_match_has_flags_deps(ir_ctx *ctx, ir_ref ref, ir_ref root)
+{
+	ir_ref pos = ctx->prev_ref[root];
+
+	while (pos > ref) {
+		if (ctx->ir_base[pos].op != IR_SNAPSHOT) {
+			return 1;
+		}
+		pos = ctx->prev_ref[pos];
+	}
+	return pos != ref;
+}
+
+static void ir_match_fuse_load(ir_ctx *ctx, ir_ref ref, ir_ref root)
+{
+	if (ir_in_same_block(ctx, ref) &&
+	    (ctx->ir_base[ref].op == IR_LOAD || ctx->ir_base[ref].op == IR_LOAD_v ||
+	     ctx->ir_base[ref].op == IR_VLOAD || ctx->ir_base[ref].op == IR_VLOAD_v)) {
+		if (ctx->use_lists[ref].count == 2
+		 && !ir_match_has_mem_deps(ctx, ref, root)) {
+			ir_ref addr_ref = ctx->ir_base[ref].op2;
+			ir_insn *addr_insn = &ctx->ir_base[addr_ref];
+
+			if (IR_IS_CONST_REF(addr_ref)) {
+				if (ir_may_fuse_addr(ctx, addr_insn)) {
+					ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
+					return;
+				}
+			} else if (addr_insn->op == IR_TLS_ADDR) {
+				// TODO: try to fuse static TLS addr ???
+				return;
 			} else {
-				return IR_STORE_FP;
-			}
-			break;
-		case IR_RLOAD:
-			if (IR_REGSET_IN(IR_REGSET_UNION((ir_regset)ctx->fixed_regset, IR_REGSET_FIXED), insn->op2)) {
-				return IR_SKIPPED | IR_RLOAD;
+				ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
+				ir_match_fuse_addr(ctx, addr_ref);
+				return;
 			}
-			return IR_RLOAD;
-		case IR_RSTORE:
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
-				if ((ctx->flags & IR_OPT_CODEGEN)
-				 && ir_in_same_block(ctx, insn->op2)
-				 && ctx->use_lists[insn->op2].count == 1
-				 && IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
-					ir_insn *op_insn = &ctx->ir_base[insn->op2];
+		}
+	}
+}

-					if (op_insn->op == IR_ADD ||
-				        op_insn->op == IR_SUB ||
-//				        op_insn->op == IR_MUL ||
-				        op_insn->op == IR_OR  ||
-				        op_insn->op == IR_AND ||
-				        op_insn->op == IR_XOR) {
-						if (insn->op1 == op_insn->op1
-						 && ctx->ir_base[op_insn->op1].op == IR_RLOAD
-						 && ctx->ir_base[op_insn->op1].op2 == insn->op3
-						 && ctx->use_lists[op_insn->op1].count == 2) {
-							/* l = RLOAD(r) ... v = BINOP(l, _) ... RSTORE(l, r, v) => SKIP ... SKIP_REG_BINOP ... REG_BINOP */
-							ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | IR_RLOAD;
-							return IR_REG_BINOP_INT;
-						} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
-						 && insn->op1 == op_insn->op2
-						 && ctx->ir_base[op_insn->op2].op == IR_RLOAD
-						 && ctx->ir_base[op_insn->op2].op2 == insn->op3
-						 && ctx->use_lists[op_insn->op2].count == 2) {
-							/* l = RLOAD(r) ... v = BINOP(x, l) ... RSTORE(l, r, v) => SKIP ... SKIP_REG_BINOP ... REG_BINOP */
-							ir_swap_ops(op_insn);
-							ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
-							ctx->rules[op_insn->op1] = IR_SKIPPED | IR_RLOAD;
-							return IR_REG_BINOP_INT;
-						}
-					}
+static bool ir_match_try_fuse_load(ir_ctx *ctx, ir_ref ref, ir_ref root)
+{
+	ir_insn *insn = &ctx->ir_base[ref];
+
+	if (ir_in_same_block(ctx, ref)
+	 && (insn->op == IR_LOAD || insn->op == IR_LOAD_v || insn->op == IR_VLOAD || insn->op == IR_VLOAD_v)) {
+		if (ctx->use_lists[ref].count == 2
+		 && !ir_match_has_mem_deps(ctx, ref, root)) {
+			ir_ref addr_ref = ctx->ir_base[ref].op2;
+			ir_insn *addr_insn = &ctx->ir_base[addr_ref];
+
+			if (IR_IS_CONST_REF(addr_ref)) {
+				if (ir_may_fuse_addr(ctx, addr_insn)) {
+					ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
+					return 1;
 				}
-			}
-			ir_match_fuse_load(ctx, insn->op2, ref);
-			return IR_RSTORE;
-		case IR_START:
-		case IR_BEGIN:
-		case IR_IF_TRUE:
-		case IR_IF_FALSE:
-		case IR_CASE_VAL:
-		case IR_CASE_RANGE:
-		case IR_CASE_DEFAULT:
-		case IR_MERGE:
-		case IR_LOOP_BEGIN:
-		case IR_UNREACHABLE:
-			return IR_SKIPPED | insn->op;
-		case IR_RETURN:
-			if (!insn->op2) {
-				return IR_RETURN_VOID;
-			} else if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
-				return IR_RETURN_INT;
+			} else if (addr_insn->op == IR_TLS_ADDR) {
+				// TODO: try to fuse static TLS addr ???
+				return 0;
 			} else {
-				return IR_RETURN_FP;
+				ctx->rules[ref] = IR_FUSED | IR_SIMPLE | IR_LOAD;
+				ir_match_fuse_addr(ctx, addr_ref);
+				return 1;
 			}
-		case IR_IF:
-			if (!IR_IS_CONST_REF(insn->op2) && (ctx->use_lists[insn->op2].count == 1 || all_usages_are_fusable(ctx, insn->op2))) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (op2_insn->op >= IR_EQ && op2_insn->op <= IR_UNORDERED) {
-					if (IR_IS_TYPE_INT(ctx->ir_base[op2_insn->op1].type)) {
-						if (IR_IS_CONST_REF(op2_insn->op2)
-						 && !IR_IS_SYM_CONST(ctx->ir_base[op2_insn->op2].op)
-						 && ctx->ir_base[op2_insn->op2].val.i64 == 0
-						 && op2_insn->op1 == insn->op2 - 1) { /* previous instruction */
-							ir_insn *op1_insn = &ctx->ir_base[op2_insn->op1];
+		}
+	} else if (insn->op == IR_PARAM) {
+		if (ctx->use_lists[ref].count == 1
+		 && ir_get_param_reg(ctx, ref) == IR_REG_NONE) {
+			return 1;
+		}
+	}
+	return 0;
+}

-							if (op1_insn->op == IR_AND && ctx->use_lists[op2_insn->op1].count == 1) {
-								/* v = AND(_, _); c = CMP(v, 0) ... IF(c) => SKIP_TEST; SKIP ... TEST_AND_BRANCH */
-								if (ctx->use_lists[insn->op2].count == 1) {
-									ir_match_fuse_load_test_int(ctx, op1_insn, ref);
-								}
-								ctx->rules[op2_insn->op1] = IR_FUSED | IR_TEST_INT;
-								ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_NOP;
-								return IR_TEST_AND_BRANCH_INT;
-							} else if (insn->op2 == ref - 1 && /* previous instruction */
-									((op1_insn->op == IR_OR || op1_insn->op == IR_AND || op1_insn->op == IR_XOR) ||
-										/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
-										((op1_insn->op == IR_ADD || op1_insn->op == IR_SUB) &&
-											(op2_insn->op == IR_EQ || op2_insn->op == IR_NE ||
-												op2_insn->op == IR_LT || op2_insn->op == IR_GE)))) {
-								/* v = BINOP(_, _); c = CMP(v, 0) ... IF(c) => BINOP; SKIP_CMP ... JCC */
-								if (ir_op_flags[op1_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
-									if (ctx->use_lists[insn->op2].count == 1) {
-										ir_match_fuse_load_commutative_int(ctx, op1_insn, ref);
-									}
-									ctx->rules[op2_insn->op1] = IR_BINOP_INT | IR_MAY_SWAP;
-								} else {
-									if (ctx->use_lists[insn->op2].count == 1) {
-										ir_match_fuse_load(ctx, op1_insn->op2, ref);
-									}
-									ctx->rules[op2_insn->op1] = IR_BINOP_INT;
-								}
-								ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
-								return IR_JCC_INT;
+static void ir_match_fuse_load_commutative_int(ir_ctx *ctx, ir_insn *insn, ir_ref root)
+{
+	if (IR_IS_CONST_REF(insn->op2)
+	 && ir_may_fuse_imm(ctx, &ctx->ir_base[insn->op2])) {
+		return;
+	} else if (ir_match_try_fuse_load(ctx, insn->op2, root)) {
+		return;
+	} else if (ir_match_try_fuse_load(ctx, insn->op1, root)) {
+		ir_swap_ops(insn);
+	}
+}
+
+static void ir_match_fuse_load_commutative_fp(ir_ctx *ctx, ir_insn *insn, ir_ref root)
+{
+	if (!IR_IS_CONST_REF(insn->op2)
+	 && !ir_match_try_fuse_load(ctx, insn->op2, root)
+	 && (IR_IS_CONST_REF(insn->op1) || ir_match_try_fuse_load(ctx, insn->op1, root))) {
+		ir_swap_ops(insn);
+	}
+}
+
+static void ir_match_fuse_load_cmp_int(ir_ctx *ctx, ir_insn *insn, ir_ref root)
+{
+	if (IR_IS_CONST_REF(insn->op2)
+	 && ir_may_fuse_imm(ctx, &ctx->ir_base[insn->op2])) {
+		ir_match_fuse_load(ctx, insn->op1, root);
+	} else if (!ir_match_try_fuse_load(ctx, insn->op2, root)
+	 && ir_match_try_fuse_load(ctx, insn->op1, root)) {
+		ir_swap_ops(insn);
+		if (insn->op != IR_EQ && insn->op != IR_NE) {
+			insn->op ^= 3;
+		}
+	}
+}
+
+static void ir_match_fuse_load_test_int(ir_ctx *ctx, ir_insn *insn, ir_ref root)
+{
+	if (IR_IS_CONST_REF(insn->op2)
+	 && ir_may_fuse_imm(ctx, &ctx->ir_base[insn->op2])) {
+		ir_match_fuse_load(ctx, insn->op1, root);
+	} else if (!ir_match_try_fuse_load(ctx, insn->op2, root)
+	 && ir_match_try_fuse_load(ctx, insn->op1, root)) {
+		ir_swap_ops(insn);
+	}
+}
+
+static void ir_match_fuse_load_cmp_fp(ir_ctx *ctx, ir_insn *insn, ir_ref root)
+{
+	if (insn->op != IR_EQ && insn->op != IR_NE) {
+		if (insn->op == IR_LT || insn->op == IR_LE) {
+			/* swap operands to avoid P flag check */
+			ir_swap_ops(insn);
+			insn->op ^= 3;
+		}
+		ir_match_fuse_load(ctx, insn->op2, root);
+	} else if (IR_IS_CONST_REF(insn->op2) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op2])) {
+		/* pass */
+	} else if (ir_match_try_fuse_load(ctx, insn->op2, root)) {
+		/* pass */
+	} else if ((IR_IS_CONST_REF(insn->op1) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op1])) || ir_match_try_fuse_load(ctx, insn->op1, root)) {
+		ir_swap_ops(insn);
+		if (insn->op != IR_EQ && insn->op != IR_NE
+		 && insn->op != IR_ORDERED && insn->op != IR_UNORDERED) {
+			insn->op ^= 3;
+		}
+	}
+}
+
+static void ir_match_fuse_load_cmp_fp_br(ir_ctx *ctx, ir_insn *insn, ir_ref root)
+{
+	if (insn->op == IR_LT || insn->op == IR_LE || insn->op == IR_UGT || insn->op == IR_UGE) {
+		/* swap operands to avoid P flag check */
+		ir_swap_ops(insn);
+		insn->op ^= 3;
+	}
+	if (IR_IS_CONST_REF(insn->op2) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op2])) {
+		/* pass */
+	} else if (ir_match_try_fuse_load(ctx, insn->op2, root)) {
+		/* pass */
+	} else if ((IR_IS_CONST_REF(insn->op1) && !IR_IS_FP_ZERO(ctx->ir_base[insn->op1])) || ir_match_try_fuse_load(ctx, insn->op1, root)) {
+		ir_swap_ops(insn);
+		if (insn->op != IR_EQ && insn->op != IR_NE
+		 && insn->op != IR_ORDERED && insn->op != IR_UNORDERED) {
+			insn->op ^= 3;
+		}
+	}
+}
+
+#define STR_EQUAL(name, name_len, str) (name_len == strlen(str) && memcmp(name, str, strlen(str)) == 0)
+
+#define IR_IS_FP_FUNC_1(proto, _type)  (proto->params_count == 1 && \
+                                        proto->param_types[0] == _type && \
+                                        proto->ret_type == _type)
+
+static uint32_t ir_match_builtin_call(ir_ctx *ctx, const ir_insn *func)
+{
+	const ir_proto_t *proto = (const ir_proto_t *)ir_get_str(ctx, func->proto);
+
+	if ((proto->flags & IR_CALL_CONV_MASK) == IR_CC_BUILTIN) {
+		size_t name_len;
+		const char *name = ir_get_strl(ctx, func->val.name, &name_len);
+
+		if (STR_EQUAL(name, name_len, "sqrt")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
+				return IR_SSE_SQRT;
+			}
+		} else if (STR_EQUAL(name, name_len, "sqrtf")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
+				return IR_SSE_SQRT;
+			}
+		} else if (!(ctx->mflags & (IR_X86_AVX|IR_X86_SSE41))) {
+			/* skip */
+		} else if (STR_EQUAL(name, name_len, "rint")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
+				return IR_SSE_RINT;
+			}
+		} else if (STR_EQUAL(name, name_len, "rintf")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
+				return IR_SSE_RINT;
+			}
+		} else if (STR_EQUAL(name, name_len, "floor")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
+				return IR_SSE_FLOOR;
+			}
+		} else if (STR_EQUAL(name, name_len, "floorf")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
+				return IR_SSE_FLOOR;
+			}
+		} else if (STR_EQUAL(name, name_len, "ceil")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
+				return IR_SSE_CEIL;
+			}
+		} else if (STR_EQUAL(name, name_len, "ceilf")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
+				return IR_SSE_CEIL;
+			}
+		} else if (STR_EQUAL(name, name_len, "trunc")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
+				return IR_SSE_TRUNC;
+			}
+		} else if (STR_EQUAL(name, name_len, "truncf")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
+				return IR_SSE_TRUNC;
+			}
+		} else if (STR_EQUAL(name, name_len, "nearbyint")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_DOUBLE)) {
+				return IR_SSE_NEARBYINT;
+			}
+		} else if (STR_EQUAL(name, name_len, "nearbyintf")) {
+			if (IR_IS_FP_FUNC_1(proto, IR_FLOAT)) {
+				return IR_SSE_NEARBYINT;
+			}
+		}
+	}
+
+	return 0;
+}
+
+static bool all_usages_are_fusable(ir_ctx *ctx, ir_ref ref)
+{
+	ir_insn *insn = &ctx->ir_base[ref];
+
+	if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+		ir_use_list *use_list = &ctx->use_lists[ref];
+		ir_ref n = use_list->count;
+
+		if (n > 0) {
+			ir_ref *p = ctx->use_edges + use_list->refs;
+
+			do {
+				insn = &ctx->ir_base[*p];
+				if (insn->op != IR_IF
+				 && insn->op != IR_GUARD
+				 && insn->op != IR_GUARD_NOT
+				 && (insn->op != IR_COND || insn->op2 == ref || insn->op3 == ref)) {
+					return 0;
+				}
+				p++;
+				n--;
+			} while (n);
+			return 1;
+		}
+	}
+	return 0;
+}
+
+bool ir_may_fuse_tls_addr(ir_ctx *ctx, ir_ref ref)
+{
+	ir_use_list *use_list = &ctx->use_lists[ref];
+	ir_ref n, *p;
+
+	if (use_list->count == 2 || (ctx->rules[ref] & IR_FUSED)) {
+		return 1;
+	}
+	n = use_list->count;
+	for (p = ctx->use_edges + use_list->refs; n > 0; p++, n--) {
+		ir_ref use = *p;
+		ir_op op = ctx->ir_base[use].op;
+		if (op == IR_LOAD || op == IR_LOAD_v) {
+			/* pass */
+		} else if (op == IR_STORE || op == IR_STORE_v) {
+			if (ctx->ir_base[use].op3 == ref) {
+				return 0;
+			}
+		} else if (ctx->ir_base[use].op1 == ref && (ir_op_flags[op] & (IR_OP_FLAG_CONTROL|IR_OP_FLAG_MEM))) {
+			/* ignore control link */
+		} else {
+			return 0;
+		}
+	}
+	return 1;
+}
+
+#if IR_SIMD
+# define IR_SHUFFLE_MASK(i) (p[(i)*s])
+
+static uint32_t ir_match_shuffle(ir_ctx *ctx, const ir_insn *insn)
+{
+	if (IR_IS_CONST_REF(insn->op3)) {
+		ir_insn *op3_insn = &ctx->ir_base[insn->op3];
+		int8_t *p;
+		uint32_t s, n, n1, n2;
+		ir_type element_type;
+
+		IR_ASSERT(IR_IS_TYPE_VECTOR(insn->type) && IR_IS_TYPE_VECTOR(op3_insn->type));
+		p = ir_long_const_ptr(ctx, insn->op3);
+		s = ir_type_size[IR_VECTOR_BASE_TYPE(op3_insn->type)];
+		n = IR_VECTOR_LENGTH(op3_insn->type);
+		n1 = IR_VECTOR_LENGTH(ctx->ir_base[insn->op1].type);
+		n2 = IR_VECTOR_LENGTH(ctx->ir_base[insn->op2].type);
+		element_type = IR_VECTOR_BASE_TYPE(insn->type);
+
+		if (element_type == IR_I64 || element_type == IR_U64) {
+			if (n == 2 && n1 == 2 && n2 == 2) {
+				// TODO:  try PUNPCKHQDQ, PUNPCKLQDQ
+			}
+			element_type = IR_DOUBLE;
+		} else if (element_type == IR_I32 || element_type == IR_U32) {
+			if (n == 4 && n1 == 4 && n2 == 4) {
+				// TODO: try PUNPCKHDQ, PUNPCKLDQ, PSHUFD ???
+			}
+			element_type = IR_FLOAT;
+		}
+
+		if (element_type == IR_DOUBLE) {
+			if (n == 2 && n1 == 2 && n2 == 2) {
+				uint32_t mask = 0;
+
+				mask |= (IR_SHUFFLE_MASK(0) & 3);
+				mask |= (IR_SHUFFLE_MASK(1) & 3) << 2;
+				switch (mask) {
+					case 0x0: /* 0000 00 <src1[0], src1[0]> SHUFPD xmm0, xmm0, 0 (UNPCKLPD) */
+						return IR_SHUFPD_11;
+					case 0x1: /* 0001 10 <src1[1], src1[0]> SHUFPD xmm0, xmm0, 1 */
+						return IR_SHUFPD_11;
+					case 0x2: /* 0010 20 <src2[0], src1[0]> SHUFPD xmm1, xmm0, 0 (UNPCKLPD) */
+						return IR_SHUFPD_21;
+					case 0x3: /* 0011 30 <src2[1], src1[0]> SHUFPD xmm0, xmm0, 1 */
+						return IR_SHUFPD_21;
+					case 0x4: /* 0100 01 <src1[0], src1[1]> SHUFPD xmm0, xmm0, 2 (MOV) */
+						return IR_SHUFPD_11;
+					case 0x5: /* 0101 11 <src1[1], src1[1]> SHUFPD xmm0, xmm0, 3 (UNPCKHPD) */
+						return IR_SHUFPD_11;
+					case 0x6: /* 0110 21 <src2[0], src1[1]> MOVSD xmm0, xmm1 */
+						return IR_MOVSD_12;
+					case 0x7: /* 0111 31 <src2[1], src1[1]> SHUFPD xmm1, xmm0, 3 (UNPCKHPD) */
+						return IR_SHUFPD_21;
+					case 0x8: /* 1000 02 <src1[0], src2[0]> SHUFPD xmm0, xmm1, 0 (UNPCKLPD) */
+						return IR_SHUFPD_12;
+					case 0x9: /* 1001 12 <src1[1], src2[0]> SHUFPD xmm0, xmm1, 1 */
+						return IR_SHUFPD_12;
+					case 0xa: /* 1010 22 <src2[0], src2[0]> SHUFPD xmm1, xmm1, 0 (UNPCKLPD)*/
+						return IR_SHUFPD_22;
+					case 0xb: /* 1011 32 <src2[1], src2[0]> SHUFPD xmm1, xmm1, 1 */
+						return IR_SHUFPD_22;
+					case 0xc: /* 1100 30 <src1[0], src2[1]> SHUFPD xmm0, xmm1, 2 */
+						return IR_SHUFPD_12;
+					case 0xd: /* 1101 13 <src1[1], src2[1]> SHUFPD xmm0, xmm1, 3 (UNPCKHPD) */
+						return IR_SHUFPD_12;
+					case 0xe: /* 1110 23 <src2[0], src2[1]> SHUFPD xmm1, xmm1, 2 (MOV)*/
+						return IR_SHUFPD_22;
+					case 0xf: /* 1111 33 <src2[1], src2[1]> SHUFPD xmm1, xmm1, 3 (UNPCKHPD) */
+						return IR_SHUFPD_22;
+					default:
+						break;
+				}
+			}
+		} else if (element_type == IR_FLOAT) {
+			if (n == 4 && n1 == 4 && n2 == 4) {
+				if (insn->op1 == insn->op2) {
+					return IR_SHUFPS_11;
+				} else {
+					uint32_t i, v2_mask = 0;
+
+					if ((ctx->mflags & IR_X86_SSE41)
+					 && (IR_SHUFFLE_MASK(0) & 3) == 0
+					 && (IR_SHUFFLE_MASK(1) & 3) == 1
+					 && (IR_SHUFFLE_MASK(2) & 3) == 2
+					 && (IR_SHUFFLE_MASK(3) & 3) == 3) {
+						i = IR_SHUFFLE_MASK(0) & 4;
+						if ((IR_SHUFFLE_MASK(1) & 4) != i || (IR_SHUFFLE_MASK(3) & 4) != i || (IR_SHUFFLE_MASK(3) & 4) != i) {
+							return IR_BLENDPS_12;
+						}
+					}
+
+					for (i = 0; i < 4; i++) {
+						if (IR_SHUFFLE_MASK(i) >= 4) {
+							v2_mask |= (1 << i);
+						}
+					}
+					switch (v2_mask) {
+						case 0x0: /* 0000 */
+							return IR_SHUFPS_11;
+						case 0x1: /* 0001 */
+						case 0x2: /* 0010 */
+							return IR_SHUFPS_12_1;
+						case 0x3: /* 0011 */
+							return IR_SHUFPS_21;
+						case 0x4: /* 0100 */
+							if (IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(1) ||
+								IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(3) ||
+								IR_SHUFFLE_MASK(1) == IR_SHUFFLE_MASK(3)) {
+								return IR_SHUFPS_12_0;
+							} else {
+								return IR_SHUFPS_1_21;
+							}
+						case 0x5: /* 0101 */
+						case 0x6: /* 0110 */
+							return IR_SHUFPS_12_0;
+						case 0x7: /* 0111 */
+							if (IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(1) ||
+								IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(2) ||
+								IR_SHUFFLE_MASK(1) == IR_SHUFFLE_MASK(2)) {
+								return IR_SHUFPS_12_0;
+							} else {
+								return IR_SHUFPS_2_12;
+							}
+						case 0x8: /* 1000 */
+							if (IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(1) ||
+								IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(2) ||
+								IR_SHUFFLE_MASK(1) == IR_SHUFFLE_MASK(2)) {
+								return IR_SHUFPS_12_0;
+							} else {
+								return IR_SHUFPS_1_21;
+							}
+						case 0x9: /* 1001 */
+						case 0xa: /* 1010 */
+							return IR_SHUFPS_12_0;
+						case 0xb: /* 1011 */
+							if (IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(1) ||
+								IR_SHUFFLE_MASK(0) == IR_SHUFFLE_MASK(3) ||
+								IR_SHUFFLE_MASK(1) == IR_SHUFFLE_MASK(3)) {
+								return IR_SHUFPS_12_0;
+							} else {
+								return IR_SHUFPS_2_12;
+							}
+							break;
+						case 0xc: /* 1100 */
+							return IR_SHUFPS_12;
+						case 0xd: /* 1101 */
+						case 0xe: /* 1110 */
+							return IR_SHUFPS_12_2;
+						case 0xf: /* 1111 */
+							return IR_SHUFPS_22;
+						default:
+							break;
+					}
+				}
+			}
+		}
+	}
+
+	return IR_SHUFFLE;
+}
+#endif
+
+static uint32_t ir_match_insn(ir_ctx *ctx, ir_ref ref)
+{
+	ir_insn *op2_insn;
+	ir_insn *insn = &ctx->ir_base[ref];
+	uint32_t store_rule;
+	ir_op load_op;
+
+	switch (insn->op) {
+		case IR_EQ:
+		case IR_NE:
+		case IR_LT:
+		case IR_GE:
+		case IR_LE:
+		case IR_GT:
+		case IR_ULT:
+		case IR_UGE:
+		case IR_ULE:
+		case IR_UGT:
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
+#if IR_X86_I64
+				if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+					ir_match_fuse_load_cmp_int(ctx, insn, ref);
+					return IR_CMP_I64;
+				}
+#endif
+				if (IR_IS_CONST_REF(insn->op2)
+				 && !IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op)
+				 && ctx->ir_base[insn->op2].val.i64 == 0
+				 && insn->op1 == ref - 1) { /* previous instruction */
+					ir_insn *op1_insn = &ctx->ir_base[insn->op1];
+
+					if (op1_insn->op == IR_AND && ctx->use_lists[insn->op1].count == 1) {
+						/* v = AND(_, _); CMP(v, 0) => SKIP_TEST; TEST */
+						ir_match_fuse_load_test_int(ctx, op1_insn, ref);
+						if (sizeof(void*) == 8
+						 && ir_type_size[op1_insn->type] == 8
+						 && IR_IS_CONST_REF(op1_insn->op2)
+						 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)
+						 && !IR_IS_SIGNED_32BIT(ctx->ir_base[op1_insn->op2].val.i64)
+						 && IR_IS_POWER_OF_TWO(ctx->ir_base[op1_insn->op2].val.u64)) {
+							ctx->rules[insn->op1] = IR_FUSED | IR_TEST_BIT;
+							return IR_TESTCC_BIT;
+						} else {
+							ctx->rules[insn->op1] = IR_FUSED | IR_TEST_INT;
+							return IR_TESTCC_INT;
+						}
+					} else if ((op1_insn->op == IR_OR || op1_insn->op == IR_AND || op1_insn->op == IR_XOR) ||
+							/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
+							((op1_insn->op == IR_ADD || op1_insn->op == IR_SUB) &&
+								(insn->op == IR_EQ || insn->op == IR_NE ||
+									insn->op == IR_LT || insn->op == IR_GE))) {
+						/* v = BINOP(_, _); CMP(v, 0) => BINOP; SETCC */
+						if (ir_op_flags[op1_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+							ir_match_fuse_load_commutative_int(ctx, op1_insn, ref);
+							ctx->rules[insn->op1] = IR_BINOP_INT | IR_MAY_SWAP;
+						} else {
+							ir_match_fuse_load(ctx, op1_insn->op2, ref);
+							ctx->rules[insn->op1] = IR_BINOP_INT;
+						}
+						return IR_SETCC_INT;
+					}
+				}
+				ir_match_fuse_load_cmp_int(ctx, insn, ref);
+				return IR_CMP_INT;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+binop_vector:
+				if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+					if (ir_may_fuse_load_vector(ctx, insn->type)) {
+						ir_match_fuse_load_commutative_fp(ctx, insn, ref);
+					}
+					if (ctx->mflags & IR_X86_AVX) {
+						return IR_VECTOR_BINOP_AVX;
+					} else {
+						return IR_VECTOR_BINOP_SSE2 | IR_MAY_SWAP;
+					}
+				} else {
+					if (ir_may_fuse_load_vector(ctx, insn->type)
+					 && !(insn->op >= IR_LT && insn->op <= IR_UGT)) {
+						/* load may be fused only into some vector comparison instructions */
+						ir_match_fuse_load(ctx, insn->op2, ref);
+					}
+					if (ctx->mflags & IR_X86_AVX) {
+						return IR_VECTOR_BINOP_AVX;
+					} else {
+						return IR_VECTOR_BINOP_SSE2;
+					}
+				}
+#endif
+			} else {
+				ir_match_fuse_load_cmp_fp(ctx, insn, ref);
+				return IR_CMP_FP;
+			}
+			break;
+		case IR_ORDERED:
+		case IR_UNORDERED:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				goto binop_vector;
+			}
+#endif
+			ir_match_fuse_load_cmp_fp(ctx, insn, ref);
+			return IR_CMP_FP;
+		case IR_ADD:
+		case IR_SUB:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+						ir_match_fuse_load_commutative_int(ctx, insn, ref);
+						return IR_TWO_REGS | IR_BINOP_I64 | IR_MAY_SWAP;
+					} else {
+						ir_match_fuse_load(ctx, insn->op2, ref);
+						return IR_TWO_REGS | IR_BINOP_I64;
+					}
+				}
+#endif
+				if (IR_IS_CONST_REF(insn->op2)) {
+					op2_insn = &ctx->ir_base[insn->op2];
+					if (IR_IS_CONST_REF(insn->op1)) {
+						ir_insn *op1_insn = &ctx->ir_base[insn->op1];
+
+						if (insn->op == IR_ADD
+						 && IR_IS_SYM_CONST(op1_insn->op)
+						 && !IR_IS_SYM_CONST(op2_insn->op)
+						 && IR_IS_SIGNED_32BIT((intptr_t)ir_sym_val(ctx, op1_insn) + (intptr_t)op2_insn->val.i64)) {
+							return IR_LEA_SYM_O;
+						} else if (insn->op == IR_ADD
+						 && IR_IS_SYM_CONST(op2_insn->op)
+						 && !IR_IS_SYM_CONST(op1_insn->op)
+						 && IR_IS_SIGNED_32BIT((intptr_t)ir_sym_val(ctx, op2_insn) + (intptr_t)op1_insn->val.i64)) {
+							return IR_LEA_O_SYM;
+						}
+						// const
+						// TODO: add support for sym+offset ???
+					} else if (IR_IS_SYM_CONST(op2_insn->op)) {
+						if (insn->op == IR_ADD && ir_may_fuse_addr(ctx, op2_insn)) {
+							goto lea;
+						}
+						/* pass */
+					} else if (op2_insn->val.i64 == 0) {
+						return IR_COPY_INT | IR_MAY_REUSE;
+					} else if ((ir_type_size[insn->type] >= 4 && insn->op == IR_ADD && IR_IS_SIGNED_32BIT(op2_insn->val.i64)) ||
+							(ir_type_size[insn->type] >= 4 && insn->op == IR_SUB && IR_IS_SIGNED_NEG_32BIT(op2_insn->val.i64))) {
+lea:
+						if (ctx->use_lists[insn->op1].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op1)) {
+							uint32_t rule = ctx->rules[insn->op1];
+
+							if (!rule) {
+								ctx->rules[insn->op1] = rule = ir_match_insn(ctx, insn->op1);
+							}
+							if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
+								/* z = MUL(Y, 2|4|8) ... ADD(z, imm32) => SKIP ... LEA [Y*2|4|8+im32] */
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
+								return IR_LEA_SI_O;
+							} else if (rule == IR_LEA_SIB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SIB)) {
+								/* z = ADD(X, MUL(Y, 2|4|8)) ... ADD(z, imm32) => SKIP ... LEA [X+Y*2|4|8+im32] */
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SIB;
+								return IR_LEA_SIB_O;
+							} else if (rule == IR_LEA_IB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_IB)) {
+								/* z = ADD(X, Y) ... ADD(z, imm32) => SKIP ... LEA [X+Y+im32] */
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_IB;
+								return IR_LEA_IB_O;
+							} else if (rule == IR_LEA_B_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_B_SI)) {
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_B_SI;
+								return IR_LEA_B_SI_O;
+							} else if (rule == IR_LEA_SI_B || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI_B)) {
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SI_B;
+								return IR_LEA_SI_B_O;
+							}
+						}
+						/* ADD(X, imm32) => LEA [X+imm32] */
+						return IR_LEA_OB;
+					} else if (op2_insn->val.i64 == 1 || op2_insn->val.i64 == -1) {
+						if (insn->op == IR_ADD) {
+							if (op2_insn->val.i64 == 1) {
+								/* ADD(_, 1) => INC */
+								return IR_INC;
+						    } else {
+								/* ADD(_, -1) => DEC */
+								return IR_DEC;
+						    }
+						} else {
+							if (op2_insn->val.i64 == 1) {
+								/* SUB(_, 1) => DEC */
+								return IR_DEC;
+						    } else {
+								/* SUB(_, -1) => INC */
+								return IR_INC;
+						    }
+						}
+					}
+				} else if (insn->op == IR_ADD && ir_type_size[insn->type] >= 4 && EXPECTED(!IR_IS_CONST_REF(insn->op1))) {
+					if (insn->op1 != insn->op2) {
+						if (ctx->use_lists[insn->op1].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op1)) {
+							uint32_t rule =ctx->rules[insn->op1];
+							if (!rule) {
+								ctx->rules[insn->op1] = rule = ir_match_insn(ctx, insn->op1);
+							}
+							if (rule == IR_LEA_OB) {
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_OB;
+								if (ctx->use_lists[insn->op2].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op2)) {
+									rule = ctx->rules[insn->op2];
+									if (!rule) {
+										ctx->rules[insn->op2] = rule = ir_match_insn(ctx, insn->op2);
+									}
+									if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
+										/* x = ADD(X, imm32) ... y = MUL(Y, 2|4|8) ... ADD(x, y) => SKIP ... SKIP ... LEA */
+										ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
+										return IR_LEA_OB_SI;
+									}
+								}
+								/* x = ADD(X, imm32) ... ADD(x, Y) => SKIP ... LEA */
+								return IR_LEA_OB_I;
+							} else if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
+								ctx->rules[insn->op1] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
+								if (ctx->use_lists[insn->op2].count == 1) {
+									rule = ctx->rules[insn->op2];
+									if (!rule) {
+										ctx->rules[insn->op2] = rule = ir_match_insn(ctx, insn->op2);
+									}
+									if (rule == IR_LEA_OB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_OB)) {
+										/* x = ADD(X, imm32) ... y = MUL(Y, 2|4|8) ... ADD(y, x) => SKIP ... SKIP ... LEA */
+										ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_OB;
+										return IR_LEA_SI_OB;
+									}
+								}
+								/* x = MUL(X, 2|4|8) ... ADD(x, Y) => SKIP ... LEA */
+								return IR_LEA_SI_B;
+							}
+						}
+						if (ctx->use_lists[insn->op2].count == 1 || ir_match_fuse_addr_all_useges(ctx, insn->op2)) {
+							uint32_t rule = ctx->rules[insn->op2];
+							if (!rule) {
+								ctx->rules[insn->op2] = rule = ir_match_insn(ctx, insn->op2);
+							}
+							if (rule == IR_LEA_OB || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_OB)) {
+								ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_OB;
+								/* x = ADD(X, imm32) ... ADD(Y, x) => SKIP ... LEA */
+								return IR_LEA_I_OB;
+							} else if (rule == IR_LEA_SI || rule == (IR_FUSED | IR_SIMPLE | IR_LEA_SI)) {
+								ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_LEA_SI;
+								/* x = MUL(X, 2|4|8) ... ADD(Y, x) => SKIP ... LEA */
+								return IR_LEA_B_SI;
+							}
+						}
+					}
+					/* ADD(X, Y) => LEA [X + Y] */
+					return IR_LEA_IB;
+				}
+binop_int:
+				if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+					ir_match_fuse_load_commutative_int(ctx, insn, ref);
+					return IR_BINOP_INT | IR_MAY_SWAP;
+				} else {
+					ir_match_fuse_load(ctx, insn->op2, ref);
+					return IR_BINOP_INT;
+				}
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				goto binop_vector;
+#endif
+			} else {
+binop_fp:
+				if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+					ir_match_fuse_load_commutative_fp(ctx, insn, ref);
+					if (ctx->mflags & IR_X86_AVX) {
+						return IR_BINOP_AVX;
+					} else {
+						return IR_BINOP_SSE2 | IR_MAY_SWAP;
+					}
+				} else {
+					ir_match_fuse_load(ctx, insn->op2, ref);
+					if (ctx->mflags & IR_X86_AVX) {
+						return IR_BINOP_AVX;
+					} else {
+						return IR_BINOP_SSE2;
+					}
+				}
+			}
+			break;
+		case IR_MUL:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					ir_match_fuse_load(ctx, insn->op2, ref);
+					return IR_TWO_REGS | IR_MUL_I64;
+				}
+#endif
+				if (IR_IS_CONST_REF(insn->op2)) {
+					op2_insn = &ctx->ir_base[insn->op2];
+					if (IR_IS_SYM_CONST(op2_insn->op)) {
+						/* pass */
+					} else if (IR_IS_CONST_REF(insn->op1)) {
+						// const
+					} else if (op2_insn->val.u64 == 0) {
+						// 0
+					} else if (op2_insn->val.u64 == 1) {
+						return IR_COPY_INT | IR_MAY_REUSE;
+					} else if (ir_type_size[insn->type] >= 4 &&
+							(op2_insn->val.u64 == 2 || op2_insn->val.u64 == 4 || op2_insn->val.u64 == 8)) {
+						/* MUL(X, 2|4|8) => LEA [X*2|4|8] */
+						return IR_LEA_SI;
+					} else if (ir_type_size[insn->type] >= 4 &&
+							(op2_insn->val.u64 == 3 || op2_insn->val.u64 == 5 || op2_insn->val.u64 == 9)) {
+						/* MUL(X, 3|5|9) => LEA [X+X*2|4|8] */
+						return IR_LEA_SIB;
+					} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
+						/* MUL(X, PWR2) => SHL */
+						return IR_MUL_PWR2;
+					} else if (IR_IS_TYPE_SIGNED(insn->type)
+					 && ir_type_size[insn->type] != 1
+					 && IR_IS_SIGNED_32BIT(op2_insn->val.i64)
+					 && !IR_IS_CONST_REF(insn->op1)) {
+						/* MUL(_, imm32) => IMUL */
+						ir_match_fuse_load(ctx, insn->op1, ref);
+						return IR_IMUL3;
+					}
+				}
+				/* Prefer IMUL over MUL because it's more flexible and uses less registers ??? */
+//				if (IR_IS_TYPE_SIGNED(insn->type) && ir_type_size[insn->type] != 1) {
+				if (ir_type_size[insn->type] != 1) {
+					goto binop_int;
+				}
+				ir_match_fuse_load(ctx, insn->op2, ref);
+				return IR_MUL_INT;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				ir_type element_type = IR_VECTOR_BASE_TYPE(insn->type);
+				if (IR_IS_TYPE_INT(element_type)
+				 && (ir_type_size[element_type] == 1 || ir_type_size[element_type] == 8)) {
+					return IR_VECTOR_BINOP_EXPAND;
+				}
+				goto binop_vector;
+#endif
+			} else {
+				goto binop_fp;
+			}
+			break;
+		case IR_ADD_OV:
+		case IR_SUB_OV:
+			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				if (ir_op_flags[insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+					ir_match_fuse_load_commutative_int(ctx, insn, ref);
+					return IR_TWO_REGS | IR_BINOP_I64 | IR_MAY_SWAP;
+				} else {
+					ir_match_fuse_load(ctx, insn->op2, ref);
+					return IR_TWO_REGS | IR_BINOP_I64;
+				}
+			}
+#endif
+			goto binop_int;
+		case IR_MUL_OV:
+			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				ir_match_fuse_load(ctx, insn->op2, ref);
+				return IR_TWO_REGS | IR_MUL_OV_I64;
+			}
+#endif
+			if (IR_IS_TYPE_SIGNED(insn->type) && ir_type_size[insn->type] != 1) {
+				if (IR_IS_CONST_REF(insn->op2)) {
+					op2_insn = &ctx->ir_base[insn->op2];
+					if (!IR_IS_SYM_CONST(op2_insn->op)
+					 && IR_IS_SIGNED_32BIT(op2_insn->val.i64)
+					 && !IR_IS_CONST_REF(insn->op1)) {
+						/* MUL(_, imm32) => IMUL */
+						ir_match_fuse_load(ctx, insn->op1, ref);
+						return IR_IMUL3;
+					}
+				}
+				goto binop_int;
+			}
+			ir_match_fuse_load(ctx, insn->op2, ref);
+			return IR_MUL_INT;
+		case IR_DIV:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					ir_match_fuse_load(ctx, insn->op1, ref);
+					ir_match_fuse_load(ctx, insn->op2, ref);
+					return IR_TWO_REGS | IR_BINOP_HELPER_I64;
+				}
+#endif
+				if (IR_IS_CONST_REF(insn->op2)) {
+					op2_insn = &ctx->ir_base[insn->op2];
+					if (IR_IS_SYM_CONST(op2_insn->op)) {
+						/* pass */
+					} else if (IR_IS_CONST_REF(insn->op1)) {
+						// const
+					} else if (op2_insn->val.u64 == 1) {
+						return IR_COPY_INT | IR_MAY_REUSE;
+					} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
+						/* DIV(X, PWR2) => SHR */
+						if (IR_IS_TYPE_UNSIGNED(insn->type)) {
+							return IR_DIV_PWR2;
+						} else {
+							return IR_SDIV_PWR2;
+						}
+					}
+				}
+				ir_match_fuse_load(ctx, insn->op2, ref);
+				return IR_DIV_INT;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (IR_IS_TYPE_INT(IR_VECTOR_BASE_TYPE(insn->type))) {
+					return IR_VECTOR_BINOP_EXPAND;
+				}
+				goto binop_vector;
+#endif
+			} else {
+				goto binop_fp;
+			}
+			break;
+		case IR_MOD:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_BINOP_EXPAND;
+			}
+#endif
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				ir_match_fuse_load(ctx, insn->op1, ref);
+				ir_match_fuse_load(ctx, insn->op2, ref);
+				return IR_TWO_REGS | IR_BINOP_HELPER_I64;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.u64 == 1) {
+					// 0
+				} else if (IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
+					/* MOD(X, PWR2) => AND */
+					if (IR_IS_TYPE_UNSIGNED(insn->type)) {
+						return IR_MOD_PWR2;
+					} else {
+						return IR_SMOD_PWR2;
+					}
+				}
+			}
+			ir_match_fuse_load(ctx, insn->op2, ref);
+			return IR_MOD_INT;
+		case IR_BSWAP:
+			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | IR_OP_I64;
+			}
+#endif
+			return IR_OP_INT;
+		case IR_NOT:
+			if (insn->type == IR_BOOL) {
+				if (ctx->ir_base[insn->op1].type == IR_BOOL) {
+					return IR_BOOL_NOT;
+				} else {
+					IR_ASSERT(IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type));
+					return IR_BOOL_NOT_INT;
+				}
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (ir_may_fuse_load_vector(ctx, insn->type)) {
+					ir_match_fuse_load(ctx, insn->op1, ref);
+				}
+				return IR_VECTOR_OP;
+#endif
+			} else {
+				IR_ASSERT(IR_IS_TYPE_INT(insn->type));
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_OP_I64;
+				}
+#endif
+				return IR_OP_INT;
+			}
+			break;
+		case IR_NEG:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_OP_I64;
+				}
+#endif
+				return IR_OP_INT;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (ir_may_fuse_load_vector(ctx, insn->type)) {
+					ir_match_fuse_load(ctx, insn->op1, ref);
+				}
+				return IR_VECTOR_OP;
+#endif
+			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+				return IR_OP_FP;
+			}
+		case IR_ABS:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_OP_I64;
+				}
+#endif
+				return IR_ABS_INT; // movl %edi, %eax; negl %eax; cmovs %edi, %eax
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_OP;
+#endif
+			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+				return IR_OP_FP;
+			}
+		case IR_OR:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				goto binop_vector;
+			}
+#endif
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				ir_match_fuse_load_commutative_int(ctx, insn, ref);
+				return IR_TWO_REGS | IR_BINOP_I64 | IR_MAY_SWAP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.i64 == 0) {
+					return IR_COPY_INT | IR_MAY_REUSE;
+				} else if (op2_insn->val.i64 == -1) {
+					// -1
+				} else if (ir_type_size[insn->type] == 8
+						&& !IR_IS_SIGNED_32BIT(op2_insn->val.i64)
+						&& IR_IS_POWER_OF_TWO(op2_insn->val.u64)) {
+					/* OR(X, PWR2) => BTS */
+					return IR_BIT_OP;
+				}
+			}
+			goto binop_int;
+		case IR_AND:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				goto binop_vector;
+			}
+#endif
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				ir_match_fuse_load_commutative_int(ctx, insn, ref);
+				return IR_TWO_REGS | IR_BINOP_I64 | IR_MAY_SWAP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.i64 == 0) {
+					// 0
+				} else if (op2_insn->val.i64 == -1) {
+					return IR_COPY_INT | IR_MAY_REUSE;
+				} else if (ir_type_size[insn->type] == 8
+						&& !IR_IS_SIGNED_32BIT(op2_insn->val.i64)
+						&& IR_IS_POWER_OF_TWO(~op2_insn->val.u64)) {
+					/* AND(X, ~PWR2) => BTR */
+					return IR_BIT_OP;
+#ifdef IR_TARGET_X64
+				} else if (op2_insn->val.u64 == 0xff || op2_insn->val.u64 == 0xffff || op2_insn->val.u64 == 0xffffffff) {
+#else
+				} else if (op2_insn->val.u64 == 0xffff) {
+#endif
+					/* AND(X, 0xff) => MOVZX */
+					if (ir_type_size[insn->type] < 8
+					 && (1ULL << (ir_type_size[insn->type] * 8)) - 1 == op2_insn->val.u64) {
+						return IR_COPY_INT | IR_MAY_REUSE;
+					} else {
+						return IR_AND_ZEXT;
+					}
+				}
+			}
+			goto binop_int;
+		case IR_XOR:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				goto binop_vector;
+			}
+#endif
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				ir_match_fuse_load_commutative_int(ctx, insn, ref);
+				return IR_TWO_REGS | IR_BINOP_I64 | IR_MAY_SWAP;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				}
+			}
+			goto binop_int;
+		case IR_SHL:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				ir_type element_type = IR_VECTOR_BASE_TYPE(insn->type);
+
+				if (IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op2].type) ||
+						element_type == IR_I8 || element_type == IR_U8) {
+					return IR_VECTOR_BINOP_EXPAND;
+				}
+				goto binop_vector;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.u64 == 0) {
+#if IR_X86_I64
+					if (insn->type == IR_I64 || insn->type == IR_U64) {
+						return IR_TWO_REGS | IR_OP_I64;
+					}
+#endif
+					return IR_COPY_INT | IR_MAY_REUSE;
+				} else if (ir_type_size[insn->type] >= 4) {
+					if (op2_insn->val.u64 == 1) {
+						// lea [op1*2]
+					} else if (op2_insn->val.u64 == 2) {
+						// lea [op1*4]
+					} else if (op2_insn->val.u64 == 3) {
+						// lea [op1*8]
+					}
+				}
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_SHIFT_CONST_I64;
+				}
+#endif
+				return IR_SHIFT_CONST;
+			}
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | IR_SHIFT_I64;
+			}
+#endif
+			return IR_SHIFT;
+		case IR_ROL:
+		case IR_ROR:
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				ir_match_fuse_load(ctx, insn->op1, ref);
+				ir_match_fuse_load(ctx, insn->op2, ref);
+				return IR_TWO_REGS | IR_BINOP_HELPER_I64;
+			}
+#endif
+			IR_FALLTHROUGH;
+		case IR_SHR:
+		case IR_SAR:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				ir_type element_type = IR_VECTOR_BASE_TYPE(insn->type);
+
+				if (IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op2].type)  ||
+						element_type == IR_I8 || element_type == IR_U8 ||
+						(element_type == IR_I64 && insn->op == IR_SAR)) {
+					return IR_VECTOR_BINOP_EXPAND;
+				}
+				goto binop_vector;
+			}
+#endif
+			if (IR_IS_CONST_REF(insn->op2)) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (IR_IS_SYM_CONST(op2_insn->op)) {
+					/* pass */
+				} else if (IR_IS_CONST_REF(insn->op1)) {
+					// const
+				} else if (op2_insn->val.u64 == 0) {
+#if IR_X86_I64
+					if (insn->type == IR_I64 || insn->type == IR_U64) {
+						return IR_TWO_REGS | IR_OP_I64;
+					}
+#endif
+					return IR_COPY_INT | IR_MAY_REUSE;
+				}
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_SHIFT_CONST_I64;
+				}
+#endif
+				return IR_SHIFT_CONST;
+			}
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | IR_SHIFT_I64;
+			}
+#endif
+			return IR_SHIFT;
+		case IR_MIN:
+		case IR_MAX:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_MIN_MAX_I64 | IR_MAY_SWAP;
+				}
+#endif
+				return IR_MIN_MAX_INT | IR_MAY_SWAP;
+#if IR_SIMD
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				goto binop_vector;
+#endif
+			} else {
+				goto binop_fp;
+			}
+			break;
+		case IR_COPY:
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_OP_I64;
+				}
+#endif
+				return IR_COPY_INT | IR_MAY_REUSE;
+			} else {
+				return IR_COPY_FP | IR_MAY_REUSE;
+			}
+			break;
+		case IR_CALL:
+			if (IR_IS_CONST_REF(insn->op2)) {
+				const ir_insn *func = &ctx->ir_base[insn->op2];
+
+				if (func->op == IR_FUNC && func->proto) {
+					uint32_t rule = ir_match_builtin_call(ctx, func);
+
+					if (rule) {
+						return rule;
+					}
+				}
+			}
+			ctx->flags2 |= IR_HAS_CALLS;
+			IR_FALLTHROUGH;
+		case IR_TAILCALL:
+		case IR_IJMP:
+			if (!IR_IS_CONST_REF(insn->op2)) {
+				if (ctx->ir_base[insn->op2].op == IR_PROTO) {
+					if (IR_IS_CONST_REF(ctx->ir_base[insn->op2].op1)) {
+						ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_PROTO;
+					} else {
+						ir_match_fuse_load(ctx, ctx->ir_base[insn->op2].op1, ref);
+						if (ctx->rules[ctx->ir_base[insn->op2].op1] & IR_FUSED) {
+							ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_PROTO;
+						}
+				   }
+				} else {
+					ir_match_fuse_load(ctx, insn->op2, ref);
+				}
+			}
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | insn->op;
+			}
+#endif
+			return insn->op;
+		case IR_IGOTO:
+			if (ctx->ir_base[insn->op1].op == IR_MERGE || ctx->ir_base[insn->op1].op == IR_LOOP_BEGIN) {
+				ir_insn *merge = &ctx->ir_base[insn->op1];
+				ir_ref *p, n = merge->inputs_count;
+
+				for (p = merge->ops + 1; n > 0; p++, n--) {
+					ir_ref input = *p;
+					IR_ASSERT(ctx->ir_base[input].op == IR_END || ctx->ir_base[input].op == IR_LOOP_END);
+					ctx->rules[input] = IR_IGOTO_DUP;
+				}
+			}
+			ir_match_fuse_load(ctx, insn->op2, ref);
+			return insn->op;
+		case IR_VAR:
+			return IR_STATIC_ALLOCA;
+		case IR_PARAM:
+			if (ctx->value_params && ctx->value_params[insn->op3 - 1].align) {
+				const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(ctx->flags);
+				if (cc->pass_struct_by_val) {
+					return IR_STATIC_ALLOCA;
+				}
+			}
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return ctx->use_lists[ref].count > 0 ? (IR_TWO_REGS | IR_PARAM_I64) : (IR_SKIPPED | IR_PARAM_I64);
+			}
+#endif
+			return ctx->use_lists[ref].count > 0 ? IR_PARAM : IR_SKIPPED | IR_PARAM;
+		case IR_ALLOCA:
+			/* alloca() may be used only in functions */
+			if (ctx->flags & IR_FUNCTION) {
+				if (IR_IS_CONST_REF(insn->op2) && ctx->cfg_map[ref] == 1) {
+					ir_insn *val = &ctx->ir_base[insn->op2];
+
+					if (!IR_IS_SYM_CONST(val->op)) {
+						return IR_STATIC_ALLOCA;
+					}
+				}
+				ctx->flags |= IR_USE_FRAME_POINTER;
+				ctx->flags2 |= IR_HAS_ALLOCA | IR_16B_FRAME_ALIGNMENT;
+			}
+			return IR_ALLOCA;
+		case IR_VSTORE:
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
+				store_rule = IR_VSTORE_INT;
+				load_op = IR_VLOAD;
+store_int:
+				if (ir_in_same_block(ctx, insn->op3)
+				 && (ctx->use_lists[insn->op3].count == 1 ||
+				     (ctx->use_lists[insn->op3].count == 2
+				   && (ctx->ir_base[insn->op3].op == IR_ADD_OV ||
+				       ctx->ir_base[insn->op3].op == IR_SUB_OV)
+				   && insn->op3 == ref - 1 /* OVERFLOW must no be between ADD_OV and STORE */))) {
+					ir_insn *op_insn = &ctx->ir_base[insn->op3];
+					uint32_t rule = ctx->rules[insn->op3];
+
+					if (!rule) {
+						ctx->rules[insn->op3] = rule = ir_match_insn(ctx, insn->op3);
+					}
+					if (((rule & IR_RULE_MASK) == IR_BINOP_INT && op_insn->op != IR_MUL) || rule == IR_LEA_OB || rule == IR_LEA_IB) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = BINOP(l, _) ... STORE(l, a, v) => SKIP ... SKIP_MEM_BINOP ... MEM_BINOP */
+							ctx->rules[insn->op3] = IR_FUSED | IR_BINOP_INT;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							if (!IR_IS_CONST_REF(op_insn->op2)
+							 && ctx->rules[op_insn->op2] == (IR_FUSED|IR_SIMPLE|IR_LOAD)) {
+								ctx->rules[op_insn->op2] = ctx->ir_base[op_insn->op2].op == IR_VLOAD ? IR_VLOAD : IR_LOAD_INT;
+							}
+							return IR_MEM_BINOP_INT;
+						} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
+						 && insn->op1 == op_insn->op2
+						 && ctx->ir_base[op_insn->op2].op == load_op
+						 && ctx->ir_base[op_insn->op2].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op2].count == 2) {
+							/* l = LOAD(_, a) ... v = BINOP(_, l) ... STORE(l, a, v) => SKIP ... SKIP_MEM_BINOP ... MEM_BINOP */
+							ir_swap_ops(op_insn);
+							ctx->rules[insn->op3] = IR_FUSED | IR_BINOP_INT;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_BINOP_INT;
+						}
+					} else if (rule == IR_INC) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = INC(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_INC */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_INC;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_INC;
+						}
+					} else if (rule == IR_DEC) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2){
+							/* l = LOAD(_, a) ... v = DEC(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_DEC */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_DEC;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_DEC;
+						}
+					} else if (rule == IR_MUL_PWR2) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = MUL_PWR2(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_MUL_PWR2 */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_MUL_PWR2;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_MUL_PWR2;
+						}
+					} else if (rule == IR_DIV_PWR2) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = DIV_PWR2(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_DIV_PWR2 */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_DIV_PWR2;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_DIV_PWR2;
+						}
+					} else if (rule == IR_MOD_PWR2) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = MOD_PWR2(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_MOD_PWR2 */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_MOD_PWR2;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_MOD_PWR2;
+						}
+					} else if (rule == IR_SHIFT) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = SHIFT(l, _) ... STORE(l, a, v) => SKIP ... SKIP_SHIFT ... MEM_SHIFT */
+							ctx->rules[insn->op3] = IR_FUSED | IR_SHIFT;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_SHIFT;
+						}
+					} else if (rule == IR_SHIFT_CONST) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = SHIFT(l, CONST) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_SHIFT_CONST */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_SHIFT_CONST;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_SHIFT_CONST;
+						}
+					} else if (rule == IR_OP_INT && op_insn->op != IR_BSWAP) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == load_op
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op2
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = LOAD(_, a) ... v = OP(l) ... STORE(l, a, v) => SKIP ... SKIP ... MEM_OP */
+							ctx->rules[insn->op3] = IR_SKIPPED | IR_OP_INT;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | load_op;
+							return IR_MEM_OP_INT;
+						}
+					} else if (rule == IR_CMP_INT && load_op == IR_LOAD) {
+						/* c = CMP(_, _) ... STORE(c) => SKIP_CMP ... CMP_AND_STORE_INT */
+						ctx->rules[insn->op3] = IR_FUSED | IR_CMP_INT;
+						return IR_CMP_AND_STORE_INT;
+					}
+				}
+				return store_rule;
+			} else {
+				return IR_VSTORE_FP;
+			}
+			break;
+		case IR_VSTORE_v:
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
+				return IR_VSTORE_INT;
+			} else {
+				return IR_VSTORE_FP;
+			}
+			break;
+		case IR_LOAD:
+		case IR_LOAD_v:
+			if (ctx->ir_base[insn->op2].op == IR_TLS_ADDR && ir_may_fuse_tls_addr(ctx, insn->op2)) {
+				ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_TLS_ADDR;
+				return IR_TLS_LOAD;
+			}
+			ir_match_fuse_addr(ctx, insn->op2);
+			if (IR_IS_TYPE_INT(insn->type)) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					return IR_TWO_REGS | IR_LOAD_I64;
+				}
+#endif
+				return IR_LOAD_INT;
+			} else {
+				return IR_LOAD_FP;
+			}
+			break;
+		case IR_STORE:
+			if (ctx->ir_base[insn->op2].op == IR_TLS_ADDR && ir_may_fuse_tls_addr(ctx, insn->op2)) {
+				ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_TLS_ADDR;
+				return IR_TLS_STORE;
+			}
+			ir_match_fuse_addr(ctx, insn->op2);
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
+				store_rule = IR_STORE_INT;
+				load_op = IR_LOAD;
+				goto store_int;
+			} else {
+				return IR_STORE_FP;
+			}
+			break;
+		case IR_STORE_v:
+			if (ctx->ir_base[insn->op2].op == IR_TLS_ADDR && ir_may_fuse_tls_addr(ctx, insn->op2)) {
+				ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_TLS_ADDR;
+				return IR_TLS_STORE;
+			}
+			ir_match_fuse_addr(ctx, insn->op2);
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type)) {
+				return IR_STORE_INT;
+			} else {
+				return IR_STORE_FP;
+			}
+			break;
+		case IR_RLOAD:
+			if (IR_REGSET_IN(IR_REGSET_UNION((ir_regset)ctx->fixed_regset, IR_REGSET_FIXED), insn->op2)) {
+				return IR_SKIPPED | IR_RLOAD;
+			}
+			return IR_RLOAD;
+		case IR_RSTORE:
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
+				if (ir_in_same_block(ctx, insn->op2)
+				 && ctx->use_lists[insn->op2].count == 1
+				 && IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
+					ir_insn *op_insn = &ctx->ir_base[insn->op2];
+
+					if (op_insn->op == IR_ADD ||
+				        op_insn->op == IR_SUB ||
+//				        op_insn->op == IR_MUL ||
+				        op_insn->op == IR_OR  ||
+				        op_insn->op == IR_AND ||
+				        op_insn->op == IR_XOR) {
+						if (insn->op1 == op_insn->op1
+						 && ctx->ir_base[op_insn->op1].op == IR_RLOAD
+						 && ctx->ir_base[op_insn->op1].op2 == insn->op3
+						 && ctx->use_lists[op_insn->op1].count == 2) {
+							/* l = RLOAD(r) ... v = BINOP(l, _) ... RSTORE(l, r, v) => SKIP ... SKIP_REG_BINOP ... REG_BINOP */
+							ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | IR_RLOAD;
+							return IR_REG_BINOP_INT;
+						} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
+						 && insn->op1 == op_insn->op2
+						 && ctx->ir_base[op_insn->op2].op == IR_RLOAD
+						 && ctx->ir_base[op_insn->op2].op2 == insn->op3
+						 && ctx->use_lists[op_insn->op2].count == 2) {
+							/* l = RLOAD(r) ... v = BINOP(x, l) ... RSTORE(l, r, v) => SKIP ... SKIP_REG_BINOP ... REG_BINOP */
+							ir_swap_ops(op_insn);
+							ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
+							ctx->rules[op_insn->op1] = IR_SKIPPED | IR_RLOAD;
+							return IR_REG_BINOP_INT;
+						}
+					}
+				}
+			}
+			ir_match_fuse_load(ctx, insn->op2, ref);
+			return IR_RSTORE;
+		case IR_START:
+		case IR_BEGIN:
+		case IR_IF_TRUE:
+		case IR_IF_FALSE:
+		case IR_CASE_VAL:
+		case IR_CASE_RANGE:
+		case IR_CASE_DEFAULT:
+		case IR_MERGE:
+		case IR_LOOP_BEGIN:
+		case IR_UNREACHABLE:
+			return IR_SKIPPED | insn->op;
+		case IR_RETURN:
+			if (!insn->op2) {
+				return IR_RETURN_VOID;
+			} else if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
+#if IR_X86_I64
+				if (ctx->ir_base[insn->op2].type == IR_I64 || ctx->ir_base[insn->op2].type == IR_U64) {
+					return IR_RETURN_I64;
+				}
+#endif
+				return IR_RETURN_INT;
+			} else {
+				return IR_RETURN_FP;
+			}
+		case IR_IF:
+			if (!IR_IS_CONST_REF(insn->op2) && (ctx->use_lists[insn->op2].count == 1 || all_usages_are_fusable(ctx, insn->op2))) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (op2_insn->op >= IR_EQ && op2_insn->op <= IR_UNORDERED) {
+					if (IR_IS_TYPE_INT(ctx->ir_base[op2_insn->op1].type)) {
+#if IR_X86_I64
+						if (ctx->ir_base[op2_insn->op1].type == IR_I64 || ctx->ir_base[op2_insn->op1].type == IR_U64) {
+							if (ctx->use_lists[insn->op2].count == 1) {
+								ir_match_fuse_load_cmp_int(ctx, op2_insn, ref);
+							}
+							ctx->rules[insn->op2] = IR_FUSED | IR_CMP_I64;
+							return IR_CMP_AND_BRANCH_I64;
+						}
+#endif
+						if (IR_IS_CONST_REF(op2_insn->op2)
+						 && !IR_IS_SYM_CONST(ctx->ir_base[op2_insn->op2].op)
+						 && ctx->ir_base[op2_insn->op2].val.i64 == 0
+						 && op2_insn->op1 == insn->op2 - 1) { /* previous instruction */
+							ir_insn *op1_insn = &ctx->ir_base[op2_insn->op1];
+
+							if (op1_insn->op == IR_AND && ctx->use_lists[op2_insn->op1].count == 1) {
+								/* v = AND(_, _); c = CMP(v, 0) ... IF(c) => SKIP_TEST; SKIP ... TEST_AND_BRANCH */
+								if (ctx->use_lists[insn->op2].count == 1) {
+									ir_match_fuse_load_test_int(ctx, op1_insn, ref);
+								}
+								if (sizeof(void*) == 8
+								 && ir_type_size[op1_insn->type] == 8
+								 && IR_IS_CONST_REF(op1_insn->op2)
+								 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)
+								 && !IR_IS_SIGNED_32BIT(ctx->ir_base[op1_insn->op2].val.i64)
+								 && IR_IS_POWER_OF_TWO(ctx->ir_base[op1_insn->op2].val.u64)) {
+									ctx->rules[op2_insn->op1] = IR_FUSED | IR_TEST_BIT;
+									ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_NOP;
+									return IR_TEST_AND_BRANCH_BIT;
+								} else {
+									ctx->rules[op2_insn->op1] = IR_FUSED | IR_TEST_INT;
+									ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_NOP;
+									return IR_TEST_AND_BRANCH_INT;
+								}
+							} else if (insn->op2 == ref - 1 && /* previous instruction */
+									((op1_insn->op == IR_OR || op1_insn->op == IR_AND || op1_insn->op == IR_XOR) ||
+										/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
+										((op1_insn->op == IR_ADD || op1_insn->op == IR_SUB) &&
+											(op2_insn->op == IR_EQ || op2_insn->op == IR_NE ||
+												op2_insn->op == IR_LT || op2_insn->op == IR_GE)))) {
+								/* v = BINOP(_, _); c = CMP(v, 0) ... IF(c) => BINOP; SKIP_CMP ... JCC */
+								if (ir_op_flags[op1_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+									if (ctx->use_lists[insn->op2].count == 1) {
+										ir_match_fuse_load_commutative_int(ctx, op1_insn, ref);
+									}
+									ctx->rules[op2_insn->op1] = IR_BINOP_INT | IR_MAY_SWAP;
+								} else {
+									if (ctx->use_lists[insn->op2].count == 1) {
+										ir_match_fuse_load(ctx, op1_insn->op2, ref);
+									}
+									ctx->rules[op2_insn->op1] = IR_BINOP_INT;
+								}
+								ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
+								return IR_JCC_INT;
+							}
+						}
+						/* c = CMP(_, _) ... IF(c) => SKIP_CMP ... CMP_AND_BRANCH */
+						if (ctx->use_lists[insn->op2].count == 1) {
+							ir_match_fuse_load_cmp_int(ctx, op2_insn, ref);
+						}
+						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
+						return IR_CMP_AND_BRANCH_INT;
+					} else {
+						/* c = CMP(_, _) ... IF(c) => SKIP_CMP ... CMP_AND_BRANCH */
+						if (ctx->use_lists[insn->op2].count == 1) {
+							ir_match_fuse_load_cmp_fp_br(ctx, op2_insn, ref);
+						}
+						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_FP;
+						return IR_CMP_AND_BRANCH_FP;
+					}
+				} else if (op2_insn->op == IR_OVERFLOW && ir_in_same_block(ctx, insn->op2)) {
+					/* c = OVERFLOW(_) ... IF(c) => SKIP_OVERFLOW ... OVERFLOW_AND_BRANCH */
+					ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_OVERFLOW;
+					return IR_OVERFLOW_AND_BRANCH;
+#if IR_X86_I64
+				} else if (ctx->ir_base[op2_insn->op1].type == IR_I64 || ctx->ir_base[op2_insn->op1].type == IR_U64) {
+					/* pass */
+#endif
+				} else if (op2_insn->op == IR_AND) {
+					/* c = AND(_, _) ... IF(c) => SKIP_TEST ... TEST_AND_BRANCH */
+					ir_match_fuse_load_test_int(ctx, op2_insn, ref);
+					if (sizeof(void*) == 8
+					 && ir_type_size[op2_insn->type] == 8
+					 && IR_IS_CONST_REF(op2_insn->op2)
+					 && !IR_IS_SYM_CONST(ctx->ir_base[op2_insn->op2].op)
+					 && !IR_IS_SIGNED_32BIT(ctx->ir_base[op2_insn->op2].val.i64)
+					 && IR_IS_POWER_OF_TWO(ctx->ir_base[op2_insn->op2].val.u64)) {
+						ctx->rules[insn->op2] = IR_FUSED | IR_TEST_BIT;
+						return IR_TEST_AND_BRANCH_BIT;
+					} else {
+						ctx->rules[insn->op2] = IR_FUSED | IR_TEST_INT;
+						return IR_TEST_AND_BRANCH_INT;
+				    }
+				}
+			}
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
+#if IR_X86_I64
+				if (ctx->ir_base[insn->op2].type == IR_I64 || ctx->ir_base[insn->op2].type == IR_U64) {
+					ir_match_fuse_load(ctx, insn->op2, ref);
+					return IR_IF_I64;
+				} else
+#endif
+				if (insn->op2 == ref - 1) { /* previous instruction */
+					op2_insn = &ctx->ir_base[insn->op2];
+					if (op2_insn->op == IR_ADD ||
+					    op2_insn->op == IR_SUB ||
+//					    op2_insn->op == IR_MUL ||
+					    op2_insn->op == IR_OR  ||
+					    op2_insn->op == IR_AND ||
+					    op2_insn->op == IR_XOR) {
+
+							/* v = BINOP(_, _); IF(v) => BINOP; JCC */
+						if (ir_op_flags[op2_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+							ir_match_fuse_load_commutative_int(ctx, op2_insn, ref);
+							ctx->rules[insn->op2] = IR_BINOP_INT | IR_MAY_SWAP;
+						} else {
+							ir_match_fuse_load(ctx, op2_insn->op2, ref);
+							ctx->rules[insn->op2] = IR_BINOP_INT;
+						}
+						return IR_JCC_INT;
+					}
+				} else if (insn->op1 == ref - 1 /* previous instruction */
+				 && insn->op2 == ref - 2 /* previous instruction */
+				 && ctx->use_lists[insn->op2].count == 2
+				 && IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
+					ir_insn *store_insn = &ctx->ir_base[insn->op1];
+
+					if (store_insn->op == IR_STORE && store_insn->op3 == insn->op2) {
+						ir_insn *op_insn = &ctx->ir_base[insn->op2];
+
+						if (op_insn->op == IR_ADD ||
+						    op_insn->op == IR_SUB ||
+//						    op_insn->op == IR_MUL ||
+						    op_insn->op == IR_OR  ||
+						    op_insn->op == IR_AND ||
+						    op_insn->op == IR_XOR) {
+							if (ctx->ir_base[op_insn->op1].op == IR_LOAD
+							 && ctx->ir_base[op_insn->op1].op2 == store_insn->op2) {
+								if (ir_in_same_block(ctx, op_insn->op1)
+								 && ctx->use_lists[op_insn->op1].count == 2
+								 && store_insn->op1 == op_insn->op1) {
+									/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; JCC */
+									ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
+									ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
+									ir_match_fuse_addr(ctx, store_insn->op2);
+									ctx->rules[insn->op1] = IR_MEM_BINOP_INT;
+									return IR_JCC_INT;
+								}
+							} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
+							 && ctx->ir_base[op_insn->op2].op == IR_LOAD
+							 && ctx->ir_base[op_insn->op2].op2 == store_insn->op2) {
+								if (ir_in_same_block(ctx, op_insn->op2)
+								 && ctx->use_lists[op_insn->op2].count == 2
+								 && store_insn->op1 == op_insn->op2) {
+									/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; JCC */
+									ir_swap_ops(op_insn);
+									ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
+									ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
+									ir_match_fuse_addr(ctx, store_insn->op2);
+									ctx->rules[insn->op1] = IR_MEM_BINOP_INT;
+									return IR_JCC_INT;
+								}
+							}
+						}
+					}
+				}
+				ir_match_fuse_load(ctx, insn->op2, ref);
+				return IR_IF_INT;
+			} else {
+				IR_ASSERT(0 && "NIY IR_IF_FP");
+				break;
+			}
+		case IR_COND:
+			if (!IR_IS_CONST_REF(insn->op1) && (ctx->use_lists[insn->op1].count == 1 || all_usages_are_fusable(ctx, insn->op1))) {
+				ir_insn *op1_insn = &ctx->ir_base[insn->op1];
+
+				if (op1_insn->op >= IR_EQ && op1_insn->op <= IR_UNORDERED) {
+					if (IR_IS_TYPE_INT(ctx->ir_base[op1_insn->op1].type)) {
+						if (ctx->use_lists[insn->op1].count == 1) {
+							ir_match_fuse_load_cmp_int(ctx, op1_insn, ref);
+						}
+#if IR_X86_I64
+						if (insn->type == IR_I64 || insn->type == IR_U64) {
+							if (ctx->ir_base[op1_insn->op1].type == IR_I64 || ctx->ir_base[op1_insn->op1].type == IR_U64) {
+								ctx->rules[insn->op1] = IR_FUSED | IR_CMP_I64;
+							}
+							return IR_TWO_REGS | IR_COND_I64_CMP_INT;
+						} else if (ctx->ir_base[op1_insn->op1].type == IR_I64 || ctx->ir_base[op1_insn->op1].type == IR_U64) {
+							ctx->rules[insn->op1] = IR_FUSED | IR_CMP_I64;
+							return IR_COND_CMP_I64;
+						}
+#endif
+						ctx->rules[insn->op1] = IR_FUSED | IR_CMP_INT;
+						return IR_COND_CMP_INT;
+					} else {
+						if (ctx->use_lists[insn->op1].count == 1) {
+							ir_match_fuse_load_cmp_fp_br(ctx, op1_insn, ref);
+						}
+						ctx->rules[insn->op1] = IR_FUSED | IR_CMP_FP;
+#if IR_X86_I64
+						if (insn->type == IR_I64 || insn->type == IR_U64) {
+							return IR_TWO_REGS | IR_COND_I64_CMP_FP;
+						}
+#endif
+						return IR_COND_CMP_FP;
+					}
+#if IR_X86_I64
+				} else if (insn->type == IR_I64 || insn->type == IR_U64) {
+					/* pass */
+#endif
+				} else if (op1_insn->op == IR_AND) {
+					/* c = AND(_, _) ... IF(c) => SKIP_TEST ... TEST_AND_BRANCH */
+					ir_match_fuse_load_test_int(ctx, op1_insn, ref);
+					if (sizeof(void*) == 8
+					 && ir_type_size[op1_insn->type] == 8
+					 && IR_IS_CONST_REF(op1_insn->op2)
+					 && !IR_IS_SYM_CONST(ctx->ir_base[op1_insn->op2].op)
+					 && !IR_IS_SIGNED_32BIT(ctx->ir_base[op1_insn->op2].val.i64)
+					 && IR_IS_POWER_OF_TWO(ctx->ir_base[op1_insn->op2].val.u64)) {
+						ctx->rules[insn->op1] = IR_FUSED | IR_TEST_BIT;
+						return IR_COND_TEST_BIT;
+					} else {
+						ctx->rules[insn->op1] = IR_FUSED | IR_TEST_INT;
+						return IR_COND_TEST_INT;
+					}
+				}
+			}
+			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
+				ir_match_fuse_load(ctx, insn->op1, ref);
+			}
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | IR_COND_I64;
+			}
+#endif
+			return IR_COND;
+		case IR_GUARD:
+		case IR_GUARD_NOT:
+			if (!IR_IS_CONST_REF(insn->op2) && (ctx->use_lists[insn->op2].count == 1 || all_usages_are_fusable(ctx, insn->op2))) {
+				op2_insn = &ctx->ir_base[insn->op2];
+				if (op2_insn->op >= IR_EQ && op2_insn->op <= IR_UNORDERED) {
+					if (IR_IS_TYPE_INT(ctx->ir_base[op2_insn->op1].type)) {
+						if (IR_IS_CONST_REF(op2_insn->op2)
+						 && !IR_IS_SYM_CONST(ctx->ir_base[op2_insn->op2].op)
+						 && ctx->ir_base[op2_insn->op2].val.i64 == 0) {
+							if (op2_insn->op1 == insn->op2 - 1 /* previous instruction */
+							 && ir_in_same_block(ctx, op2_insn->op1)
+							 && !ir_match_has_flags_deps(ctx, insn->op2, ref)) {
+								ir_insn *op1_insn = &ctx->ir_base[op2_insn->op1];
+
+								if ((op1_insn->op == IR_OR || op1_insn->op == IR_AND || op1_insn->op == IR_XOR) ||
+										/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
+										((op1_insn->op == IR_ADD || op1_insn->op == IR_SUB) &&
+											(op2_insn->op == IR_EQ || op2_insn->op == IR_NE ||
+												op2_insn->op == IR_LT || op2_insn->op == IR_GE))) {
+									if (ir_op_flags[op1_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
+										if (ctx->use_lists[insn->op2].count == 1) {
+											ir_match_fuse_load_commutative_int(ctx, op1_insn, ref);
+										}
+										ctx->rules[op2_insn->op1] = IR_BINOP_INT | IR_MAY_SWAP;
+									} else {
+										if (ctx->use_lists[insn->op2].count == 1) {
+											ir_match_fuse_load(ctx, op1_insn->op2, ref);
+										}
+										ctx->rules[op2_insn->op1] = IR_BINOP_INT;
+									}
+									/* v = BINOP(_, _); c = CMP(v, 0) ... IF(c) => BINOP; SKIP_CMP ... GUARD_JCC */
+									ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
+									return IR_GUARD_JCC_INT;
+								}
+							} else if (ctx->use_lists[insn->op2].count == 1
+							 && op2_insn->op1 == insn->op2 - 2 /* before previous instruction */
+							 && ir_in_same_block(ctx, op2_insn->op1)
+							 && ctx->use_lists[op2_insn->op1].count == 2
+							 && !ir_match_has_flags_deps(ctx, insn->op2, ref)) {
+								ir_insn *store_insn = &ctx->ir_base[insn->op2 - 1];
+
+								if (store_insn->op == IR_STORE && store_insn->op3 == op2_insn->op1) {
+									ir_insn *op_insn = &ctx->ir_base[op2_insn->op1];
+
+									if ((op_insn->op == IR_OR || op_insn->op == IR_AND || op_insn->op == IR_XOR) ||
+											/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
+											((op_insn->op == IR_ADD || op_insn->op == IR_SUB) &&
+												(op2_insn->op == IR_EQ || op2_insn->op == IR_NE ||
+													op2_insn->op == IR_LT || op2_insn->op == IR_GE))) {
+										if (ctx->ir_base[op_insn->op1].op == IR_LOAD
+										 && ctx->ir_base[op_insn->op1].op2 == store_insn->op2) {
+											if (ir_in_same_block(ctx, op_insn->op1)
+											 && ctx->use_lists[op_insn->op1].count == 2
+											 && store_insn->op1 == op_insn->op1) {
+												/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; GUARD_JCC */
+												ctx->rules[op2_insn->op1] = IR_FUSED | IR_BINOP_INT;
+												ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
+												ir_match_fuse_addr(ctx, store_insn->op2);
+												ctx->rules[insn->op2 - 1] = IR_MEM_BINOP_INT;
+												ctx->rules[insn->op2] = IR_SKIPPED | IR_NOP;
+												return IR_GUARD_JCC_INT;
+											}
+										} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
+										 && ctx->ir_base[op_insn->op2].op == IR_LOAD
+										 && ctx->ir_base[op_insn->op2].op2 == store_insn->op2) {
+											if (ir_in_same_block(ctx, op_insn->op2)
+											 && ctx->use_lists[op_insn->op2].count == 2
+											 && store_insn->op1 == op_insn->op2) {
+												/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; JCC */
+												ir_swap_ops(op_insn);
+												ctx->rules[op2_insn->op1] = IR_FUSED | IR_BINOP_INT;
+												ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
+												ir_match_fuse_addr(ctx, store_insn->op2);
+												ctx->rules[insn->op2 - 1] = IR_MEM_BINOP_INT;
+												ctx->rules[insn->op2] = IR_SKIPPED | IR_NOP;
+												return IR_GUARD_JCC_INT;
+											}
+										}
+									}
+								}
+							}
+						}
+						/* c = CMP(_, _) ... GUARD(c) => SKIP_CMP ... GUARD_CMP */
+						if (ctx->use_lists[insn->op2].count == 1) {
+							ir_match_fuse_load_cmp_int(ctx, op2_insn, ref);
+						}
+#if IR_X86_I64
+						if (ctx->ir_base[op2_insn->op1].type == IR_I64 || ctx->ir_base[op2_insn->op1].type == IR_U64) {
+							if (ctx->use_lists[insn->op2].count == 1) {
+								ir_match_fuse_load_cmp_int(ctx, op2_insn, ref);
+							}
+							ctx->rules[insn->op2] = IR_FUSED | IR_CMP_I64;
+							return IR_GUARD_CMP_I64;
+						}
+#endif
+						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
+						return IR_GUARD_CMP_INT;
+					} else {
+						/* c = CMP(_, _) ... GUARD(c) => SKIP_CMP ... GUARD_CMP */
+						if (ctx->use_lists[insn->op2].count == 1) {
+							ir_match_fuse_load_cmp_fp_br(ctx, op2_insn, ref);
+						}
+						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_FP;
+						return IR_GUARD_CMP_FP;
+					}
+				} else if (op2_insn->op == IR_OVERFLOW && ir_in_same_block(ctx, insn->op2)) {
+					/* c = OVERFLOW(_) ... GUARD(c) => SKIP_OVERFLOW ... GUARD_OVERFLOW */
+					ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_OVERFLOW;
+					return IR_GUARD_OVERFLOW;
+#if IR_X86_I64
+				} else if (op2_insn->type == IR_I64 || op2_insn->type == IR_U64) {
+					/* pass */
+#endif
+				} else if (op2_insn->op == IR_AND) { // TODO: OR, XOR. etc
+					/* c = AND(_, _) ... GUARD(c) => SKIP_TEST ... GUARD_TEST */
+					ir_match_fuse_load_test_int(ctx, op2_insn, ref);
+					if (sizeof(void*) == 8
+					 && ir_type_size[op2_insn->type] == 8
+					 && IR_IS_CONST_REF(op2_insn->op2)
+					 && !IR_IS_SYM_CONST(ctx->ir_base[op2_insn->op2].op)
+					 && !IR_IS_SIGNED_32BIT(ctx->ir_base[op2_insn->op2].val.i64)
+					 && IR_IS_POWER_OF_TWO(ctx->ir_base[op2_insn->op2].val.u64)) {
+						ctx->rules[insn->op2] = IR_FUSED | IR_TEST_BIT;
+						return IR_GUARD_TEST_BIT;
+					} else {
+						ctx->rules[insn->op2] = IR_FUSED | IR_TEST_INT;
+						return IR_GUARD_TEST_INT;
+					}
+				}
+			}
+			ir_match_fuse_load(ctx, insn->op2, ref);
+#if IR_X86_I64
+			if (ctx->ir_base[insn->op2].type == IR_I64 || ctx->ir_base[insn->op2].type == IR_U64) {
+				return IR_GUARD_I64;
+			}
+#endif
+			return insn->op;
+		case IR_INT2FP:
+#if IR_X86_I64
+			if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				return IR_INT2FP_I64;
+			}
+#endif
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op1].type)) {
+				return IR_VECTOR_INT2FP;
+			}
+#endif
+			if (ir_type_size[ctx->ir_base[insn->op1].type] > (IR_IS_TYPE_SIGNED(ctx->ir_base[insn->op1].type) ? 2 : 4)) {
+				ir_match_fuse_load(ctx, insn->op1, ref);
+			}
+			return insn->op;
+		case IR_FP2INT:
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | IR_FP2INT_I64;
+			}
+#endif
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_FP2INT;
+			}
+#endif
+			if (IR_IS_TYPE_SIGNED(insn->type) || ir_type_size[insn->type] != sizeof(void*)) {
+				ir_match_fuse_load(ctx, insn->op1, ref);
+			}
+			return insn->op;
+		case IR_SEXT:
+		case IR_ZEXT:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (ir_may_fuse_load_vector(ctx, insn->type)) {
+					ir_match_fuse_load(ctx, insn->op1, ref);
+				}
+				return IR_VECTOR_EXT;
+			}
+#endif
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				if (insn->op == IR_SEXT) {
+					return IR_TWO_REGS | IR_SEXT_I64;
+				} else {
+					return IR_TWO_REGS | IR_ZEXT_I64;
+				}
+			}
+#endif
+			IR_FALLTHROUGH;
+		case IR_FP2FP:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (ir_may_fuse_load_vector(ctx, insn->type)) {
+					ir_match_fuse_load(ctx, insn->op1, ref);
+				}
+				return IR_VECTOR_FP2FP;
+			}
+#endif
+			ir_match_fuse_load(ctx, insn->op1, ref);
+			return insn->op;
+		case IR_TRUNC:
+#if IR_SIMD
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				return IR_VECTOR_TRUNC;
+			}
+#endif
+			ir_match_fuse_load(ctx, insn->op1, ref);
+#if IR_X86_I64
+			if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				return insn->op;
+			}
+#endif
+#ifdef IR_TARGET_X86
+			if (ir_type_size[insn->type] == 1) {
+				/* disable register reuse, becuse in 32-bit mode some 32-bit registers (%ebp) don't have 8-bit part */
+				return insn->op;
+			}
+#endif
+			return insn->op | IR_MAY_REUSE;
+		case IR_PROTO:
+			ir_match_fuse_load(ctx, insn->op1, ref);
+			return insn->op | IR_MAY_REUSE;
+		case IR_BITCAST:
+			ir_match_fuse_load(ctx, insn->op1, ref);
+#if IR_X86_I64
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				if (IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
+					return IR_TWO_REGS | IR_OP_I64 | IR_MAY_REUSE;
+				} else {
+					return IR_TWO_REGS | IR_BITCAST_I64;
+				}
+			} else if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				return IR_BITCAST_I64;
+			}
+#endif
+			if (IR_IS_TYPE_INT(insn->type) == IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
+				return insn->op | IR_MAY_REUSE;
+			} else {
+				return insn->op;
+			}
+		case IR_CTLZ:
+		case IR_CTTZ:
+			ir_match_fuse_load(ctx, insn->op1, ref);
+#if IR_X86_I64
+			if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				IR_ASSERT(IR_IS_TYPE_INT(insn->type) && ir_type_size[insn->type] <= sizeof(void*));
+				return IR_BIT_COUNT_I64;
+			}
+#endif
+			return IR_BIT_COUNT;
+		case IR_CTPOP:
+			ir_match_fuse_load(ctx, insn->op1, ref);
+#if IR_X86_I64
+			if (ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64) {
+				IR_ASSERT(IR_IS_TYPE_INT(insn->type) && ir_type_size[insn->type] <= sizeof(void*));
+				return (ctx->mflags & IR_X86_BMI1) ? IR_BIT_COUNT_I64 : IR_BIT_COUNT_HELPER_I64;
+			}
+#endif
+			return (ctx->mflags & IR_X86_BMI1) ? IR_BIT_COUNT : IR_CTPOP;
+		case IR_VA_START:
+			ctx->flags2 |= IR_HAS_VA_START;
+			if ((ctx->ir_base[insn->op2].op == IR_ALLOCA) || (ctx->ir_base[insn->op2].op == IR_VADDR)) {
+				ir_use_list *use_list = &ctx->use_lists[insn->op2];
+				ir_ref *p, n = use_list->count;
+				for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
+					ir_insn *use_insn = &ctx->ir_base[*p];
+					if (use_insn->op == IR_VA_START || use_insn->op == IR_VA_END) {
+					} else if (use_insn->op == IR_VA_COPY) {
+						if (use_insn->op3 == insn->op2) {
+							ctx->flags2 |= IR_HAS_VA_COPY;
+						}
+					} else if (use_insn->op == IR_VA_ARG) {
+						if (use_insn->op2 == insn->op2) {
+							if (IR_IS_TYPE_INT(use_insn->type)) {
+								ctx->flags2 |= IR_HAS_VA_ARG_GP;
+							} else {
+								IR_ASSERT(IR_IS_TYPE_FP(use_insn->type));
+								ctx->flags2 |= IR_HAS_VA_ARG_FP;
+							}
+						}
+					} else if (*p > ref) {
+						/* diriect va_list access */
+						ctx->flags2 |= IR_HAS_VA_ARG_GP|IR_HAS_VA_ARG_FP;
+					}
+				}
+			} else {
+				/* va_list may escape */
+				ctx->flags2 |= IR_HAS_VA_ARG_GP|IR_HAS_VA_ARG_FP;
+			}
+			return IR_VA_START;
+		case IR_VA_END:
+			return IR_SKIPPED | IR_NOP;
+		case IR_VADDR:
+			if (ctx->use_lists[ref].count > 0) {
+				ir_use_list *use_list = &ctx->use_lists[ref];
+				ir_ref *p, n = use_list->count;
+
+				for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
+					if (ctx->ir_base[*p].op != IR_VA_END) {
+						return IR_STATIC_ALLOCA;
+					}
+				}
+			}
+			return IR_SKIPPED | IR_NOP;
+		case IR_ARGVAL:
+			return IR_FUSED | IR_ARGVAL;
+		case IR_NOP:
+			return IR_SKIPPED | IR_NOP;
+		case IR_ASM:
+		case IR_ASM_OUT:
+		case IR_ASM_GOTO:
+			fprintf(stderr, "ERROR: IR_ASM is not implemented yet\n");
+			exit(1);
+			return IR_SKIPPED | IR_NOP;
+#if IR_SIMD
+		case IR_SHUFFLE:
+			return ir_match_shuffle(ctx, insn);
+#endif
+#if IR_X86_I64
+		case IR_PHI:
+		case IR_VLOAD:
+		case IR_VLOAD_v:
+		case IR_VA_ARG:
+		case IR_EXTRACT:
+			if (insn->type == IR_I64 || insn->type == IR_U64) {
+				return IR_TWO_REGS | insn->op;
+			}
+			return insn->op;
+#endif
+		default:
+			break;
+	}
+
+	return insn->op;
+}
+
+static void ir_match_insn2(ir_ctx *ctx, ir_ref ref, uint32_t rule)
+{
+	if (rule == IR_LEA_IB) {
+		if (!ir_match_try_revert_lea_to_add(ctx, ref) && !(ctx->flags & IR_OPT_CODEGEN)) {
+			/* revert to ADD to avoid extra register spill load with -O0 */
+			ctx->rules[ref] = IR_BINOP_INT | IR_MAY_SWAP;
+		}
+	}
+}
+
+/* code generation */
+static int32_t ir_ref_spill_slot_offset(ir_ctx *ctx, ir_ref ref, ir_reg *reg)
+{
+	int32_t offset;
+
+	IR_ASSERT(ref >= 0 && ctx->vregs[ref] && ctx->live_intervals[ctx->vregs[ref]]);
+	offset = ctx->live_intervals[ctx->vregs[ref]]->stack_spill_pos;
+	IR_ASSERT(offset != -1);
+	if (ctx->live_intervals[ctx->vregs[ref]]->flags & IR_LIVE_INTERVAL_SPILL_SPECIAL) {
+		IR_ASSERT(ctx->spill_base != IR_REG_NONE);
+		*reg = ctx->spill_base;
+		return offset;
+	}
+	*reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	return IR_SPILL_POS_TO_OFFSET(offset);
+}
+
+static ir_mem ir_vreg_spill_slot(ir_ctx *ctx, ir_ref v)
+{
+	int32_t offset;
+	ir_reg base;
+
+	IR_ASSERT(v > 0 && v <= ctx->vregs_count && ctx->live_intervals[v]);
+	offset = ctx->live_intervals[v]->stack_spill_pos;
+	IR_ASSERT(offset != -1);
+	if (ctx->live_intervals[v]->flags & IR_LIVE_INTERVAL_SPILL_SPECIAL) {
+		IR_ASSERT(ctx->spill_base != IR_REG_NONE);
+		return IR_MEM_BO(ctx->spill_base, offset);
+	}
+	base = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	offset = IR_SPILL_POS_TO_OFFSET(offset);
+	return IR_MEM_BO(base, offset);
+}
+
+static ir_mem ir_ref_spill_slot(ir_ctx *ctx, ir_ref ref)
+{
+	IR_ASSERT(!IR_IS_CONST_REF(ref));
+	return ir_vreg_spill_slot(ctx, ctx->vregs[ref]);
+}
+
+static bool ir_is_same_spill_slot(ir_ctx *ctx, ir_ref ref, ir_mem mem)
+{
+	ir_mem m = ir_ref_spill_slot(ctx, ref);
+	return IR_MEM_VAL(m) == IR_MEM_VAL(mem);
+}
+
+static ir_mem ir_var_spill_slot(ir_ctx *ctx, ir_ref ref)
+{
+	ir_insn *var_insn = &ctx->ir_base[ref];
+	ir_reg reg;
+
+	IR_ASSERT(var_insn->op == IR_VAR);
+	reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	return IR_MEM_BO(reg, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
+}
+
+static bool ir_may_avoid_spill_load(ir_ctx *ctx, ir_ref ref, ir_ref use)
+{
+	ir_live_interval *ival;
+
+	IR_ASSERT(ctx->vregs[ref] && ctx->live_intervals[ctx->vregs[ref]]);
+	ival = ctx->live_intervals[ctx->vregs[ref]];
+	while (ival) {
+		ir_use_pos *use_pos = ival->use_pos;
+		while (use_pos) {
+			if (IR_LIVE_POS_TO_REF(use_pos->pos) == use) {
+				return !use_pos->next || use_pos->next->op_num == 0;
+			}
+			use_pos = use_pos->next;
+		}
+		ival = ival->next;
+	}
+	return 0;
+}
+
+static void ir_emit_mov_imm_int(ir_ctx *ctx, ir_type type, ir_reg reg, int64_t val)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	if (ir_type_size[type] == 8) {
+		IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+		if (IR_IS_UNSIGNED_32BIT(val)) {
+			|	mov Rd(reg), (uint32_t)val // zero extended load
+		} else if (IR_IS_SIGNED_32BIT(val)) {
+			|	mov Rq(reg), (int32_t)val // sign extended load
+		} else if (type == IR_ADDR && IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, (intptr_t)val)) {
+			|	lea Ra(reg), [&val]
+		} else {
+			|	mov64 Ra(reg), val
+		}
+|.endif
+	} else {
+		|	ASM_REG_IMM_OP mov, type, reg, (int32_t)val // sign extended load
+	}
+}
+
+static void ir_emit_load_imm_int(ir_ctx *ctx, ir_type type, ir_reg reg, int64_t val)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	IR_ASSERT(IR_IS_TYPE_INT(type));
+	if (val == 0) {
+		|	ASM_REG_REG_OP xor, type, reg, reg
+	} else {
+		ir_emit_mov_imm_int(ctx, type, reg, val);
+	}
+}
+
+static void ir_emit_load_mem_int(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem mem)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	|	ASM_REG_MEM_OP mov, type, reg, mem
+}
+
+#if IR_SIMD
+static bool ir_is_zero_vector(const void * p, uint32_t size, uint32_t width)
+{
+	while (width > 0) {
+		if (*(uint8_t*)p != 0) return 0;
+		p = (uint8_t*)p + 1;
+		width--;
+	}
+	return 1;
+}
+#endif
+
+static void ir_emit_load_imm_fp(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref src)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *insn = &ctx->ir_base[src];
+	int label;
+
+	if (type == IR_FLOAT && insn->val.u32 == 0) {
+		if (ctx->mflags & IR_X86_AVX) {
+			|	vxorps xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+		} else {
+			|	xorps xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+		}
+	} else if (type == IR_DOUBLE && insn->val.u64 == 0) {
+		if (ctx->mflags & IR_X86_AVX) {
+			|	vxorpd xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+		} else {
+			|	xorpd xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+		}
+#if IR_SIMD
+	} else if (IR_IS_TYPE_VECTOR(type)) {
+		ir_type element_type = IR_VECTOR_BASE_TYPE(type);
+		uint32_t width = IR_VECTOR_SIZE(type);
+		void *p = ir_long_const_ptr(ctx, src);
+
+		if (ir_is_zero_vector(p, ir_type_size[element_type], width)) {
+			if (ctx->mflags & IR_X86_AVX) {
+				if (width <= 16) {
+					|	vpxor xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(width == 32);
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpxor ymm(reg-IR_REG_FP_FIRST), ymm(reg-IR_REG_FP_FIRST), ymm(reg-IR_REG_FP_FIRST)
+					} else {
+						|	vxorps ymm(reg-IR_REG_FP_FIRST), ymm(reg-IR_REG_FP_FIRST), ymm(reg-IR_REG_FP_FIRST)
+					}
+				}
+			} else {
+				IR_ASSERT(width <= 16);
+				|	pxor xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+			}
+		} else {
+			label = ir_get_const_label(ctx, src);
+
+			if (ctx->mflags & IR_X86_AVX) {
+				if (width <= 16) {
+					if (IR_IS_TYPE_INT(element_type)) {
+						if (width == 16) {
+							|	vmovdqa xmm(reg-IR_REG_FP_FIRST), [=>label]
+						} else if (width == 8) {
+							|	vmovq xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+						} else if (width == 4) {
+							|	vmovd xmm(reg-IR_REG_FP_FIRST), dword [=>label]
+						} else {
+							// TODO: 4-byte read may be unsafe ???
+							// IR_ASSERT(width >= 4 && width <= 16);
+							IR_ASSERT(width <= 4);
+							|	vmovd xmm(reg-IR_REG_FP_FIRST), dword [=>label]
+						}
+					} else if (element_type == IR_DOUBLE) {
+						if (width == 16) {
+							|	vmovapd xmm(reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(width == 8);
+							|	vmovlpd xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+						}
+					} else {
+						IR_ASSERT(element_type == IR_FLOAT);
+						if (width == 16) {
+							|	vmovaps xmm(reg-IR_REG_FP_FIRST), [=>label]
+						} else if (width == 8) {
+							|	vmovlps xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+						} else {
+							IR_ASSERT(width == 4);
+						|	vmovss xmm(reg-IR_REG_FP_FIRST), dword [=>label]
+						}
+					}
+				} else {
+					IR_ASSERT(width == 32);
+					if (IR_IS_TYPE_INT(element_type)) {
+						|	vmovdqa ymm(reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_DOUBLE) {
+						|	vmovapd ymm(reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						IR_ASSERT(element_type == IR_FLOAT);
+						|	vmovaps ymm(reg-IR_REG_FP_FIRST), [=>label]
+					}
+				}
+			} else {
+				IR_ASSERT(width <= 16);
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						|	movdqa xmm(reg-IR_REG_FP_FIRST), [=>label]
+					} else if (width == 8) {
+|.if X64
+						|	movq xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+|.else
+						|	movlpd xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+|.endif
+					} else if (width == 4) {
+						|	movd xmm(reg-IR_REG_FP_FIRST), dword [=>label]
+					} else {
+						// TODO: 4-byte read may be unsafe ???
+						// IR_ASSERT(width >= 4 && width <= 16);
+						IR_ASSERT(width <= 4);
+						|	movd xmm(reg-IR_REG_FP_FIRST), dword [=>label]
+					}
+				} else if (element_type == IR_DOUBLE) {
+					if (width == 16) {
+						|	movapd xmm(reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						IR_ASSERT(width == 8);
+						|	movlpd xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+					}
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					if (width == 16) {
+						|	movaps xmm(reg-IR_REG_FP_FIRST), [=>label]
+					} else if (width == 8) {
+						|	movlps xmm(reg-IR_REG_FP_FIRST), qword [=>label]
+					} else {
+						IR_ASSERT(width == 4);
+						|	movss xmm(reg-IR_REG_FP_FIRST), dword [=>label]
+					}
+				}
+			}
+		}
+#endif
+	} else {
+		label = ir_get_const_label(ctx, src);
+		|	ASM_FP_REG_TXT_OP movs, type, reg, [=>label]
+	}
+}
+
+static void ir_emit_load_mem_fp(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem mem)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+#if IR_SIMD
+	if (IR_IS_TYPE_VECTOR(type)) {
+		ir_type element_type = IR_VECTOR_BASE_TYPE(type);
+		uint32_t width = IR_VECTOR_SIZE(type);
+
+		if (ctx->mflags & IR_X86_AVX) {
+			if (width <= 16) {
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						|	ASM_TXT_TMEM_OP vmovdqa, xmm(reg-IR_REG_FP_FIRST), oword, mem
+					} else if (width == 8) {
+						|	ASM_TXT_TMEM_OP vmovq, xmm(reg-IR_REG_FP_FIRST), qword, mem
+					} else if (width == 4) {
+						|	ASM_TXT_TMEM_OP vmovd, xmm(reg-IR_REG_FP_FIRST), dword, mem
+					} else {
+						// TODO: 4-byte read may be unsafe ???
+						// IR_ASSERT(width >= 4 && width <= 16);
+						IR_ASSERT(width <= 4);
+						|	ASM_TXT_TMEM_OP vmovd, xmm(reg-IR_REG_FP_FIRST), dword, mem
+					}
+				} else if (element_type == IR_DOUBLE) {
+					if (width == 16) {
+						|	ASM_TXT_TMEM_OP vmovapd, xmm(reg-IR_REG_FP_FIRST), oword, mem
+					} else {
+						IR_ASSERT(width == 8);
+						|	ASM_TXT_TXT_TMEM_OP vmovlpd, xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), qword, mem
+					}
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					if (width == 16) {
+						|	ASM_TXT_TMEM_OP vmovaps, xmm(reg-IR_REG_FP_FIRST), oword, mem
+					} else if (width == 8) {
+						|	ASM_TXT_TXT_TMEM_OP vmovlps, xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), qword, mem
+					} else {
+						IR_ASSERT(width == 4);
+						|	ASM_TXT_TMEM_OP vmovss, xmm(reg-IR_REG_FP_FIRST), dword, mem
+					}
+				}
+			} else {
+				IR_ASSERT(width == 32);
+				// TODO: Use 32-byte stack alihnment and "aligned" moves ???
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_TXT_TMEM_OP vmovdqu, ymm(reg-IR_REG_FP_FIRST), yword, mem
+				} else if (element_type == IR_DOUBLE) {
+					|	ASM_TXT_TMEM_OP vmovupd, ymm(reg-IR_REG_FP_FIRST), yword, mem
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					|	ASM_TXT_TMEM_OP vmovups, ymm(reg-IR_REG_FP_FIRST), yword, mem
+				}
+			}
+		} else {
+			IR_ASSERT(width <= 16);
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (width == 16) {
+					|	ASM_TXT_TMEM_OP movdqa, xmm(reg-IR_REG_FP_FIRST), oword, mem
+				} else if (width == 8) {
+|.if X64
+					|	ASM_TXT_TMEM_OP movq, xmm(reg-IR_REG_FP_FIRST), qword, mem
+|.else
+					|	ASM_TXT_TMEM_OP movlpd, xmm(reg-IR_REG_FP_FIRST), qword, mem
+|.endif
+				} else if (width == 4) {
+					|	ASM_TXT_TMEM_OP movd, xmm(reg-IR_REG_FP_FIRST), dword, mem
+				} else {
+					// TODO: 4-byte read may be unsafe ???
+					// IR_ASSERT(width >= 4 && width <= 16);
+					IR_ASSERT(width <= 4);
+					|	ASM_TXT_TMEM_OP movd, xmm(reg-IR_REG_FP_FIRST), dword, mem
+				}
+			} else if (element_type == IR_DOUBLE) {
+				if (width == 16) {
+					|	ASM_TXT_TMEM_OP movapd, xmm(reg-IR_REG_FP_FIRST), oword, mem
+				} else {
+					IR_ASSERT(width == 8);
+					|	ASM_TXT_TMEM_OP movlpd, xmm(reg-IR_REG_FP_FIRST), qword, mem
+				}
+			} else {
+				IR_ASSERT(element_type == IR_FLOAT);
+				if (width == 16) {
+					|	ASM_TXT_TMEM_OP movaps, xmm(reg-IR_REG_FP_FIRST), oword, mem
+				} else if (width == 8) {
+					|	ASM_TXT_TMEM_OP movlps, xmm(reg-IR_REG_FP_FIRST), qword, mem
+				} else {
+					IR_ASSERT(width == 4);
+					|	ASM_TXT_TMEM_OP movss, xmm(reg-IR_REG_FP_FIRST), dword, mem
+				}
+			}
+		}
+		return;
+	}
+#endif
+	|	ASM_FP_REG_MEM_OP movs, type, reg, mem
+}
+
+static void ir_emit_load_mem(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem mem)
+{
+	if (IR_IS_TYPE_INT(type)) {
+		ir_emit_load_mem_int(ctx, type, reg, mem);
+	} else {
+		ir_emit_load_mem_fp(ctx, type, reg, mem);
+	}
+}
+
+static int32_t ir_local_offset(ir_ctx *ctx, ir_insn *insn)
+{
+	if (insn->op != IR_PARAM) {
+		IR_ASSERT(insn->op == IR_VAR || insn->op == IR_ALLOCA || insn->op == IR_VADDR);
+		return IR_SPILL_POS_TO_OFFSET(insn->op3);
+	} else {
+		IR_ASSERT(ctx->value_params && ctx->value_params[insn->op3 - 1].align);
+		return IR_SPILL_POS_TO_OFFSET(ctx->value_params[insn->op3 - 1].offset);
+	}
+}
+
+static void ir_load_local_addr(ir_ctx *ctx, ir_reg reg, ir_ref src)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg base = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	ir_insn *var_insn;
+	int32_t offset;
+
+	IR_ASSERT(ir_rule(ctx, src) == IR_STATIC_ALLOCA);
+	var_insn = &ctx->ir_base[src];
+	if (var_insn->op == IR_VADDR) {
+		var_insn = &ctx->ir_base[var_insn->op1];
+	}
+	offset = ir_local_offset(ctx, var_insn);
+	if (offset == 0) {
+		| mov Ra(reg), Ra(base)
+	} else {
+		| lea Ra(reg), [Ra(base)+offset]
+	}
+}
+
+static void ir_resolve_label_syms(ir_ctx *ctx)
+{
+	uint32_t b;
+	ir_block *bb;
+
+	for (b = 1, bb = &ctx->cfg_blocks[b]; b <= ctx->cfg_blocks_count; bb++, b++) {
+		ir_insn *insn = &ctx->ir_base[bb->start];
+
+		if (insn->op == IR_BEGIN && insn->op2) {
+			IR_ASSERT(ctx->ir_base[insn->op2].op == IR_LABEL);
+			ctx->ir_base[insn->op2].val.u32_hi = b;
+		}
+	}
+}
+
+static void ir_emit_load_label_addr(ir_ctx *ctx, ir_reg reg, ir_insn *label)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	if (!data->resolved_label_syms) {
+		data->resolved_label_syms = 1;
+		ir_resolve_label_syms(ctx);
+	}
+
+	IR_ASSERT(label->op == IR_LABEL);
+	int b = label->val.u32_hi;
+
+	b = ir_skip_empty_target_blocks(ctx, b);
+	|	lea Ra(reg), aword [=>b]
+}
+
+static void ir_emit_load(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref src)
+{
+	if (IR_IS_CONST_REF(src)) {
+		if (IR_IS_TYPE_INT(type)) {
+			ir_insn *insn = &ctx->ir_base[src];
+
+			if (insn->op == IR_SYM || insn->op == IR_FUNC) {
+				void *addr = ir_sym_val(ctx, insn);
+				ir_emit_load_imm_int(ctx, type, reg, (intptr_t)addr);
+			} else if (insn->op == IR_STR) {
+				ir_backend_data *data = ctx->data;
+				dasm_State **Dst = &data->dasm_state;
+				int label = ir_get_const_label(ctx, src);
+
+				|	lea Ra(reg), aword [=>label]
+			} else if (insn->op == IR_LABEL) {
+				ir_emit_load_label_addr(ctx, reg, insn);
+			} else {
+				ir_emit_load_imm_int(ctx, type, reg, insn->val.i64);
+			}
+		} else {
+			ir_emit_load_imm_fp(ctx, type, reg, src);
+		}
+	} else if (ctx->vregs[src]) {
+		ir_emit_load_mem(ctx, type, reg, ir_ref_spill_slot(ctx, src));
+	} else {
+		ir_load_local_addr(ctx, reg, src);
+	}
+}
+
+static void ir_emit_store_mem_int(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg reg)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	|	ASM_MEM_REG_OP mov, type, mem, reg
+}
+
+static void ir_emit_store_mem_fp(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg reg)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+#if IR_SIMD
+	if (IR_IS_TYPE_VECTOR(type)) {
+		ir_type element_type = IR_VECTOR_BASE_TYPE(type);
+		uint32_t width = IR_VECTOR_SIZE(type);
+
+		if (ctx->mflags & IR_X86_AVX) {
+			if (width <= 16) {
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						|	ASM_TMEM_TXT_OP vmovdqa, oword, mem, xmm(reg-IR_REG_FP_FIRST)
+					} else if (width == 8) {
+						|	ASM_TMEM_TXT_OP vmovq, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+					} else {
+						// TODO: 4-byte write may be unsafe ???
+						// IR_ASSERT(width >= 4 && width <= 16);
+						IR_ASSERT(width <= 4);
+						|	ASM_TMEM_TXT_OP vmovd, dword, mem, xmm(reg-IR_REG_FP_FIRST)
+					}
+				} else if (element_type == IR_DOUBLE) {
+					if (width == 16) {
+						|	ASM_TMEM_TXT_OP vmovapd, oword, mem, xmm(reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(width == 8);
+						|	ASM_TMEM_TXT_OP vmovlpd, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					if (width == 16) {
+						|	ASM_TMEM_TXT_OP vmovaps, oword, mem, xmm(reg-IR_REG_FP_FIRST)
+					} else if (width == 8) {
+						|	ASM_TMEM_TXT_OP vmovlps, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(width == 4);
+						|	ASM_TMEM_TXT_OP vmovss, dword, mem, xmm(reg-IR_REG_FP_FIRST)
+					}
+				}
+			} else {
+				IR_ASSERT(width == 32);
+				// TODO: Use 32-byte stack alignment and "aligned" moves ???
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_TMEM_TXT_OP vmovdqu, yword, mem, ymm(reg-IR_REG_FP_FIRST)
+				} else if (element_type == IR_DOUBLE) {
+					|	ASM_TMEM_TXT_OP vmovupd, yword, mem, ymm(reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					|	ASM_TMEM_TXT_OP vmovups, yword, mem, ymm(reg-IR_REG_FP_FIRST)
+				}
+			}
+		} else {
+			IR_ASSERT(width <= 16);
+			if (IR_IS_TYPE_INT(element_type)) {
+				if (width == 16) {
+					|	ASM_TMEM_TXT_OP movdqa, oword, mem, xmm(reg-IR_REG_FP_FIRST)
+				} else if (width == 8) {
+|.if X64
+					|	ASM_TMEM_TXT_OP movq, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+|.else
+					|	ASM_TMEM_TXT_OP movlpd, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+|.endif
+				} else {
+					// TODO: 4-byte write may be unsafe ???
+					// IR_ASSERT(width >= 4 && width <= 16);
+					IR_ASSERT(width <= 4);
+					|	ASM_TMEM_TXT_OP movd, dword, mem, xmm(reg-IR_REG_FP_FIRST)
+				}
+			} else if (element_type == IR_DOUBLE) {
+				if (width == 16) {
+					|	ASM_TMEM_TXT_OP movapd, oword, mem, xmm(reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(width == 8);
+					|	ASM_TMEM_TXT_OP movlpd, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(element_type == IR_FLOAT);
+				if (width == 16) {
+					|	ASM_TMEM_TXT_OP movaps, oword, mem, xmm(reg-IR_REG_FP_FIRST)
+				} else if (width == 8) {
+					|	ASM_TMEM_TXT_OP movlps, qword, mem, xmm(reg-IR_REG_FP_FIRST)
+				} else {
+					|	ASM_TMEM_TXT_OP movss, dword, mem, xmm(reg-IR_REG_FP_FIRST)
+					IR_ASSERT(width == 4);
+				}
+			}
+		}
+		return;
+	}
+#endif
+	|	ASM_FP_MEM_REG_OP movs, type, mem, reg
+}
+
+static void ir_emit_store_mem_imm(ir_ctx *ctx, ir_type type, ir_mem mem, int32_t imm)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	|	ASM_MEM_IMM_OP mov, type, mem, imm
+}
+
+#if IR_X86_I64
+|.if not X64
+static void ir_emit_load_i64_lo(ir_ctx *ctx, ir_reg reg, ir_ref src)
+{
+	if (IR_IS_CONST_REF(src)) {
+		ir_insn *insn = &ctx->ir_base[src];
+
+		IR_ASSERT(IR_IS_TYPE_INT(insn->type) && !IR_IS_SYM_CONST(insn->op));
+		ir_emit_load_imm_int(ctx, IR_U32, reg, insn->val.u32);
+	} else if (ctx->vregs[src]) {
+		ir_emit_load_mem_int(ctx, IR_U32, reg, ir_ref_spill_slot(ctx, src));
+	} else {
+		IR_ASSERT(0);
+	}
+}
+
+static void ir_emit_load_i64_hi(ir_ctx *ctx, ir_reg reg, ir_ref src)
+{
+	if (IR_IS_CONST_REF(src)) {
+		ir_insn *insn = &ctx->ir_base[src];
+
+		IR_ASSERT(IR_IS_TYPE_INT(insn->type) && !IR_IS_SYM_CONST(insn->op));
+		ir_emit_load_imm_int(ctx, IR_U32, reg, insn->val.u32_hi);
+	} else if (ctx->vregs[src]) {
+		ir_mem mem = IR_MEM_I64_HI(ir_ref_spill_slot(ctx, src));
+		ir_emit_load_mem_int(ctx, IR_U32, reg, mem);
+	} else {
+		IR_ASSERT(0);
+	}
+}
+
+static void ir_emit_store_i64_lo(ir_ctx *ctx, ir_ref dst, ir_reg reg)
+{
+	if (ctx->vregs[dst]) {
+		ir_emit_store_mem_int(ctx, IR_U32, ir_ref_spill_slot(ctx, dst), reg);
+	} else {
+		IR_ASSERT(0);
+	}
+}
+
+static void ir_emit_store_i64_hi(ir_ctx *ctx, ir_ref dst, ir_reg reg)
+{
+	if (ctx->vregs[dst]) {
+		ir_mem mem = IR_MEM_I64_HI(ir_ref_spill_slot(ctx, dst));
+		ir_emit_store_mem_int(ctx, IR_U32, mem, reg);
+	} else {
+		IR_ASSERT(0);
+	}
+}
+|.endif
+#endif
+
+static void ir_emit_store_mem_int_const(ir_ctx *ctx, ir_type type, ir_mem mem, ir_ref src, ir_reg tmp_reg, bool is_arg)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *val_insn = &ctx->ir_base[src];
+
+	IR_ASSERT(IR_IS_CONST_REF(src));
+	if (val_insn->op == IR_STR) {
+		int label = ir_get_const_label(ctx, src);
+
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+|.if X64
+		|	lea Ra(tmp_reg), aword [=>label]
+||		ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
+|.else
+		|	ASM_TMEM_TXT_OP mov, aword, mem, =>label
+|.endif
+	} else if (val_insn->op == IR_LABEL) {
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		tmp_reg = IR_REG_NUM(tmp_reg);
+		ir_emit_load_label_addr(ctx, tmp_reg, val_insn);
+		ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
+	} else {
+		int64_t val = val_insn->val.i64;
+
+		if (val_insn->op == IR_FUNC || val_insn->op == IR_SYM) {
+			val = (int64_t)(intptr_t)ir_sym_val(ctx, val_insn);
+		}
+
+		if (ir_type_size[val_insn->type] <= 4 || (sizeof(void*) == 8 && IR_IS_SIGNED_32BIT(val))) {
+			if (is_arg && ir_type_size[type] < 4) {
+				type = IR_U32;
+			}
+			ir_emit_store_mem_imm(ctx, type, mem, (int32_t)val);
+		} else if (sizeof(void*) == 4) {
+			ir_mem mem_hi = IR_MEM_I64_HI(mem);
+
+			ir_emit_store_mem_imm(ctx, IR_U32, mem, (uint32_t)val);
+			ir_emit_store_mem_imm(ctx, IR_U32, mem_hi, (uint32_t)(val >> 32));
+		} else {
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+			tmp_reg = IR_REG_NUM(tmp_reg);
+			ir_emit_load_imm_int(ctx, type, tmp_reg, val);
+			ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
+		}
+	}
+}
+
+static void ir_emit_store_mem_fp_const(ir_ctx *ctx, ir_type type, ir_mem mem, ir_ref src, ir_reg tmp_reg, ir_reg tmp_fp_reg)
+{
+	ir_val *val = &ctx->ir_base[src].val;
+
+	if (type == IR_FLOAT) {
+		ir_emit_store_mem_imm(ctx, IR_U32, mem, val->i32);
+	} else if (type == IR_DOUBLE) {
+		if (sizeof(void*) == 8 && val->i64 == 0) {
+			ir_emit_store_mem_imm(ctx, IR_U64, mem, 0);
+		} else if (sizeof(void*) == 8 && tmp_reg != IR_REG_NONE) {
+			ir_emit_load_imm_int(ctx, IR_U64, tmp_reg, val->i64);
+			ir_emit_store_mem_int(ctx, IR_U64, mem, tmp_reg);
+		} else {
+			tmp_fp_reg = IR_REG_NUM(tmp_fp_reg);
+			ir_emit_load(ctx, type, tmp_fp_reg, src);
+			ir_emit_store_mem_fp(ctx, IR_DOUBLE, mem, tmp_fp_reg);
+		}
+#if IR_SIMD
+	} else if (IR_IS_TYPE_VECTOR(type)) {
+		tmp_fp_reg = IR_REG_NUM(tmp_fp_reg);
+		IR_ASSERT(tmp_fp_reg != IR_REG_NONE);
+		ir_emit_load_imm_fp(ctx, type, tmp_fp_reg, src);
+		ir_emit_store_mem_fp(ctx, type, mem, tmp_fp_reg);
+#endif
+	} else {
+		IR_ASSERT(0);
+	}
+}
+
+static void ir_emit_store_mem(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg reg)
+{
+	if (IR_IS_TYPE_INT(type)) {
+		ir_emit_store_mem_int(ctx, type, mem, reg);
+	} else {
+		ir_emit_store_mem_fp(ctx, type, mem, reg);
+	}
+}
+
+static void ir_emit_store(ir_ctx *ctx, ir_type type, ir_ref dst, ir_reg reg)
+{
+	IR_ASSERT(dst >= 0);
+	ir_emit_store_mem(ctx, type, ir_ref_spill_slot(ctx, dst), reg);
+}
+
+static void ir_emit_mov(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	|	ASM_REG_REG_OP mov, type, dst, src
+}
+
+#define IR_HAVE_SWAP_INT
+
+static void ir_emit_swap(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	|	ASM_REG_REG_OP xchg, type, dst, src
+}
+
+static void ir_emit_mov_ext(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	if (ir_type_size[type] > 2) {
+		|	ASM_REG_REG_OP mov, type, dst, src
+	} else if (ir_type_size[type] == 2) {
+		if (IR_IS_TYPE_SIGNED(type)) {
+			if (dst == IR_REG_RAX && src == IR_REG_RAX) {
+				|	cwde
+			} else {
+				|	movsx Rd(dst), Rw(src)
+			}
+		} else {
+			|	movzx Rd(dst), Rw(src)
+		}
+	} else /* if (ir_type_size[type] == 1) */ {
+		if (IR_IS_TYPE_SIGNED(type)) {
+			|	movsx Rd(dst), Rb(src)
+		} else {
+			|	movzx Rd(dst), Rb(src)
+		}
+	}
+}
+
+static void ir_emit_fp_mov(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+#if IR_SIMD
+	if (IR_IS_TYPE_VECTOR(type)) {
+		ir_type element_type = IR_VECTOR_BASE_TYPE(type);
+		uint32_t width = IR_VECTOR_SIZE(type);
+
+		if (ctx->mflags & IR_X86_AVX) {
+			if (width <= 16) {
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	vmovdqa xmm(dst-IR_REG_FP_FIRST), xmm(src-IR_REG_FP_FIRST)
+				} else if (element_type == IR_DOUBLE) {
+					|	vmovapd xmm(dst-IR_REG_FP_FIRST), xmm(src-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					|	vmovaps xmm(dst-IR_REG_FP_FIRST), xmm(src-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(width == 32);
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	vmovdqa ymm(dst-IR_REG_FP_FIRST), ymm(src-IR_REG_FP_FIRST)
+				} else if (element_type == IR_DOUBLE) {
+					|	vmovapd ymm(dst-IR_REG_FP_FIRST), ymm(src-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(element_type == IR_FLOAT);
+					|	vmovaps ymm(dst-IR_REG_FP_FIRST), ymm(src-IR_REG_FP_FIRST)
+				}
+			}
+		} else {
+			IR_ASSERT(width <= 16);
+			if (IR_IS_TYPE_INT(element_type)) {
+				|	movdqa xmm(dst-IR_REG_FP_FIRST), xmm(src-IR_REG_FP_FIRST)
+			} else if (element_type == IR_DOUBLE) {
+				|	movapd xmm(dst-IR_REG_FP_FIRST), xmm(src-IR_REG_FP_FIRST)
+			} else {
+				IR_ASSERT(element_type == IR_FLOAT);
+				|	movaps xmm(dst-IR_REG_FP_FIRST), xmm(src-IR_REG_FP_FIRST)
+			}
+		}
+		return;
+	}
+#endif
+	|	ASM_FP_REG_REG_OP movap, type, dst, src
+}
+
+static ir_mem ir_fuse_addr_const(ir_ctx *ctx, ir_ref ref)
+{
+	ir_mem mem;
+	ir_insn *addr_insn = &ctx->ir_base[ref];
+
+	IR_ASSERT(IR_IS_CONST_REF(ref));
+	if (IR_IS_SYM_CONST(addr_insn->op)) {
+		void *addr = ir_sym_val(ctx, addr_insn);
+		IR_ASSERT(sizeof(void*) == 4 || IR_IS_SIGNED_32BIT((intptr_t)addr));
+		mem = IR_MEM_O((int32_t)(intptr_t)addr);
+	} else {
+		IR_ASSERT(sizeof(void*) == 4 || IR_IS_SIGNED_32BIT(addr_insn->val.i64));
+		mem = IR_MEM_O(addr_insn->val.i32);
+	}
+	return mem;
+}
+
+static ir_mem ir_fuse_addr(ir_ctx *ctx, ir_ref root, ir_ref ref)
+{
+	uint32_t rule = ctx->rules[ref];
+	ir_insn *insn = &ctx->ir_base[ref];
+	ir_insn *op1_insn, *op2_insn, *offset_insn;
+	ir_ref base_reg_ref, index_reg_ref;
+	ir_reg base_reg = IR_REG_NONE, index_reg;
+	int32_t offset = 0, scale;
+
+	IR_ASSERT(((rule & IR_RULE_MASK) >= IR_LEA_FIRST &&
+			(rule & IR_RULE_MASK) <= IR_LEA_LAST) ||
+		rule == IR_STATIC_ALLOCA);
+	switch (rule & IR_RULE_MASK) {
+		default:
+			IR_ASSERT(0);
+		case IR_LEA_OB:
+			offset_insn = insn;
+			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = ref * sizeof(ir_ref) + 1;
+			}
+			index_reg_ref = IR_UNUSED;
+			scale = 1;
+			break;
+		case IR_LEA_SI:
+			scale = ctx->ir_base[insn->op2].val.i32;
+			index_reg_ref = ref * sizeof(ir_ref) + 1;
+			base_reg_ref = IR_UNUSED;
+			offset_insn = NULL;
+			break;
+		case IR_LEA_SIB:
+			base_reg_ref = index_reg_ref = ref * sizeof(ir_ref) + 1;
+			scale = ctx->ir_base[insn->op2].val.i32 - 1;
+			offset_insn = NULL;
+			break;
+		case IR_LEA_IB:
+			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = ref * sizeof(ir_ref) + 2;
+			} else if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = ref * sizeof(ir_ref) + 1;
+			} else {
+				base_reg_ref = ref * sizeof(ir_ref) + 1;
+				index_reg_ref = ref * sizeof(ir_ref) + 2;
+			}
+			offset_insn = NULL;
+			scale = 1;
+			break;
+		case IR_LEA_OB_I:
+			op1_insn = &ctx->ir_base[insn->op1];
+			offset_insn = op1_insn;
+			scale = 1;
+			if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			} else if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = ref * sizeof(ir_ref) + 2;
+			} else {
+				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+				index_reg_ref = ref * sizeof(ir_ref) + 2;
+			}
+			break;
+		case IR_LEA_I_OB:
+			op2_insn = &ctx->ir_base[insn->op2];
+			offset_insn = op2_insn;
+			scale = 1;
+			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+			} else if (ir_rule(ctx, op2_insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op2_insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = ref * sizeof(ir_ref) + 1;
+			} else {
+				base_reg_ref = ref * sizeof(ir_ref) + 1;
+				index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+			}
+			break;
+		case IR_LEA_SI_O:
+			index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			op1_insn = &ctx->ir_base[insn->op1];
+			scale = ctx->ir_base[op1_insn->op2].val.i32;
+			offset_insn = insn;
+			base_reg_ref = IR_UNUSED;
+			break;
+		case IR_LEA_SIB_O:
+			base_reg_ref = index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			op1_insn = &ctx->ir_base[insn->op1];
+			scale = ctx->ir_base[op1_insn->op2].val.i32 - 1;
+			offset_insn = insn;
+			break;
+		case IR_LEA_IB_O:
+			op1_insn = &ctx->ir_base[insn->op1];
+			offset_insn = insn;
+			scale = 1;
+			if (ir_rule(ctx, op1_insn->op2) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op2]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			} else if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+				index_reg_ref = insn->op1 * sizeof(ir_ref) + 2;
+			} else {
+				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+				index_reg_ref = insn->op1 * sizeof(ir_ref) + 2;
+			}
+			break;
+		case IR_LEA_OB_SI:
+			index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+			op1_insn = &ctx->ir_base[insn->op1];
+			offset_insn = op1_insn;
+			op2_insn = &ctx->ir_base[insn->op2];
+			scale = ctx->ir_base[op2_insn->op2].val.i32;
+			if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			}
+			break;
+		case IR_LEA_SI_OB:
+			index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			op1_insn = &ctx->ir_base[insn->op1];
+			scale = ctx->ir_base[op1_insn->op2].val.i32;
+			op2_insn = &ctx->ir_base[insn->op2];
+			offset_insn = op2_insn;
+			if (ir_rule(ctx, op2_insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op2_insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+			}
+			break;
+		case IR_LEA_B_SI:
+			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = ref * sizeof(ir_ref) + 1;
+			}
+			index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+			op2_insn = &ctx->ir_base[insn->op2];
+			scale = ctx->ir_base[op2_insn->op2].val.i32;
+			offset_insn = NULL;
+			break;
+		case IR_LEA_SI_B:
+			index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = ref * sizeof(ir_ref) + 2;
+			}
+			op1_insn = &ctx->ir_base[insn->op1];
+			scale = ctx->ir_base[op1_insn->op2].val.i32;
+			offset_insn = NULL;
+			break;
+		case IR_LEA_B_SI_O:
+			offset_insn = insn;
+			op1_insn = &ctx->ir_base[insn->op1];
+			if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+			}
+			index_reg_ref = op1_insn->op2 * sizeof(ir_ref) + 1;
+			op2_insn = &ctx->ir_base[op1_insn->op2];
+			scale = ctx->ir_base[op2_insn->op2].val.i32;
+			break;
+		case IR_LEA_SI_B_O:
+			offset_insn = insn;
+			op1_insn = &ctx->ir_base[insn->op1];
+			index_reg_ref = op1_insn->op1 * sizeof(ir_ref) + 1;
+			if (ir_rule(ctx, op1_insn->op2) == IR_STATIC_ALLOCA) {
+				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op2]);
+				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				base_reg_ref = IR_UNUSED;
+			} else {
+				base_reg_ref = insn->op1 * sizeof(ir_ref) + 2;
+			}
+			op1_insn = &ctx->ir_base[op1_insn->op1];
+			scale = ctx->ir_base[op1_insn->op2].val.i32;
+			break;
+		case IR_LEA_SYM_O:
+			op1_insn = &ctx->ir_base[insn->op1];
+			op2_insn = &ctx->ir_base[insn->op2];
+			offset = (intptr_t)ir_sym_val(ctx, op1_insn) + (intptr_t)op2_insn->val.i64;
+			base_reg_ref = index_reg_ref = IR_UNUSED;
+			scale = 1;
+			offset_insn = NULL;
+			break;
+		case IR_LEA_O_SYM:
+			op1_insn = &ctx->ir_base[insn->op1];
+			op2_insn = &ctx->ir_base[insn->op2];
+			offset = (intptr_t)ir_sym_val(ctx, op2_insn) + (intptr_t)op1_insn->val.i64;
+			base_reg_ref = index_reg_ref = IR_UNUSED;
+			scale = 1;
+			offset_insn = NULL;
+			break;
+		case IR_ALLOCA:
+			offset = ir_local_offset(ctx, insn);
+			base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			base_reg_ref = index_reg_ref = IR_UNUSED;
+			scale = 1;
+			offset_insn = NULL;
+			break;
+	}
+
+	if (offset_insn) {
+		ir_insn *addr_insn = &ctx->ir_base[offset_insn->op2];
+
+		if (IR_IS_SYM_CONST(addr_insn->op)) {
+			void *addr = ir_sym_val(ctx, addr_insn);
+			IR_ASSERT(sizeof(void*) != 8 || IR_IS_SIGNED_32BIT((intptr_t)addr));
+			offset += (int32_t)(intptr_t)(addr);
+		} else {
+			if (offset_insn->op == IR_SUB) {
+				offset -= addr_insn->val.i32;
+			} else {
+				offset += addr_insn->val.i32;
+			}
+		}
+	}
+
+	if (base_reg_ref) {
+		if (UNEXPECTED(ctx->rules[base_reg_ref / sizeof(ir_ref)] & IR_FUSED_REG)) {
+			base_reg = ir_get_fused_reg(ctx, root, base_reg_ref);
+		} else {
+			base_reg = ((int8_t*)ctx->regs)[base_reg_ref];
+		}
+		IR_ASSERT(base_reg != IR_REG_NONE);
+		if (IR_REG_SPILLED(base_reg)) {
+			base_reg = IR_REG_NUM(base_reg);
+			ir_emit_load(ctx, insn->type, base_reg, ((ir_ref*)ctx->ir_base)[base_reg_ref]);
+		}
+	}
+
+	index_reg = IR_REG_NONE;
+	if (index_reg_ref) {
+		if (base_reg_ref
+			&& ((ir_ref*)ctx->ir_base)[index_reg_ref]
+				== ((ir_ref*)ctx->ir_base)[base_reg_ref]) {
+			index_reg = base_reg;
+		} else {
+			if (UNEXPECTED(ctx->rules[index_reg_ref / sizeof(ir_ref)] & IR_FUSED_REG)) {
+				index_reg = ir_get_fused_reg(ctx, root, index_reg_ref);
+			} else {
+				index_reg = ((int8_t*)ctx->regs)[index_reg_ref];
+			}
+			IR_ASSERT(index_reg != IR_REG_NONE);
+			if (IR_REG_SPILLED(index_reg)) {
+				index_reg = IR_REG_NUM(index_reg);
+				ir_emit_load(ctx, insn->type, index_reg, ((ir_ref*)ctx->ir_base)[index_reg_ref]);
+			}
+		}
+	}
+
+	return IR_MEM(base_reg, offset, index_reg, scale);
+}
+
+static ir_mem ir_fuse_mem(ir_ctx *ctx, ir_ref root, ir_ref ref, ir_insn *mem_insn, ir_reg reg)
+{
+	if (reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(reg)) {
+			reg = IR_REG_NUM(reg);
+			ir_emit_load(ctx, IR_ADDR, reg, mem_insn->op2);
+		}
+		return IR_MEM_B(reg);
+	} else if (IR_IS_CONST_REF(mem_insn->op2)) {
+		return ir_fuse_addr_const(ctx, mem_insn->op2);
+	} else {
+		return ir_fuse_addr(ctx, root, mem_insn->op2);
+	}
+}
+
+static ir_mem ir_fuse_load(ir_ctx *ctx, ir_ref root, ir_ref ref)
+{
+	ir_insn *load_insn = &ctx->ir_base[ref];
+	ir_reg reg;
+
+	IR_ASSERT(load_insn->op == IR_LOAD || load_insn->op == IR_LOAD_v ||
+		load_insn->op == IR_VLOAD || load_insn->op == IR_VLOAD_v);
+	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
+		reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 2);
+	} else {
+		reg = ctx->regs[ref][2];
+	}
+	return ir_fuse_mem(ctx, root, ref, load_insn, reg);
+}
+
+static int32_t ir_fuse_imm(ir_ctx *ctx, ir_ref ref)
+{
+	ir_insn *val_insn = &ctx->ir_base[ref];
+
+	IR_ASSERT(IR_IS_CONST_REF(ref));
+	if (IR_IS_SYM_CONST(val_insn->op)) {
+		void *addr = ir_sym_val(ctx, val_insn);
+		IR_ASSERT(IR_IS_SIGNED_32BIT((intptr_t)addr));
+		return (int32_t)(intptr_t)addr;
+	} else {
+		IR_ASSERT(ir_type_size[val_insn->type] == 4 || IR_IS_SIGNED_32BIT(val_insn->val.i64));
+		return val_insn->val.i32;
+	}
+}
+
+static void ir_emit_load_ex(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref src, ir_ref root)
+{
+	if (IR_IS_CONST_REF(src)) {
+		if (IR_IS_TYPE_INT(type)) {
+			ir_insn *insn = &ctx->ir_base[src];
+
+			if (insn->op == IR_SYM || insn->op == IR_FUNC) {
+				void *addr = ir_sym_val(ctx, insn);
+				ir_emit_load_imm_int(ctx, type, reg, (intptr_t)addr);
+			} else if (insn->op == IR_STR) {
+				ir_backend_data *data = ctx->data;
+				dasm_State **Dst = &data->dasm_state;
+				int label = ir_get_const_label(ctx, src);
+
+				|	lea Ra(reg), aword [=>label]
+			} else if (insn->op == IR_LABEL) {
+				ir_emit_load_label_addr(ctx, reg, insn);
+			} else {
+				ir_emit_load_imm_int(ctx, type, reg, insn->val.i64);
+			}
+		} else {
+			ir_emit_load_imm_fp(ctx, type, reg, src);
+		}
+	} else if (ir_rule(ctx, src) == IR_STATIC_ALLOCA) {
+		ir_load_local_addr(ctx, reg, src);
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, src) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, root, src);
+		} else {
+			mem = ir_ref_spill_slot(ctx, src);
+		}
+		ir_emit_load_mem(ctx, type, reg, mem);
+	}
+}
+
+static void ir_rodata(ir_ctx *ctx)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	|.rodata
+	if (!data->rodata_label) {
+		int label = data->rodata_label = ctx->cfg_blocks_count + ctx->consts_count + 2;
+		|=>label:
+	}
+}
+
+#ifdef _WIN32
+# ifdef _WIN64
+#  define WIN_CHKSTK_LIMIT (8 * 1024)
+extern size_t __chkstk(size_t);
+# else
+#  define WIN_CHKSTK_LIMIT (4 * 1024)
+#  define __chkstk _chkstk
+extern size_t _chkstk(size_t);
+# endif
+#endif
+
+static void ir_stack_alloca(ir_ctx *ctx, int32_t size)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+#ifdef _WIN32
+# ifdef _WIN64
+	if (size >= WIN_CHKSTK_LIMIT) {
+		void *addr = __chkstk;
+		if (IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+			|	mov Ra(IR_REG_RAX), size
+			|	call aword &addr
+			|	sub Ra(IR_REG_RSP), Ra(IR_REG_RAX)
+		} else {
+			uint32_t addr_hi = (uintptr_t)addr >> 32;
+			uint32_t addr_lo = (uintptr_t)addr & 0xffffffff;
+
+			if (!data->chkstk_addr) {
+				data->chkstk_addr = 1;
+				ir_rodata(ctx);
+				|.align 8
+				|->chkstk_addr:
+				|	.dword addr_lo, addr_hi
+				|.code
+			}
+			|	mov Ra(IR_REG_RAX), size
+			|	call aword [->chkstk_addr]
+			|	sub Ra(IR_REG_RSP), Ra(IR_REG_RAX)
+		}
+	} else
+# else
+	if (size >= WIN_CHKSTK_LIMIT) {
+		void *addr = __chkstk;
+		|	mov Ra(IR_REG_RAX), size
+		|	call aword &addr
+	} else
+# endif
+#endif
+	{
+		|	sub Ra(IR_REG_RSP), size
+	}
+}
+
+static void ir_emit_prologue(ir_ctx *ctx)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	int offset = ctx->stack_frame_size + ctx->call_stack_size;
+
+
+	if (ctx->flags & IR_USE_FRAME_POINTER) {
+		|	push Ra(IR_REG_RBP)
+		|	mov Ra(IR_REG_RBP), Ra(IR_REG_RSP)
+	}
+	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP)) {
+		int i;
+		ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP);
+
+		for (i = IR_REG_GP_FIRST; i <= IR_REG_GP_LAST; i++) {
+			if (IR_REGSET_IN(used_preserved_regs, i)) {
+				offset -= sizeof(void*);
+				|	push Ra(i)
+			}
+		}
+	}
+	if (ctx->stack_frame_size + ctx->call_stack_size) {
+		if (ctx->fixed_stack_red_zone) {
+			IR_ASSERT(ctx->stack_frame_size + ctx->call_stack_size <= ctx->fixed_stack_red_zone);
+		} else if (offset) {
+			ir_stack_alloca(ctx, offset);
+		}
+	}
+	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_FP)) {
+		ir_reg fp;
+		int i;
+		ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_FP);
+
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			fp = IR_REG_FRAME_POINTER;
+			offset -= ctx->stack_frame_size + ctx->call_stack_size;
+		} else {
+			fp = IR_REG_STACK_POINTER;
+		}
+		for (i = IR_REG_FP_FIRST; i <= IR_REG_FP_LAST; i++) {
+			if (IR_REGSET_IN(used_preserved_regs, i)) {
+				offset -= sizeof(void*);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vmovsd qword [Ra(fp)+offset], xmm(i-IR_REG_FP_FIRST)
+				} else {
+					|	movsd qword [Ra(fp)+offset], xmm(i-IR_REG_FP_FIRST)
+				}
+			}
+		}
+	}
+	if ((ctx->flags & IR_VARARG_FUNC) && (ctx->flags2 & IR_HAS_VA_START)) {
+		const ir_call_conv_dsc *cc = data->ra_data.cc;
+
+		if (cc->shadow_store_size) {
+			ir_reg fp;
+			int shadow_store;
+			int offset = 0;
+			int n = 0;
+
+			if (ctx->flags & IR_USE_FRAME_POINTER) {
+				fp = IR_REG_FRAME_POINTER;
+				shadow_store = sizeof(void*) * 2;
+			} else {
+				fp = IR_REG_STACK_POINTER;
+				shadow_store = ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*);
+			}
+
+			while (offset < cc->shadow_store_size && n < cc->int_param_regs_count) {
+				|	mov [Ra(fp)+shadow_store+offset], Ra(cc->int_param_regs[n])
+				n++;
+				offset += sizeof(void*);
+			}
+		}
+
+		if (cc->sysv_varargs) {
+			IR_ASSERT(sizeof(void*) == 8);
+#ifdef IR_TARGET_X64
+|.if X64
+			int32_t i;
+			ir_reg fp;
+			int offset;
+
+			if (ctx->flags & IR_USE_FRAME_POINTER) {
+				fp = IR_REG_FRAME_POINTER;
+
+				offset = -(int32_t)(ctx->stack_frame_size - ctx->locals_area_size);
+			} else {
+				fp = IR_REG_STACK_POINTER;
+				offset = ctx->locals_area_size + ctx->call_stack_size;
+			}
+
+			if ((ctx->flags2 & (IR_HAS_VA_ARG_GP|IR_HAS_VA_COPY)) && ctx->gp_reg_params < cc->int_param_regs_count) {
+				/* skip named args */
+				offset += sizeof(void*) * ctx->gp_reg_params;
+				for (i = ctx->gp_reg_params; i < cc->int_param_regs_count; i++) {
+					|	mov qword [Ra(fp)+offset], Rq(cc->int_param_regs[i])
+					offset += sizeof(void*);
+				}
+			}
+			if ((ctx->flags2 & (IR_HAS_VA_ARG_FP|IR_HAS_VA_COPY)) && ctx->fp_reg_params < cc->fp_param_regs_count) {
+				|	test al, al
+				|	je	>1
+				/* skip named args */
+				offset += 16 * ctx->fp_reg_params;
+				for (i = ctx->fp_reg_params; i < cc->fp_param_regs_count; i++) {
+					|	movaps [Ra(fp)+offset], xmm(cc->fp_param_regs[i]-IR_REG_FP_FIRST)
+					offset += 16;
+				}
+				|1:
+			}
+|.endif
+#endif
+		}
+	}
+}
+
+static void ir_emit_epilogue(ir_ctx *ctx)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_FP)) {
+		int i;
+		int offset;
+		ir_reg fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+		ir_regset used_preserved_regs = (ir_regset)ctx->used_preserved_regs;
+
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			fp = IR_REG_FRAME_POINTER;
+			offset = 0;
+		} else {
+			fp = IR_REG_STACK_POINTER;
+			offset = ctx->stack_frame_size + ctx->call_stack_size;
+		}
+		for (i = 0; i < IR_REG_NUM; i++) {
+			if (IR_REGSET_IN(used_preserved_regs, i)) {
+				if (i < IR_REG_FP_FIRST) {
+					offset -= sizeof(void*);
+				} else {
+					offset -= sizeof(void*);
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovsd xmm(i-IR_REG_FP_FIRST), qword [Ra(fp)+offset]
+					} else {
+						|	movsd xmm(i-IR_REG_FP_FIRST), qword [Ra(fp)+offset]
+					}
+				}
+			}
+		}
+	}
+
+	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP)) {
+		int i;
+		ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP);
+		int offset;
+
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			offset = 0;
+		} else {
+			offset = ctx->stack_frame_size + ctx->call_stack_size;
+		}
+		if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP)) {
+			int i;
+			ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP);
+
+			for (i = IR_REG_GP_LAST; i >= IR_REG_GP_FIRST; i--) {
+				if (IR_REGSET_IN(used_preserved_regs, i)) {
+					offset -= sizeof(void*);
+				}
+			}
+		}
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			|	lea Ra(IR_REG_RSP), [Ra(IR_REG_RBP)+offset]
+		} else if (offset) {
+			|	add Ra(IR_REG_RSP), offset
+		}
+		for (i = IR_REG_GP_LAST; i >= IR_REG_GP_FIRST; i--) {
+			if (IR_REGSET_IN(used_preserved_regs, i)) {
+				|	pop Ra(i)
+			}
+		}
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			|	pop Ra(IR_REG_RBP)
+		}
+	} else if (ctx->flags & IR_USE_FRAME_POINTER) {
+		|	mov Ra(IR_REG_RSP), Ra(IR_REG_RBP)
+		|	pop Ra(IR_REG_RBP)
+	} else if (ctx->stack_frame_size + ctx->call_stack_size) {
+		if (ctx->fixed_stack_red_zone) {
+			IR_ASSERT(ctx->stack_frame_size + ctx->call_stack_size <= ctx->fixed_stack_red_zone);
+		} else {
+			|	add Ra(IR_REG_RSP), (ctx->stack_frame_size + ctx->call_stack_size)
+		}
+	}
+}
+
+static void ir_emit_binop_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+		if (op1 == op2) {
+			op2_reg = def_reg;
+		}
+	}
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load(ctx, type, op2_reg, op2);
+			}
+		}
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	ASM_REG_REG_OP add, type, def_reg, op2_reg
+				break;
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	ASM_REG_REG_OP sub, type, def_reg, op2_reg
+				break;
+			case IR_MUL:
+			case IR_MUL_OV:
+				|	ASM_REG_REG_MUL imul, type, def_reg, op2_reg
+				break;
+			case IR_OR:
+				|	ASM_REG_REG_OP or, type, def_reg, op2_reg
+				break;
+			case IR_AND:
+				|	ASM_REG_REG_OP and, type, def_reg, op2_reg
+				break;
+			case IR_XOR:
+				|	ASM_REG_REG_OP xor, type, def_reg, op2_reg
+				break;
+		}
+	} else if (IR_IS_CONST_REF(op2)) {
+		int32_t val = ir_fuse_imm(ctx, op2);
+
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	ASM_REG_IMM_OP add, type, def_reg, val
+				break;
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	ASM_REG_IMM_OP sub, type, def_reg, val
+				break;
+			case IR_MUL:
+			case IR_MUL_OV:
+				|	ASM_REG_IMM_MUL imul, type, def_reg, val
+				break;
+			case IR_OR:
+				|	ASM_REG_IMM_OP or, type, def_reg, val
+				break;
+			case IR_AND:
+				|	ASM_REG_IMM_OP and, type, def_reg, val
+				break;
+			case IR_XOR:
+				|	ASM_REG_IMM_OP xor, type, def_reg, val
+				break;
+		}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+		}
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	ASM_REG_MEM_OP add, type, def_reg, mem
+				break;
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	ASM_REG_MEM_OP sub, type, def_reg, mem
+				break;
+			case IR_MUL:
+			case IR_MUL_OV:
+				|	ASM_REG_MEM_MUL imul, type, def_reg, mem
+				break;
+			case IR_OR:
+				|	ASM_REG_MEM_OP or, type, def_reg, mem
+				break;
+			case IR_AND:
+				|	ASM_REG_MEM_OP and, type, def_reg, mem
+				break;
+			case IR_XOR:
+				|	ASM_REG_MEM_OP xor, type, def_reg, mem
+				break;
+		}
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_imul3(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	int32_t val = ir_fuse_imm(ctx, op2);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	IR_ASSERT(!IR_IS_CONST_REF(op1));
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, op1);
+		}
+		switch (ir_type_size[type]) {
+			default:
+				IR_ASSERT(0);
+			case 2:
+				|	imul Rw(def_reg), Rw(op1_reg), val
+				break;
+			case 4:
+				|	imul Rd(def_reg), Rd(op1_reg), val
+				break;
+|.if X64
+||			case 8:
+|				imul Rq(def_reg), Rq(op1_reg), val
+||				break;
+|.endif
+		}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op1);
+		}
+		|	ASM_REG_MEM_TXT_MUL imul, type, def_reg, mem, val
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_min_max_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+
+	if (op1 == op2) {
+		goto done;
+	}
+
+	IR_ASSERT(op2_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+	}
+
+	|	ASM_REG_REG_OP cmp, type, def_reg, op2_reg
+	if (insn->op == IR_MIN) {
+		if (IR_IS_TYPE_SIGNED(type)) {
+			|	ASM_REG_REG_OP2 cmovg, type, def_reg, op2_reg
+		} else {
+			|	ASM_REG_REG_OP2 cmova, type, def_reg, op2_reg
+		}
+	} else {
+		IR_ASSERT(insn->op == IR_MAX);
+		if (IR_IS_TYPE_SIGNED(type)) {
+			|	ASM_REG_REG_OP2 cmovl, type, def_reg, op2_reg
+		} else {
+			|	ASM_REG_REG_OP2 cmovb, type, def_reg, op2_reg
+		}
+	}
+
+done:
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_overflow(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_type type = ctx->ir_base[insn->op1].type;
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	IR_ASSERT(IR_IS_TYPE_INT(type));
+	if (IR_IS_TYPE_SIGNED(type)) {
+		|	seto Rb(def_reg)
+	} else {
+		|	setc Rb(def_reg)
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_overflow_and_branch(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *overflow_insn = &ctx->ir_base[insn->op2];
+	ir_type type = ctx->ir_base[overflow_insn->op1].type;
+	uint32_t true_block, false_block;
+	bool reverse = 0;
+
+	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+	if (true_block == next_block) {
+		reverse = 1;
+		true_block = false_block;
+		false_block = 0;
+	} else if (false_block == next_block) {
+		false_block = 0;
+	}
+
+	if (IR_IS_TYPE_SIGNED(type)) {
+		if (reverse) {
+			|	jno =>true_block
+		} else {
+			|	jo =>true_block
+		}
+	} else {
+		if (reverse) {
+			|	jnc =>true_block
+		} else {
+			|	jc =>true_block
+		}
+	}
+	if (false_block) {
+		|	jmp =>false_block
+	}
+}
+
+static void ir_emit_mem_binop_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *op_insn = &ctx->ir_base[insn->op3];
+	ir_type type = op_insn->type;
+	ir_ref op2 = op_insn->op2;
+	ir_reg op2_reg = ctx->regs[insn->op3][2];
+	ir_mem mem;
+
+	if (insn->op == IR_STORE) {
+		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
+	} else {
+		IR_ASSERT(insn->op == IR_VSTORE);
+		mem = ir_var_spill_slot(ctx, insn->op2);
+	}
+
+	if (op2_reg == IR_REG_NONE) {
+		int32_t val = ir_fuse_imm(ctx, op2);
+
+		switch (op_insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	ASM_MEM_IMM_OP add, type, mem, val
+				break;
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	ASM_MEM_IMM_OP sub, type, mem, val
+				break;
+			case IR_OR:
+				|	ASM_MEM_IMM_OP or, type, mem, val
+				break;
+			case IR_AND:
+				|	ASM_MEM_IMM_OP and, type, mem, val
+				break;
+			case IR_XOR:
+				|	ASM_MEM_IMM_OP xor, type, mem, val
+				break;
+		}
+	} else {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		switch (op_insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	ASM_MEM_REG_OP add, type, mem, op2_reg
+				break;
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	ASM_MEM_REG_OP sub, type, mem, op2_reg
+				break;
+			case IR_OR:
+				|	ASM_MEM_REG_OP or, type, mem, op2_reg
+				break;
+			case IR_AND:
+				|	ASM_MEM_REG_OP and, type, mem, op2_reg
+				break;
+			case IR_XOR:
+				|	ASM_MEM_REG_OP xor, type, mem, op2_reg
+				break;
+		}
+	}
+}
+
+static void ir_emit_reg_binop_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *op_insn = &ctx->ir_base[insn->op2];
+	ir_type type = op_insn->type;
+	ir_ref op2 = op_insn->op2;
+	ir_reg op2_reg = ctx->regs[insn->op2][2];
+	ir_reg reg;
+
+	IR_ASSERT(insn->op == IR_RSTORE);
+	reg = insn->op3;
+
+	if (op2_reg == IR_REG_NONE) {
+		int32_t val = ir_fuse_imm(ctx, op2);
+
+		switch (op_insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_REG_IMM_OP add, type, reg, val
+				break;
+			case IR_SUB:
+				|	ASM_REG_IMM_OP sub, type, reg, val
+				break;
+			case IR_OR:
+				|	ASM_REG_IMM_OP or, type, reg, val
+				break;
+			case IR_AND:
+				|	ASM_REG_IMM_OP and, type, reg, val
+				break;
+			case IR_XOR:
+				|	ASM_REG_IMM_OP xor, type, reg, val
+				break;
+		}
+	} else {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		switch (op_insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_REG_REG_OP add, type, reg, op2_reg
+				break;
+			case IR_SUB:
+				|	ASM_REG_REG_OP sub, type, reg, op2_reg
+				break;
+			case IR_OR:
+				|	ASM_REG_REG_OP or, type, reg, op2_reg
+				break;
+			case IR_AND:
+				|	ASM_REG_REG_OP and, type, reg, op2_reg
+				break;
+			case IR_XOR:
+				|	ASM_REG_REG_OP xor, type, reg, op2_reg
+				break;
+		}
+	}
+}
+
+static void ir_emit_mul_div_mod_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+	if (insn->op == IR_MUL) {
+		uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
+
+		if (shift == 1) {
+			|	ASM_REG_REG_OP add, type, def_reg, def_reg
+		} else {
+			|	ASM_REG_IMM_OP shl, type, def_reg, shift
+		}
+	} else if (insn->op == IR_DIV) {
+		uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
+
+		|	ASM_REG_IMM_OP shr, type, def_reg, shift
+	} else {
+		IR_ASSERT(insn->op == IR_MOD);
+		uint64_t mask = ctx->ir_base[insn->op2].val.u64 - 1;
+
+|.if X64
+||		if (ir_type_size[type] == 8 && ctx->regs[def][2] != IR_REG_NONE) {
+||			ir_reg op2_reg = ctx->regs[def][2];
+||
+||			op2_reg = IR_REG_NUM(op2_reg);
+||			ir_emit_load_imm_int(ctx, type, op2_reg, mask);
+			|	ASM_REG_REG_OP and, type, def_reg, op2_reg
+||		} else {
+|.endif
+			|	ASM_REG_IMM_OP and, type, def_reg, mask
+|.if X64
+||		}
+|.endif
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_bit_op(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+	if (insn->op == IR_OR) {
+		uint32_t bit = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
+
+		|	ASM_REG16_IMM_OP, bts, type, def_reg, bit
+	} else {
+		IR_ASSERT(insn->op == IR_AND);
+		uint32_t bit = IR_LOG2(~ctx->ir_base[insn->op2].val.u64);
+
+		|	ASM_REG16_IMM_OP, btr, type, def_reg, bit
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_sdiv_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
+	int64_t offset = ctx->ir_base[insn->op2].val.u64 - 1;
+
+	IR_ASSERT(shift != 0);
+	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
+	IR_ASSERT(op1_reg != IR_REG_NONE && def_reg != IR_REG_NONE && op1_reg != def_reg);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+
+	if (shift == 1) {
+|.if X64
+||		if (ir_type_size[type] == 8) {
+			|	mov Rq(def_reg), Rq(op1_reg)
+			|	ASM_REG_IMM_OP shr, type, def_reg, 63
+			|	add Rq(def_reg), Rq(op1_reg)
+||		} else {
+|.endif
+			|	mov Rd(def_reg), Rd(op1_reg)
+			|	ASM_REG_IMM_OP shr, type, def_reg, (ir_type_size[type]*8-1)
+			|	add Rd(def_reg), Rd(op1_reg)
+|.if X64
+||		}
+|.endif
+	} else {
+|.if X64
+||		if (ir_type_size[type] == 8) {
+||			ir_reg op2_reg = ctx->regs[def][2];
+||
+||			if (op2_reg != IR_REG_NONE) {
+||				op2_reg =  IR_REG_NUM(op2_reg);
+||				ir_emit_load_imm_int(ctx, type, op2_reg, offset);
+				|	lea Rq(def_reg), [Rq(op1_reg)+Rq(op2_reg)]
+||			} else {
+				|	lea Rq(def_reg), [Rq(op1_reg)+(int32_t)offset]
+||			}
+||		} else {
+|.endif
+			|	lea Rd(def_reg), [Rd(op1_reg)+(int32_t)offset]
+|.if X64
+||		}
+|.endif
+		|	ASM_REG_REG_OP test, type, op1_reg, op1_reg
+		|	ASM_REG_REG_OP2 cmovns, type, def_reg, op1_reg
+	}
+	|	ASM_REG_IMM_OP sar, type, def_reg, shift
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_smod_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
+	uint64_t mask = ctx->ir_base[insn->op2].val.u64 - 1;
+
+	IR_ASSERT(shift != 0);
+	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
+	IR_ASSERT(def_reg != IR_REG_NONE && tmp_reg != IR_REG_NONE && def_reg != tmp_reg);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+	if (tmp_reg != op1_reg) {
+		ir_emit_mov(ctx, type, tmp_reg, def_reg);
+	}
+
+
+	if (shift == 1) {
+		|	ASM_REG_IMM_OP shr, type, tmp_reg, (ir_type_size[type]*8-1)
+	} else {
+		|	ASM_REG_IMM_OP sar, type, tmp_reg, (ir_type_size[type]*8-1)
+		|	ASM_REG_IMM_OP shr, type, tmp_reg, (ir_type_size[type]*8-shift)
+	}
+	|	ASM_REG_REG_OP add, type, def_reg, tmp_reg
+
+|.if X64
+||	if (ir_type_size[type] == 8 && ctx->regs[def][2] != IR_REG_NONE) {
+||		ir_reg op2_reg = ctx->regs[def][2];
+||
+||		op2_reg = IR_REG_NUM(op2_reg);
+||		ir_emit_load_imm_int(ctx, type, op2_reg, mask);
+		|	ASM_REG_REG_OP and, type, def_reg, op2_reg
+||	} else {
+|.endif
+		|	ASM_REG_IMM_OP and, type, def_reg, mask
+|.if X64
+||	}
+|.endif
+
+	|	ASM_REG_REG_OP sub, type, def_reg, tmp_reg
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_mem_mul_div_mod_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *op_insn = &ctx->ir_base[insn->op3];
+	ir_type type = op_insn->type;
+	ir_mem mem;
+
+	IR_ASSERT(IR_IS_CONST_REF(op_insn->op2));
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[op_insn->op2].op));
+
+	if (insn->op == IR_STORE) {
+		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
+	} else {
+		IR_ASSERT(insn->op == IR_VSTORE);
+		mem = ir_var_spill_slot(ctx, insn->op2);
+	}
+
+	if (op_insn->op == IR_MUL) {
+		uint32_t shift = IR_LOG2(ctx->ir_base[op_insn->op2].val.u64);
+		|	ASM_MEM_IMM_OP shl, type, mem, shift
+	} else if (op_insn->op == IR_DIV) {
+		uint32_t shift = IR_LOG2(ctx->ir_base[op_insn->op2].val.u64);
+		|	ASM_MEM_IMM_OP shr, type, mem, shift
+	} else {
+		IR_ASSERT(op_insn->op == IR_MOD);
+		uint64_t mask = ctx->ir_base[op_insn->op2].val.u64 - 1;
+		IR_ASSERT(IR_IS_UNSIGNED_32BIT(mask));
+		|	ASM_MEM_IMM_OP and, type, mem, mask
+	}
+}
+
+static void ir_emit_shift(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+
+	IR_ASSERT(def_reg != IR_REG_NONE && def_reg != IR_REG_RCX);
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, insn->op2);
+	}
+	if (op2_reg != IR_REG_RCX) {
+		if (op1_reg == IR_REG_RCX) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+			op1_reg = def_reg;
+		}
+		if (op2_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, IR_REG_RCX, op2_reg);
+		} else {
+			ir_emit_load(ctx, type, IR_REG_RCX, insn->op2);
+		}
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, insn->op1);
+		}
+	}
+	switch (insn->op) {
+		default:
+			IR_ASSERT(0);
+		case IR_SHL:
+			|	ASM_REG_TXT_OP shl, insn->type, def_reg, cl
+			break;
+		case IR_SHR:
+			|	ASM_REG_TXT_OP shr, insn->type, def_reg, cl
+			break;
+		case IR_SAR:
+			|	ASM_REG_TXT_OP sar, insn->type, def_reg, cl
+			break;
+		case IR_ROL:
+			|	ASM_REG_TXT_OP rol, insn->type, def_reg, cl
+			break;
+		case IR_ROR:
+			|	ASM_REG_TXT_OP ror, insn->type, def_reg, cl
+			break;
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_mem_shift(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *op_insn = &ctx->ir_base[insn->op3];
+	ir_type type = op_insn->type;
+	ir_ref op2 = op_insn->op2;
+	ir_reg op2_reg = ctx->regs[insn->op3][2];
+	ir_mem mem;
+
+	if (insn->op == IR_STORE) {
+		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
+	} else {
+		IR_ASSERT(insn->op == IR_VSTORE);
+		mem = ir_var_spill_slot(ctx, insn->op2);
+	}
+
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+	}
+	if (op2_reg != IR_REG_RCX) {
+		if (op2_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, IR_REG_RCX, op2_reg);
+		} else {
+			ir_emit_load(ctx, type, IR_REG_RCX, op2);
+		}
+	}
+	switch (op_insn->op) {
+		default:
+			IR_ASSERT(0);
+		case IR_SHL:
+			|	ASM_MEM_TXT_OP shl, type, mem, cl
+			break;
+		case IR_SHR:
+			|	ASM_MEM_TXT_OP shr, type, mem, cl
+			break;
+		case IR_SAR:
+			|	ASM_MEM_TXT_OP sar, type, mem, cl
+			break;
+		case IR_ROL:
+			|	ASM_MEM_TXT_OP rol, type, mem, cl
+			break;
+		case IR_ROR:
+			|	ASM_MEM_TXT_OP ror, type, mem, cl
+			break;
+	}
+}
+
+static void ir_emit_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	int32_t shift;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
+	IR_ASSERT(IR_IS_SIGNED_32BIT(ctx->ir_base[insn->op2].val.i64));
+	shift = ctx->ir_base[insn->op2].val.u8;
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+	switch (insn->op) {
+		default:
+			IR_ASSERT(0);
+		case IR_SHL:
+			|	ASM_REG_IMM_OP shl, insn->type, def_reg, shift
+			break;
+		case IR_SHR:
+			|	ASM_REG_IMM_OP shr, insn->type, def_reg, shift
+			break;
+		case IR_SAR:
+			|	ASM_REG_IMM_OP sar, insn->type, def_reg, shift
+			break;
+		case IR_ROL:
+			|	ASM_REG_IMM_OP rol, insn->type, def_reg, shift
+			break;
+		case IR_ROR:
+			|	ASM_REG_IMM_OP ror, insn->type, def_reg, shift
+			break;
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_mem_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *op_insn = &ctx->ir_base[insn->op3];
+	ir_type type = op_insn->type;
+	int32_t shift;
+	ir_mem mem;
+
+	IR_ASSERT(IR_IS_CONST_REF(op_insn->op2));
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[op_insn->op2].op));
+	IR_ASSERT(IR_IS_SIGNED_32BIT(ctx->ir_base[op_insn->op2].val.i64));
+	shift = ctx->ir_base[op_insn->op2].val.i32;
+	if (insn->op == IR_STORE) {
+		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
+	} else {
+		IR_ASSERT(insn->op == IR_VSTORE);
+		mem = ir_var_spill_slot(ctx, insn->op2);
+	}
+
+	switch (op_insn->op) {
+		default:
+			IR_ASSERT(0);
+		case IR_SHL:
+			|	ASM_MEM_IMM_OP shl, type, mem, shift
+			break;
+		case IR_SHR:
+			|	ASM_MEM_IMM_OP shr, type, mem, shift
+			break;
+		case IR_SAR:
+			|	ASM_MEM_IMM_OP sar, type, mem, shift
+			break;
+		case IR_ROL:
+			|	ASM_MEM_IMM_OP rol, type, mem, shift
+			break;
+		case IR_ROR:
+			|	ASM_MEM_IMM_OP ror, type, mem, shift
+			break;
+	}
+}
+
+static void ir_emit_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+	if (rule == IR_INC) {
+		|	ASM_REG_OP inc, insn->type, def_reg
+	} else if (rule == IR_DEC) {
+		|	ASM_REG_OP dec, insn->type, def_reg
+	} else if (insn->op == IR_NOT) {
+		|	ASM_REG_OP not, insn->type, def_reg
+	} else if (insn->op == IR_NEG) {
+		|	ASM_REG_OP neg, insn->type, def_reg
+	} else {
+		IR_ASSERT(insn->op == IR_BSWAP);
+		switch (ir_type_size[insn->type]) {
+			default:
+				IR_ASSERT(0);
+			case 4:
+				|	bswap Rd(def_reg)
+				break;
+			case 8:
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	bswap Rq(def_reg)
+|.endif
+				break;
+		}
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_bit_count(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, op1);
+		}
+		switch (ir_type_size[src_type]) {
+			default:
+				IR_ASSERT(0);
+			case 2:
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	lzcnt Rw(def_reg), Rw(op1_reg)
+					} else {
+						|	bsr Rw(def_reg), Rw(op1_reg)
+						|	xor Rw(def_reg), 0xf
+					}
+				} else if (insn->op == IR_CTTZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	tzcnt Rw(def_reg), Rw(op1_reg)
+					} else {
+						|	bsf Rw(def_reg), Rw(op1_reg)
+					}
+				} else {
+					IR_ASSERT(insn->op == IR_CTPOP);
+					|	popcnt Rw(def_reg), Rw(op1_reg)
+				}
+				break;
+			case 1:
+				|   movzx Rd(op1_reg), Rb(op1_reg)
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	lzcnt Rd(def_reg), Rd(op1_reg)
+						|	sub Rd(def_reg), 24
+					} else {
+						|	bsr Rd(def_reg), Rd(op1_reg)
+						|	xor Rw(def_reg), 0x7
+					}
+					break;
+				}
+				IR_FALLTHROUGH;
+			case 4:
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	lzcnt Rd(def_reg), Rd(op1_reg)
+					} else {
+						|	bsr Rd(def_reg), Rd(op1_reg)
+						|	xor Rw(def_reg), 0x1f
+					}
+				} else if (insn->op == IR_CTTZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	tzcnt Rd(def_reg), Rd(op1_reg)
+					} else {
+						|	bsf Rd(def_reg), Rd(op1_reg)
+					}
+				} else {
+					IR_ASSERT(insn->op == IR_CTPOP);
+					|	popcnt Rd(def_reg), Rd(op1_reg)
+				}
+				break;
+|.if X64
+			case 8:
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	lzcnt Rq(def_reg), Rq(op1_reg)
+					} else {
+						|	bsr Rq(def_reg), Rq(op1_reg)
+						|	xor Rw(def_reg), 0x3f
+					}
+				} else if (insn->op == IR_CTTZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	tzcnt Rq(def_reg), Rq(op1_reg)
+					} else {
+						|	bsf Rq(def_reg), Rq(op1_reg)
+					}
+				} else {
+					IR_ASSERT(insn->op == IR_CTPOP);
+					|	popcnt Rq(def_reg), Rq(op1_reg)
+				}
+				break;
+|.endif
+		}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op1);
+		}
+		switch (ir_type_size[src_type]) {
+			default:
+				IR_ASSERT(0);
+			case 2:
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	ASM_TXT_TMEM_OP lzcnt, Rw(def_reg), word, mem
+					} else {
+						|	ASM_TXT_TMEM_OP bsr, Rw(def_reg), word, mem
+						|	xor Rw(def_reg), 0xf
+					}
+				} else if (insn->op == IR_CTTZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	ASM_TXT_TMEM_OP tzcnt, Rw(def_reg), word, mem
+					} else {
+						|	ASM_TXT_TMEM_OP bsf, Rw(def_reg), word, mem
+					}
+				} else {
+					|	ASM_TXT_TMEM_OP popcnt, Rw(def_reg), word, mem
+				}
+				break;
+			case 4:
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	ASM_TXT_TMEM_OP lzcnt, Rd(def_reg), dword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP bsr, Rd(def_reg), dword, mem
+						|	xor Rw(def_reg), 0x1f
+					}
+				} else if (insn->op == IR_CTTZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	ASM_TXT_TMEM_OP tzcnt, Rd(def_reg), dword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP bsf, Rd(def_reg), dword, mem
+					}
+				} else {
+					|	ASM_TXT_TMEM_OP popcnt, Rd(def_reg), dword, mem
+				}
+				break;
+|.if X64
+			case 8:
+				if (insn->op == IR_CTLZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	ASM_TXT_TMEM_OP lzcnt, Rq(def_reg), qword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP bsr, Rq(def_reg), qword, mem
+						|	xor Rw(def_reg), 0x3f
+					}
+				} else if (insn->op == IR_CTTZ) {
+					if (ctx->mflags & IR_X86_BMI1) {
+						|	ASM_TXT_TMEM_OP tzcnt, Rq(def_reg), qword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP bsf, Rq(def_reg), qword, mem
+					}
+				} else {
+					|	ASM_TXT_TMEM_OP popcnt, Rq(def_reg), qword, mem
+				}
+				break;
+|.endif
+		}
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_ctpop(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][2];
+|.if X64
+||	ir_reg const_reg = ctx->regs[def][3];
+|.endif
+
+	IR_ASSERT(def_reg != IR_REG_NONE && tmp_reg != IR_REG_NONE);
+	if (op1_reg == IR_REG_NONE) {
+		ir_emit_load(ctx, src_type, def_reg, op1);
+		if (ir_type_size[src_type] == 1) {
+			|	movzx Rd(def_reg), Rb(def_reg)
+		} else if (ir_type_size[src_type] == 2) {
+			|	movzx Rd(def_reg), Rw(def_reg)
+		}
+	} else {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, op1);
+		}
+		switch (ir_type_size[src_type]) {
+			default:
+				IR_ASSERT(0);
+			case 1:
+				|	movzx Rd(def_reg), Rb(op1_reg)
+				break;
+			case 2:
+				|	movzx Rd(def_reg), Rw(op1_reg)
+				break;
+			case 4:
+				|	mov Rd(def_reg), Rd(op1_reg)
+				break;
+|.if X64
+||			case 8:
+				|	mov Rq(def_reg), Rq(op1_reg)
+||				break;
+|.endif
+		}
+	}
+	switch (ir_type_size[src_type]) {
+		default:
+			IR_ASSERT(0);
+		case 1:
+			|	mov Rd(tmp_reg), Rd(def_reg)
+			|	shr Rd(def_reg), 1
+			|	and Rd(def_reg), 0x55
+			|	sub Rd(tmp_reg), Rd(def_reg)
+			|	mov Rd(def_reg), Rd(tmp_reg)
+			|	and Rd(def_reg), 0x33
+			|	shr Rd(tmp_reg), 2
+			|	and Rd(tmp_reg), 0x33
+			|	add Rd(tmp_reg), Rd(def_reg)
+			|	mov Rd(def_reg), Rd(tmp_reg)
+			|	shr Rd(def_reg), 4
+			|	add Rd(def_reg), Rd(tmp_reg)
+			|	and Rd(def_reg), 0x0f
+			break;
+		case 2:
+			|	mov Rd(tmp_reg), Rd(def_reg)
+			|	shr Rd(def_reg), 1
+			|	and Rd(def_reg), 0x5555
+			|	sub Rd(tmp_reg), Rd(def_reg)
+			|	mov Rd(def_reg), Rd(tmp_reg)
+			|	and Rd(def_reg), 0x3333
+			|	shr Rd(tmp_reg), 2
+			|	and Rd(tmp_reg), 0x3333
+			|	add Rd(tmp_reg), Rd(def_reg)
+			|	mov Rd(def_reg), Rd(tmp_reg)
+			|	shr Rd(def_reg), 4
+			|	add Rd(def_reg), Rd(tmp_reg)
+			|	and Rd(def_reg), 0x0f0f
+			|	mov	Rd(tmp_reg), Rd(def_reg)
+			|	shr Rd(tmp_reg), 8
+			|	and Rd(def_reg), 0x0f
+			|	add Rd(def_reg), Rd(tmp_reg)
+			break;
+		case 4:
+			|	mov Rd(tmp_reg), Rd(def_reg)
+			|	shr Rd(def_reg), 1
+			|	and Rd(def_reg), 0x55555555
+			|	sub Rd(tmp_reg), Rd(def_reg)
+			|	mov Rd(def_reg), Rd(tmp_reg)
+			|	and Rd(def_reg), 0x33333333
+			|	shr Rd(tmp_reg), 2
+			|	and Rd(tmp_reg), 0x33333333
+			|	add Rd(tmp_reg), Rd(def_reg)
+			|	mov Rd(def_reg), Rd(tmp_reg)
+			|	shr Rd(def_reg), 4
+			|	add Rd(def_reg), Rd(tmp_reg)
+			|	and Rd(def_reg), 0x0f0f0f0f
+			|	imul Rd(def_reg), 0x01010101
+			|	shr Rd(def_reg), 24
+			break;
+|.if X64
+||		case 8:
+||			IR_ASSERT(const_reg != IR_REG_NONE);
+			|	mov Rq(tmp_reg), Rq(def_reg)
+			|	shr Rq(def_reg), 1
+			|	mov64 Rq(const_reg), 0x5555555555555555
+			|	and Rq(def_reg), Rq(const_reg)
+			|	sub Rq(tmp_reg), Rq(def_reg)
+			|	mov Rq(def_reg), Rq(tmp_reg)
+			|	mov64 Rq(const_reg), 0x3333333333333333
+			|	and Rq(def_reg), Rq(const_reg)
+			|	shr Rq(tmp_reg), 2
+			|	and Rq(tmp_reg), Rq(const_reg)
+			|	add Rq(tmp_reg), Rq(def_reg)
+			|	mov Rq(def_reg), Rq(tmp_reg)
+			|	shr Rq(def_reg), 4
+			|	add Rq(def_reg), Rq(tmp_reg)
+			|	mov64 Rq(const_reg), 0x0f0f0f0f0f0f0f0f
+			|	and Rq(def_reg), Rq(const_reg)
+			|	mov64 Rq(const_reg), 0x0101010101010101
+			|	imul Rq(def_reg), Rq(const_reg)
+			|	shr Rq(def_reg), 56
+||			break;
+|.endif
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_mem_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *op_insn = &ctx->ir_base[insn->op3];
+	ir_type type = op_insn->type;
+	ir_mem mem;
+
+	if (insn->op == IR_STORE) {
+		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
+	} else {
+		IR_ASSERT(insn->op == IR_VSTORE);
+		mem = ir_var_spill_slot(ctx, insn->op2);
+	}
+
+	if (rule == IR_MEM_INC) {
+		|	ASM_MEM_OP inc, type, mem
+	} else if (rule == IR_MEM_DEC) {
+		|	ASM_MEM_OP dec, type, mem
+	} else if (op_insn->op == IR_NOT) {
+		|	ASM_MEM_OP not, type, mem
+	} else {
+		IR_ASSERT(op_insn->op == IR_NEG);
+		|	ASM_MEM_OP neg, type, mem
+	}
+}
+
+static void ir_emit_abs_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+
+	IR_ASSERT(def_reg != op1_reg);
+
+	ir_emit_mov(ctx, insn->type, def_reg, op1_reg);
+	|	ASM_REG_OP neg, insn->type, def_reg
+	|	ASM_REG_REG_OP2, cmovs, type, def_reg, op1_reg
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_bool_not(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[insn->op1].type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, op1);
+		}
+
+		if (def_reg != op1_reg) {
+			|	mov Rb(def_reg), Rb(op1_reg)
+		}
+	} else {
+		ir_emit_load(ctx, type, def_reg, op1);
+	}
+
+	|	xor Rb(def_reg), 1
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_bool_not_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[insn->op1].type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+
+	if (op1_reg != IR_REG_NONE) {
+		|	ASM_REG_REG_OP test, type, op1_reg, op1_reg
+	} else {
+		ir_mem mem = ir_ref_spill_slot(ctx, op1);
+
+		|	ASM_MEM_IMM_OP cmp, type, mem, 0
+	}
+	|	sete Rb(def_reg)
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_mul_div_mod(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_mem mem;
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op1_reg != IR_REG_RAX) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_mov(ctx, type, IR_REG_RAX, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, IR_REG_RAX, op1);
+		}
+	}
+	if (op2_reg == IR_REG_NONE && op1 == op2) {
+		op2_reg = IR_REG_RAX;
+	} else if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+	} else if (IR_IS_CONST_REF(op2)
+	 && (insn->op == IR_MUL || insn->op == IR_MUL_OV)) {
+		op2_reg = IR_REG_RDX;
+		ir_emit_load(ctx, type, op2_reg, op2);
+	}
+	if (insn->op == IR_MUL || insn->op == IR_MUL_OV) {
+		if (IR_IS_TYPE_SIGNED(insn->type)) {
+			if (op2_reg != IR_REG_NONE) {
+				|	ASM_REG_OP imul, type, op2_reg
+			} else {
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, def, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				|	ASM_MEM_OP imul, type, mem
+			}
+		} else {
+			if (op2_reg != IR_REG_NONE) {
+				|	ASM_REG_OP mul, type, op2_reg
+			} else {
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, def, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				|	ASM_MEM_OP mul, type, mem
+			}
+		}
+	} else {
+		if (IR_IS_TYPE_SIGNED(type)) {
+			if (ir_type_size[type] == 8) {
+				|	cqo
+			} else if (ir_type_size[type] == 4) {
+				|	cdq
+			} else if (ir_type_size[type] == 2) {
+				|	cwd
+			} else {
+				|	cbw
+			}
+			if (op2_reg != IR_REG_NONE) {
+				|	ASM_REG_OP idiv, type, op2_reg
+			} else {
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, def, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				|	ASM_MEM_OP idiv, type, mem
+			}
+		} else {
+			if (ir_type_size[type] == 1) {
+				|	movzx ax, al
+			} else {
+				|	ASM_REG_REG_OP xor, type, IR_REG_RDX, IR_REG_RDX
+			}
+			if (op2_reg != IR_REG_NONE) {
+				|	ASM_REG_OP div, type, op2_reg
+			} else {
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, def, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				|	ASM_MEM_OP div, type, mem
+			}
+		}
+	}
+
+	if (insn->op == IR_MUL || insn->op == IR_MUL_OV || insn->op == IR_DIV) {
+		if (def_reg != IR_REG_NONE) {
+			if (def_reg != IR_REG_RAX) {
+				ir_emit_mov(ctx, type, def_reg, IR_REG_RAX);
+			}
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store(ctx, type, def, def_reg);
+			}
+		} else {
+			ir_emit_store(ctx, type, def, IR_REG_RAX);
+		}
+	} else {
+		IR_ASSERT(insn->op == IR_MOD);
+		if (ir_type_size[type] == 1) {
+			if (def_reg != IR_REG_NONE) {
+				|	mov al, ah
+				if (def_reg != IR_REG_RAX) {
+					|	mov Rb(def_reg), al
+				}
+				if (IR_REG_SPILLED(ctx->regs[def][0])) {
+					ir_emit_store(ctx, type, def, def_reg);
+				}
+			} else {
+				ir_reg fp;
+				int32_t offset = ir_ref_spill_slot_offset(ctx, def, &fp);
+
+//?????
+				|	mov byte [Ra(fp)+offset], ah
+			}
+		} else {
+			if (def_reg != IR_REG_NONE) {
+				if (def_reg != IR_REG_RDX) {
+					ir_emit_mov(ctx, type, def_reg, IR_REG_RDX);
+				}
+				if (IR_REG_SPILLED(ctx->regs[def][0])) {
+					ir_emit_store(ctx, type, def, def_reg);
+				}
+			} else {
+				ir_emit_store(ctx, type, def, IR_REG_RDX);
+			}
+		}
+	}
+}
+
+static void ir_emit_op_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+	}
+	if (insn->op == IR_NEG) {
+		if (insn->type == IR_DOUBLE) {
+			if (!data->double_neg_const) {
+				data->double_neg_const = 1;
+				ir_rodata(ctx);
+				|.align 16
+				|->double_neg_const:
+				|.dword 0, 0x80000000, 0, 0
+				|.code
+			}
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vxorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->double_neg_const]
+			} else {
+				|	xorpd xmm(def_reg-IR_REG_FP_FIRST), [->double_neg_const]
+			}
+		} else {
+			IR_ASSERT(insn->type == IR_FLOAT);
+			if (!data->float_neg_const) {
+				data->float_neg_const = 1;
+				ir_rodata(ctx);
+				|.align 16
+				|->float_neg_const:
+				|.dword 0x80000000, 0, 0, 0
+				|.code
+			}
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->float_neg_const]
+			} else {
+				|	xorps xmm(def_reg-IR_REG_FP_FIRST), [->float_neg_const]
+			}
+		}
+	} else {
+		IR_ASSERT(insn->op == IR_ABS);
+		if (insn->type == IR_DOUBLE) {
+			if (!data->double_abs_const) {
+				data->double_abs_const = 1;
+				ir_rodata(ctx);
+				|.align 16
+				|->double_abs_const:
+				|.dword 0xffffffff, 0x7fffffff, 0, 0
+				|.code
+			}
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vandpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->double_abs_const]
+			} else {
+				|	andpd xmm(def_reg-IR_REG_FP_FIRST), [->double_abs_const]
+			}
+		} else {
+			IR_ASSERT(insn->type == IR_FLOAT);
+			if (!data->float_abs_const) {
+				data->float_abs_const = 1;
+				ir_rodata(ctx);
+				|.align 16
+				|->float_abs_const:
+				|.dword 0x7fffffff, 0, 0, 0
+				|.code
+			}
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vandps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->float_abs_const]
+			} else {
+				|	andps xmm(def_reg-IR_REG_FP_FIRST), [->float_abs_const]
+			}
+		}
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_binop_sse2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (def_reg != op1_reg) {
+		if (op1_reg != IR_REG_NONE) {
+			ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
+		} else {
+			ir_emit_load(ctx, type, def_reg, op1);
+		}
+		if (op1 == op2) {
+			op2_reg = def_reg;
+		}
+	}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load(ctx, type, op2_reg, op2);
+			}
+		}
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_SSE2_REG_REG_OP adds, type, def_reg, op2_reg
+				break;
+			case IR_SUB:
+				|	ASM_SSE2_REG_REG_OP subs, type, def_reg, op2_reg
+				break;
+			case IR_MUL:
+				|	ASM_SSE2_REG_REG_OP muls, type, def_reg, op2_reg
+				break;
+			case IR_DIV:
+				|	ASM_SSE2_REG_REG_OP divs, type, def_reg, op2_reg
+				break;
+			case IR_MIN:
+				|	ASM_SSE2_REG_REG_OP mins, type, def_reg, op2_reg
+				break;
+			case IR_MAX:
+				|	ASM_SSE2_REG_REG_OP maxs, type, def_reg, op2_reg
+				break;
+		}
+	} else if (IR_IS_CONST_REF(op2)) {
+		int label = ir_get_const_label(ctx, op2);
+
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_SSE2_REG_TXT_OP adds, type, def_reg, [=>label]
+				break;
+			case IR_SUB:
+				|	ASM_SSE2_REG_TXT_OP subs, type, def_reg, [=>label]
+				break;
+			case IR_MUL:
+				|	ASM_SSE2_REG_TXT_OP muls, type, def_reg, [=>label]
+				break;
+			case IR_DIV:
+				|	ASM_SSE2_REG_TXT_OP divs, type, def_reg, [=>label]
+				break;
+			case IR_MIN:
+				|	ASM_SSE2_REG_TXT_OP mins, type, def_reg, [=>label]
+				break;
+			case IR_MAX:
+				|	ASM_SSE2_REG_TXT_OP maxs, type, def_reg, [=>label]
+				break;
+		}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+		}
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_SSE2_REG_MEM_OP adds, type, def_reg, mem
+				break;
+			case IR_SUB:
+				|	ASM_SSE2_REG_MEM_OP subs, type, def_reg, mem
+				break;
+			case IR_MUL:
+				|	ASM_SSE2_REG_MEM_OP muls, type, def_reg, mem
+				break;
+			case IR_DIV:
+				|	ASM_SSE2_REG_MEM_OP divs, type, def_reg, mem
+				break;
+			case IR_MIN:
+				|	ASM_SSE2_REG_MEM_OP mins, type, def_reg, mem
+				break;
+			case IR_MAX:
+				|	ASM_SSE2_REG_MEM_OP maxs, type, def_reg, mem
+				break;
+		}
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_binop_avx(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load(ctx, type, op2_reg, op2);
+			}
+		}
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_AVX_REG_REG_REG_OP vadds, type, def_reg, op1_reg, op2_reg
+				break;
+			case IR_SUB:
+				|	ASM_AVX_REG_REG_REG_OP vsubs, type, def_reg, op1_reg, op2_reg
+				break;
+			case IR_MUL:
+				|	ASM_AVX_REG_REG_REG_OP vmuls, type, def_reg, op1_reg, op2_reg
+				break;
+			case IR_DIV:
+				|	ASM_AVX_REG_REG_REG_OP vdivs, type, def_reg, op1_reg, op2_reg
+				break;
+			case IR_MIN:
+				|	ASM_AVX_REG_REG_REG_OP vmins, type, def_reg, op1_reg, op2_reg
+				break;
+			case IR_MAX:
+				|	ASM_AVX_REG_REG_REG_OP vmaxs, type, def_reg, op1_reg, op2_reg
+				break;
+		}
+	} else if (IR_IS_CONST_REF(op2)) {
+		int label = ir_get_const_label(ctx, op2);
+
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_AVX_REG_REG_TXT_OP vadds, type, def_reg, op1_reg, [=>label]
+				break;
+			case IR_SUB:
+				|	ASM_AVX_REG_REG_TXT_OP vsubs, type, def_reg, op1_reg, [=>label]
+				break;
+			case IR_MUL:
+				|	ASM_AVX_REG_REG_TXT_OP vmuls, type, def_reg, op1_reg, [=>label]
+				break;
+			case IR_DIV:
+				|	ASM_AVX_REG_REG_TXT_OP vdivs, type, def_reg, op1_reg, [=>label]
+				break;
+			case IR_MIN:
+				|	ASM_AVX_REG_REG_TXT_OP vmins, type, def_reg, op1_reg, [=>label]
+				break;
+			case IR_MAX:
+				|	ASM_AVX_REG_REG_TXT_OP vmaxs, type, def_reg, op1_reg, [=>label]
+				break;
+		}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+		}
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				|	ASM_AVX_REG_REG_MEM_OP vadds, type, def_reg, op1_reg, mem
+				break;
+			case IR_SUB:
+				|	ASM_AVX_REG_REG_MEM_OP vsubs, type, def_reg, op1_reg, mem
+				break;
+			case IR_MUL:
+				|	ASM_AVX_REG_REG_MEM_OP vmuls, type, def_reg, op1_reg, mem
+				break;
+			case IR_DIV:
+				|	ASM_AVX_REG_REG_MEM_OP vdivs, type, def_reg, op1_reg, mem
+				break;
+			case IR_MIN:
+				|	ASM_AVX_REG_REG_MEM_OP vmins, type, def_reg, op1_reg, mem
+				break;
+			case IR_MAX:
+				|	ASM_AVX_REG_REG_MEM_OP vmaxs, type, def_reg, op1_reg, mem
+				break;
+		}
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_cmp_int_common(ir_ctx *ctx, ir_type type, ir_ref root, ir_insn *insn, ir_reg op1_reg, ir_ref op1, ir_reg op2_reg, ir_ref op2)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	if (op2_reg == IR_REG_NONE && op1 == op2) {
+	    IR_ASSERT(op1_reg != IR_REG_NONE);
+		op2_reg = op1_reg;
+	}
+
+	if (op1_reg != IR_REG_NONE) {
+		if (op2_reg != IR_REG_NONE) {
+			|	ASM_REG_REG_OP cmp, type, op1_reg, op2_reg
+		} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
+			|	ASM_REG_REG_OP test, type, op1_reg, op1_reg
+		} else if (IR_IS_CONST_REF(op2)) {
+			int32_t val = ir_fuse_imm(ctx, op2);
+			|	ASM_REG_IMM_OP cmp, type, op1_reg, val
+		} else {
+			ir_mem mem;
+
+			if (ir_rule(ctx, op2) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, root, op2);
+			} else {
+				mem = ir_ref_spill_slot(ctx, op2);
+			}
+			|	ASM_REG_MEM_OP cmp, type, op1_reg, mem
+		}
+	} else if (IR_IS_CONST_REF(op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, root, op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op1);
+		}
+		if (op2_reg != IR_REG_NONE) {
+			|	ASM_MEM_REG_OP cmp, type, mem, op2_reg
+		} else {
+			int32_t val = ir_fuse_imm(ctx, op2);
+			|	ASM_MEM_IMM_OP cmp, type, mem, val
+		}
+	}
+}
+
+static void ir_emit_cmp_int_common2(ir_ctx *ctx, ir_ref root, ir_ref ref, ir_insn *cmp_insn)
+{
+	ir_type type = ctx->ir_base[cmp_insn->op1].type;
+	ir_ref op1 = cmp_insn->op1;
+	ir_ref op2 = cmp_insn->op2;
+	ir_reg op1_reg, op2_reg;
+
+	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[ref][1];
+		op2_reg = ctx->regs[ref][2];
+	}
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		if (op1 != op2) {
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+	}
+
+	ir_emit_cmp_int_common(ctx, type, root, cmp_insn, op1_reg, op1, op2_reg, op2);
+}
+
+static void _ir_emit_setcc_int(ir_ctx *ctx, uint8_t op, ir_reg def_reg, bool after_op)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	switch (op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	sete Rb(def_reg)
+			break;
+		case IR_NE:
+			|	setne Rb(def_reg)
+			break;
+		case IR_LT:
+			if (after_op) {
+				|	sets Rb(def_reg)
+			} else {
+				|	setl Rb(def_reg)
+			}
+			break;
+		case IR_GE:
+			if (after_op) {
+				|	setns Rb(def_reg)
+			} else {
+				|	setge Rb(def_reg)
+			}
+			break;
+		case IR_LE:
+			|	setle Rb(def_reg)
+			break;
+		case IR_GT:
+			|	setg Rb(def_reg)
+			break;
+		case IR_ULT:
+			|	setb Rb(def_reg)
+			break;
+		case IR_UGE:
+			|	setae Rb(def_reg)
+			break;
+		case IR_ULE:
+			|	setbe Rb(def_reg)
+			break;
+		case IR_UGT:
+			|	seta Rb(def_reg)
+			break;
+	}
+}
+
+static void _ir_emit_setcc_int_mem(ir_ctx *ctx, uint8_t op, ir_mem mem)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+
+	switch (op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	ASM_TMEM_OP sete, byte, mem
+			break;
+		case IR_NE:
+			|	ASM_TMEM_OP setne, byte, mem
+			break;
+		case IR_LT:
+			|	ASM_TMEM_OP setl, byte, mem
+			break;
+		case IR_GE:
+			|	ASM_TMEM_OP setge, byte, mem
+			break;
+		case IR_LE:
+			|	ASM_TMEM_OP setle, byte, mem
+			break;
+		case IR_GT:
+			|	ASM_TMEM_OP setg, byte, mem
+			break;
+		case IR_ULT:
+			|	ASM_TMEM_OP setb, byte, mem
+			break;
+		case IR_UGE:
+			|	ASM_TMEM_OP setae, byte, mem
+			break;
+		case IR_ULE:
+			|	ASM_TMEM_OP setbe, byte, mem
+			break;
+		case IR_UGT:
+			|	ASM_TMEM_OP seta, byte, mem
+			break;
+	}
+}
+
+static void ir_emit_cmp_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[insn->op1].type;
+	ir_op op = insn->op;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		if (op1 != op2) {
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+	}
+	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
+		if (op == IR_ULT) {
+			/* always false */
+			|	xor Ra(def_reg), Ra(def_reg)
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store(ctx, insn->type, def, def_reg);
+			}
+			return;
+		} else if (op == IR_UGE) {
+			/* always true */
+			|	ASM_REG_IMM_OP mov, insn->type, def_reg, 1
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store(ctx, insn->type, def, def_reg);
+			}
+			return;
+		} else if (op == IR_ULE) {
+			op = IR_EQ;
+		} else if (op == IR_UGT) {
+			op = IR_NE;
+		}
+	}
+	ir_emit_cmp_int_common(ctx, type, def, insn, op1_reg, op1, op2_reg, op2);
+	_ir_emit_setcc_int(ctx, op, def_reg, 0);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_test_int_common(ir_ctx *ctx, ir_ref root, ir_ref ref, ir_op op)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *binop_insn = &ctx->ir_base[ref];
+	ir_type type = binop_insn->type;
+	ir_ref op1 = binop_insn->op1;
+	ir_ref op2 = binop_insn->op2;
+	ir_reg op1_reg, op2_reg;
+
+	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[ref][1];
+		op2_reg = ctx->regs[ref][2];
+	}
+
+	IR_ASSERT(binop_insn->op == IR_AND);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, op1);
+		}
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				if (op1 != op2) {
+					ir_emit_load(ctx, type, op2_reg, op2);
+				}
+			}
+			|	ASM_REG_REG_OP test, type, op1_reg, op2_reg
+		} else if (IR_IS_CONST_REF(op2)) {
+			int32_t val = ir_fuse_imm(ctx, op2);
+
+			if ((op == IR_EQ || op == IR_NE) && val == 0xff && (sizeof(void*) == 8 || op1_reg <= IR_REG_R3)) {
+				|	test Rb(op1_reg), Rb(op1_reg)
+			} else if ((op == IR_EQ || op == IR_NE) && val == 0xff00 && op1_reg <= IR_REG_R3) {
+				if (op1_reg == IR_REG_RAX) {
+					|	test ah, ah
+				} else if (op1_reg == IR_REG_RBX) {
+					|	test bh, bh
+				} else if (op1_reg == IR_REG_RCX) {
+					|	test ch, ch
+				} else if (op1_reg == IR_REG_RDX) {
+					|	test dh, dh
+				} else {
+					IR_ASSERT(0);
+				}
+			} else if ((op == IR_EQ || op == IR_NE) && val == 0xffff) {
+				|	test Rw(op1_reg), Rw(op1_reg)
+			} else if ((op == IR_EQ || op == IR_NE) && val == -1) {
+				|	test Rd(op1_reg), Rd(op1_reg)
+			} else {
+				|	ASM_REG_IMM_OP test, type, op1_reg, val
+			}
+		} else {
+			ir_mem mem;
+
+			if (ir_rule(ctx, op2) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, root, op2);
+			} else {
+				mem = ir_ref_spill_slot(ctx, op2);
+			}
+			|	ASM_REG_MEM_OP test, type, op1_reg, mem
+		}
+	} else if (IR_IS_CONST_REF(op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, root, op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op1);
+		}
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				if (op1 != op2) {
+					ir_emit_load(ctx, type, op2_reg, op2);
+				}
+			}
+			|	ASM_MEM_REG_OP test, type, mem, op2_reg
+		} else {
+			IR_ASSERT(!IR_IS_CONST_REF(op1));
+			int32_t val = ir_fuse_imm(ctx, op2);
+			|	ASM_MEM_IMM_OP test, type, mem, val
+		}
+	}
+}
+
+static void ir_emit_testcc_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	ir_emit_test_int_common(ctx, def, insn->op1, insn->op);
+	_ir_emit_setcc_int(ctx, insn->op, def_reg, 0);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_setcc_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	_ir_emit_setcc_int(ctx, insn->op, def_reg, 1);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_test_bit_common(ir_ctx *ctx, ir_ref root, ir_ref ref)
+{
+#ifdef IR_TARGET_X64
+|.if X64
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *binop_insn = &ctx->ir_base[ref];
+	ir_type type = binop_insn->type;
+	ir_ref op1 = binop_insn->op1;
+	ir_reg op1_reg;
+	uint32_t bit;
+
+	IR_ASSERT(ir_type_size[type] == 8 && IR_IS_CONST_REF(binop_insn->op2));
+
+	bit = IR_LOG2(ctx->ir_base[binop_insn->op2].val.u64);
+	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 1);
+	} else {
+		op1_reg = ctx->regs[ref][1];
+	}
+
+	IR_ASSERT(binop_insn->op == IR_AND);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, op1);
+		}
+
+		|	bt Rq(op1_reg), bit
+	} else if (IR_IS_CONST_REF(op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, root, op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op1);
+		}
+		|	ASM_TMEM_TXT_OP bt, qword, mem, bit
+	}
+|.endif
+#endif
+}
+
+static void ir_emit_testcc_bit(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	ir_emit_test_bit_common(ctx, def, insn->op1);
+	if (insn->op == IR_EQ) {
+		|	setnc Rb(def_reg)
+	} else {
+		IR_ASSERT(insn->op == IR_NE);
+		|	setc Rb(def_reg)
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static ir_op ir_emit_cmp_fp_common(ir_ctx *ctx, ir_ref root, ir_ref cmp_ref, ir_insn *cmp_insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[cmp_insn->op1].type;
+	ir_op op = cmp_insn->op;
+	ir_ref op1, op2;
+	ir_reg op1_reg, op2_reg;
+
+	op1 = cmp_insn->op1;
+	op2 = cmp_insn->op2;
+	if (UNEXPECTED(ctx->rules[cmp_ref] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, root, cmp_ref * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, root, cmp_ref * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[cmp_ref][1];
+		op2_reg = ctx->regs[cmp_ref][2];
+	}
+
+	if (op1_reg == IR_REG_NONE && op2_reg != IR_REG_NONE && (op == IR_EQ || op == IR_NE)) {
+		ir_reg tmp_reg;
+
+		SWAP_REFS(op1, op2);
+		tmp_reg = op1_reg;
+		op1_reg = op2_reg;
+		op2_reg = tmp_reg;
+	}
+
+
+	IR_ASSERT(op1_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load(ctx, type, op2_reg, op2);
+			}
+		}
+		|	ASM_FP_REG_REG_OP ucomis, type, op1_reg, op2_reg
+	} else if (IR_IS_CONST_REF(op2)) {
+		int label = ir_get_const_label(ctx, op2);
+
+		|	ASM_FP_REG_TXT_OP ucomis, type, op1_reg, [=>label]
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, root, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+		}
+		|	ASM_FP_REG_MEM_OP ucomis, type, op1_reg, mem
+	}
+	return op;
+}
+
+static void ir_emit_cmp_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_op op = ir_emit_cmp_fp_common(ctx, def, def, insn);
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg tmp_reg = ctx->regs[def][3];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	switch (op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	setnp Rb(def_reg)
+			|	mov Rd(tmp_reg), 0
+			|	cmovne Rd(def_reg), Rd(tmp_reg)
+			break;
+		case IR_NE:
+			|	setp Rb(def_reg)
+			|	mov Rd(tmp_reg), 1
+			|	cmovne Rd(def_reg), Rd(tmp_reg)
+			break;
+		case IR_LT:
+			|	setnp Rb(def_reg)
+			|	mov Rd(tmp_reg), 0
+			|	cmovae Rd(def_reg), Rd(tmp_reg)
+			break;
+		case IR_GE:
+			|	setae Rb(def_reg)
+			break;
+		case IR_LE:
+			|	setnp Rb(def_reg)
+			|	mov Rd(tmp_reg), 0
+			|	cmova Rd(def_reg), Rd(tmp_reg)
+			break;
+		case IR_GT:
+			|	seta Rb(def_reg)
+			break;
+		case IR_ULT:
+			|	setb Rb(def_reg)
+			break;
+		case IR_UGE:
+			|	setp Rb(def_reg)
+			|	mov Rd(tmp_reg), 1
+			|	cmovae Rd(def_reg), Rd(tmp_reg)
+			break;
+		case IR_ULE:
+			|	setbe Rb(def_reg)
+			break;
+		case IR_UGT:
+			|	setp Rb(def_reg)
+			|	mov Rd(tmp_reg), 1
+			|	cmova Rd(def_reg), Rd(tmp_reg)
+			break;
+		case IR_ORDERED:
+			|	setnp Rb(def_reg)
+			break;
+		case IR_UNORDERED:
+			|	setp Rb(def_reg)
+			break;
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_jmp_true(ir_ctx *ctx, uint32_t b, ir_ref def, uint32_t next_block)
+{
+	uint32_t true_block, false_block;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+	if (true_block != next_block) {
+		|	jmp =>true_block
+	}
+}
+
+static void ir_emit_jmp_false(ir_ctx *ctx, uint32_t b, ir_ref def, uint32_t next_block)
+{
+	uint32_t true_block, false_block;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+	if (false_block != next_block) {
+		|	jmp =>false_block
+	}
+}
+
+static void ir_emit_jcc(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block, uint8_t op, bool int_cmp, bool after_op)
+{
+	uint32_t true_block, false_block;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+	if (true_block == next_block) {
+		/* swap to avoid unconditional JMP */
+		if (int_cmp || op == IR_EQ || op == IR_NE || op == IR_ORDERED || op == IR_UNORDERED) {
+			op ^= 1; // reverse
+		} else {
+			op ^= 5; // reverse
+		}
+		true_block = false_block;
+		false_block = 0;
+	} else if (false_block == next_block) {
+		false_block = 0;
+	}
+
+	if (int_cmp) {
+		switch (op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_EQ:
+				|	je =>true_block
+				break;
+			case IR_NE:
+				|	jne =>true_block
+				break;
+			case IR_LT:
+				if (after_op) {
+					|	js =>true_block
+				} else {
+					|	jl =>true_block
+				}
+				break;
+			case IR_GE:
+				if (after_op) {
+					|	jns =>true_block
+				} else {
+					|	jge =>true_block
+				}
+				break;
+			case IR_LE:
+				|	jle =>true_block
+				break;
+			case IR_GT:
+				|	jg =>true_block
+				break;
+			case IR_ULT:
+				|	jb =>true_block
+				break;
+			case IR_UGE:
+				|	jae =>true_block
+				break;
+			case IR_ULE:
+				|	jbe =>true_block
+				break;
+			case IR_UGT:
+				|	ja =>true_block
+				break;
+		}
+	} else {
+		switch (op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_EQ:
+				if (!false_block) {
+					|	jp >1
+					|	je =>true_block
+					|1:
+				} else {
+					|	jp =>false_block
+					|	je =>true_block
+				}
+				break;
+			case IR_NE:
+				|	jne =>true_block
+				|	jp =>true_block
+				break;
+			case IR_LT:
+				if (!false_block) {
+					|	jp >1
+					|	jb =>true_block
+					|1:
+				} else {
+					|	jp =>false_block
+					|	jb =>true_block
+				}
+				break;
+			case IR_GE:
+				|	jae =>true_block
+				break;
+			case IR_LE:
+				if (!false_block) {
+					|	jp >1
+					|	jbe =>true_block
+					|1:
+				} else {
+					|	jp =>false_block
+					|	jbe =>true_block
+				}
+				break;
+			case IR_GT:
+				|	ja =>true_block
+				break;
+			case IR_ULT:
+				|	jb =>true_block
+				break;
+			case IR_UGE:
+				|	jp =>true_block
+				|	jae =>true_block
+				break;
+			case IR_ULE:
+				|	jbe =>true_block
+				break;
+			case IR_UGT:
+				|	jp =>true_block
+				|	ja =>true_block
+				break;
+			case IR_ORDERED:
+				|	jnp =>true_block
+				break;
+			case IR_UNORDERED:
+				|	jp =>true_block
+				break;
+		}
+	}
+	if (false_block) {
+		|	jmp =>false_block
+	}
+}
+
+static void ir_emit_cmp_and_branch_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op2];
+	ir_op op = cmp_insn->op;
+	ir_type type = ctx->ir_base[cmp_insn->op1].type;
+	ir_ref op1 = cmp_insn->op1;
+	ir_ref op2 = cmp_insn->op2;
+	ir_reg op1_reg, op2_reg;
+
+	if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[insn->op2][1];
+		op2_reg = ctx->regs[insn->op2][2];
+	}
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		if (op1 != op2) {
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+	}
+	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
+		if (op == IR_ULT) {
+			/* always false */
+			ir_emit_jmp_false(ctx, b, def, next_block);
+			return;
+		} else if (op == IR_UGE) {
+			/* always true */
+			ir_emit_jmp_true(ctx, b, def, next_block);
+			return;
+		} else if (op == IR_ULE) {
+			op = IR_EQ;
+		} else if (op == IR_UGT) {
+			op = IR_NE;
+		}
+	}
+
+	bool same_comparison = 0;
+	ir_insn *prev_insn = &ctx->ir_base[insn->op1];
+	if (prev_insn->op == IR_IF_TRUE || prev_insn->op == IR_IF_FALSE) {
+		if (ir_rule(ctx, prev_insn->op1) == IR_CMP_AND_BRANCH_INT) {
+			prev_insn = &ctx->ir_base[prev_insn->op1];
+			prev_insn = &ctx->ir_base[prev_insn->op2];
+			if (prev_insn->op1 == cmp_insn->op1 && prev_insn->op2 == cmp_insn->op2) {
+				same_comparison = true;
+			}
+		}
+	}
+	if (!same_comparison) {
+		ir_emit_cmp_int_common(ctx, type, def, cmp_insn, op1_reg, op1, op2_reg, op2);
+	}
+	ir_emit_jcc(ctx, b, def, insn, next_block, op, 1, 0);
+}
+
+static void ir_emit_test_and_branch_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_ref op2 = insn->op2;
+	ir_op op = ctx->ir_base[op2].op;
+
+	if (op >= IR_EQ && op <= IR_UGT) {
+		op2 = ctx->ir_base[op2].op1;
+	} else {
+		IR_ASSERT(op == IR_AND);
+		op = IR_NE;
+	}
+
+	ir_emit_test_int_common(ctx, def, op2, op);
+	ir_emit_jcc(ctx, b, def, insn, next_block, op, 1, 0);
+}
+
+static void ir_emit_test_and_branch_bit(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_ref op2 = insn->op2;
+	ir_op op = ctx->ir_base[op2].op;
+	uint32_t true_block, false_block;
+
+	if (op == IR_EQ || op == IR_NE) {
+		op2 = ctx->ir_base[op2].op1;
+	} else {
+		IR_ASSERT(op == IR_AND);
+		op = IR_NE;
+	}
+
+	ir_emit_test_bit_common(ctx, def, op2);
+
+	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+	if (true_block == next_block) {
+		op ^= 1; // reverse
+		true_block = false_block;
+		false_block = 0;
+	} else if (false_block == next_block) {
+		false_block = 0;
+	}
+
+	if (op == IR_EQ) {
+		|	jnc =>true_block
+	} else {
+		IR_ASSERT(op == IR_NE);
+		|	jc =>true_block
+	}
+	if (false_block) {
+		|	jmp =>false_block
+	}
+}
+
+static void ir_emit_cmp_and_branch_fp(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_op op = ir_emit_cmp_fp_common(ctx, def, insn->op2, &ctx->ir_base[insn->op2]);
+	ir_emit_jcc(ctx, b, def, insn, next_block, op, 0, 0);
+}
+
+static void ir_emit_if_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_type type = ctx->ir_base[insn->op2].type;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
+		}
+		|	ASM_REG_REG_OP test, type, op2_reg, op2_reg
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		uint32_t true_block, false_block;
+
+		ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+		if (ir_const_is_true(&ctx->ir_base[insn->op2])) {
+			if (true_block != next_block) {
+				|	jmp =>true_block
+			}
+		} else {
+			if (false_block != next_block) {
+				|	jmp =>false_block
+			}
+		}
+		return;
+	} else if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
+		uint32_t true_block, false_block;
+
+		ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+		if (true_block != next_block) {
+			|	jmp =>true_block
+		}
+		return;
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, insn->op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op2);
+		}
+		|	ASM_MEM_IMM_OP cmp, type, mem, 0
+	}
+	ir_emit_jcc(ctx, b, def, insn, next_block, IR_NE, 1, 0);
+}
+
+static void ir_emit_cond(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_type op1_type = ctx->ir_base[op1].type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+		if (op1 == op2) {
+			op1_reg = op2_reg;
+		}
+		if (op3 == op2) {
+			op3_reg = op2_reg;
+		}
+	}
+	if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, op3);
+		if (op1 == op3) {
+			op1_reg = op2_reg;
+		}
+	}
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, op1_type, op1_reg, op1);
+	}
+
+	if (IR_IS_TYPE_INT(op1_type)) {
+		if (op1_reg != IR_REG_NONE) {
+			|	ASM_REG_REG_OP test, op1_type, op1_reg, op1_reg
+		} else {
+			ir_mem mem;
+
+			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, insn->op1);
+			} else {
+				mem = ir_ref_spill_slot(ctx, insn->op1);
+			}
+
+			|	ASM_MEM_IMM_OP cmp, op1_type, mem, 0
+		}
+		if (IR_IS_TYPE_INT(type)) {
+			IR_ASSERT(op2_reg != IR_REG_NONE || op3_reg != IR_REG_NONE);
+			if (op3_reg != IR_REG_NONE) {
+				if (op3_reg == def_reg) {
+					IR_ASSERT(op2_reg != IR_REG_NONE);
+					|	ASM_REG_REG_OP2 cmovne, type, def_reg, op2_reg
+				} else {
+					if (op2_reg != IR_REG_NONE) {
+						if (def_reg != op2_reg) {
+							if (IR_IS_TYPE_INT(type)) {
+								ir_emit_mov(ctx, type, def_reg, op2_reg);
+							} else {
+								ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
 							}
 						}
-						/* c = CMP(_, _) ... IF(c) => SKIP_CMP ... CMP_AND_BRANCH */
-						if (ctx->use_lists[insn->op2].count == 1) {
-							ir_match_fuse_load_cmp_int(ctx, op2_insn, ref);
-						}
-						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
-						return IR_CMP_AND_BRANCH_INT;
+					} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
+						/* prevent "xor" and flags clobbering */
+						ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
 					} else {
-						/* c = CMP(_, _) ... IF(c) => SKIP_CMP ... CMP_AND_BRANCH */
-						if (ctx->use_lists[insn->op2].count == 1) {
-							ir_match_fuse_load_cmp_fp_br(ctx, op2_insn, ref);
-						}
-						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_FP;
-						return IR_CMP_AND_BRANCH_FP;
+						ir_emit_load_ex(ctx, type, def_reg, op2, def);
 					}
-				} else if (op2_insn->op == IR_AND) {
-					/* c = AND(_, _) ... IF(c) => SKIP_TEST ... TEST_AND_BRANCH */
-					ir_match_fuse_load_test_int(ctx, op2_insn, ref);
-					ctx->rules[insn->op2] = IR_FUSED | IR_TEST_INT;
-					return IR_TEST_AND_BRANCH_INT;
-				} else if (op2_insn->op == IR_OVERFLOW && ir_in_same_block(ctx, insn->op2)) {
-					/* c = OVERFLOW(_) ... IF(c) => SKIP_OVERFLOW ... OVERFLOW_AND_BRANCH */
-					ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_OVERFLOW;
-					return IR_OVERFLOW_AND_BRANCH;
+					|	ASM_REG_REG_OP2 cmove, type, def_reg, op3_reg
+				}
+			} else {
+				IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
+				if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
+					/* prevent "xor" and flags clobbering */
+					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
+				} else {
+					ir_emit_load_ex(ctx, type, def_reg, op3, def);
 				}
+				|	ASM_REG_REG_OP2 cmovne, type, def_reg, op2_reg
 			}
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
-				if (insn->op2 == ref - 1) { /* previous instruction */
-					op2_insn = &ctx->ir_base[insn->op2];
-					if (op2_insn->op == IR_ADD ||
-					    op2_insn->op == IR_SUB ||
-//					    op2_insn->op == IR_MUL ||
-					    op2_insn->op == IR_OR  ||
-					    op2_insn->op == IR_AND ||
-					    op2_insn->op == IR_XOR) {

-							/* v = BINOP(_, _); IF(v) => BINOP; JCC */
-						if (ir_op_flags[op2_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
-							ir_match_fuse_load_commutative_int(ctx, op2_insn, ref);
-							ctx->rules[insn->op2] = IR_BINOP_INT | IR_MAY_SWAP;
-						} else {
-							ir_match_fuse_load(ctx, op2_insn->op2, ref);
-							ctx->rules[insn->op2] = IR_BINOP_INT;
-						}
-						return IR_JCC_INT;
-					}
-				} else if ((ctx->flags & IR_OPT_CODEGEN)
-				 && insn->op1 == ref - 1 /* previous instruction */
-				 && insn->op2 == ref - 2 /* previous instruction */
-				 && ctx->use_lists[insn->op2].count == 2
-				 && IR_IS_TYPE_INT(ctx->ir_base[insn->op2].type)) {
-					ir_insn *store_insn = &ctx->ir_base[insn->op1];
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store(ctx, type, def, def_reg);
+			}
+			return;
+		}
+		|	je >2
+	} else {
+		if (!data->double_zero_const) {
+			data->double_zero_const = 1;
+			ir_rodata(ctx);
+			|.align 16
+			|->double_zero_const:
+			|.dword 0, 0
+			|.code
+		}
+		|	ASM_FP_REG_TXT_OP ucomis, op1_type, op1_reg, [->double_zero_const]
+		|	jp >1
+		|	je >2
+		|1:
+	}

-					if (store_insn->op == IR_STORE && store_insn->op3 == insn->op2) {
-						ir_insn *op_insn = &ctx->ir_base[insn->op2];
+	if (op2_reg != IR_REG_NONE) {
+		if (def_reg != op2_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, op2_reg);
+			} else {
+				ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+			}
+		}
+	} else {
+		ir_emit_load_ex(ctx, type, def_reg, op2, def);
+	}
+	|	jmp >3
+	|2:
+	if (op3_reg != IR_REG_NONE) {
+		if (def_reg != op3_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, op3_reg);
+			} else {
+				ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
+			}
+		}
+	} else {
+		ir_emit_load_ex(ctx, type, def_reg, op3, def);
+	}
+	|3:

-						if (op_insn->op == IR_ADD ||
-						    op_insn->op == IR_SUB ||
-//						    op_insn->op == IR_MUL ||
-						    op_insn->op == IR_OR  ||
-						    op_insn->op == IR_AND ||
-						    op_insn->op == IR_XOR) {
-							if (ctx->ir_base[op_insn->op1].op == IR_LOAD
-							 && ctx->ir_base[op_insn->op1].op2 == store_insn->op2) {
-								if (ir_in_same_block(ctx, op_insn->op1)
-								 && ctx->use_lists[op_insn->op1].count == 2
-								 && store_insn->op1 == op_insn->op1) {
-									/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; JCC */
-									ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
-									ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
-									ir_match_fuse_addr(ctx, store_insn->op2);
-									ctx->rules[insn->op1] = IR_MEM_BINOP_INT;
-									return IR_JCC_INT;
-								}
-							} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
-							 && ctx->ir_base[op_insn->op2].op == IR_LOAD
-							 && ctx->ir_base[op_insn->op2].op2 == store_insn->op2) {
-								if (ir_in_same_block(ctx, op_insn->op2)
-								 && ctx->use_lists[op_insn->op2].count == 2
-								 && store_insn->op1 == op_insn->op2) {
-									/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; JCC */
-									ir_swap_ops(op_insn);
-									ctx->rules[insn->op2] = IR_FUSED | IR_BINOP_INT;
-									ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
-									ir_match_fuse_addr(ctx, store_insn->op2);
-									ctx->rules[insn->op1] = IR_MEM_BINOP_INT;
-									return IR_JCC_INT;
-								}
-							}
-						}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}
+
+static void ir_emit_cond_test_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+
+	if (op2 != op3) {
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, op3);
+		}
+	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+		op3_reg = op2_reg;
+	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, op3);
+		op2_reg = op3_reg;
+	}
+
+	ir_emit_test_int_common(ctx, def, insn->op1, IR_NE);
+
+	if (IR_IS_TYPE_INT(type)) {
+		bool eq = 0;
+
+		if (op3_reg != IR_REG_NONE) {
+			if (op3_reg == def_reg) {
+				IR_ASSERT(op2_reg != IR_REG_NONE);
+				op3_reg = op2_reg;
+				eq = 1; // reverse
+			} else {
+				if (op2_reg != IR_REG_NONE) {
+					if (def_reg != op2_reg) {
+//						if (IR_IS_TYPE_INT(type)) {
+							ir_emit_mov(ctx, type, def_reg, op2_reg);
+//						} else {
+//							ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+//						}
 					}
+				} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
+					/* prevent "xor" and flags clobbering */
+					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
+				} else {
+					ir_emit_load_ex(ctx, type, def_reg, op2, def);
 				}
-				ir_match_fuse_load(ctx, insn->op2, ref);
-				return IR_IF_INT;
+			}
+		} else {
+			IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
+			if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
+				/* prevent "xor" and flags clobbering */
+				ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
 			} else {
-				IR_ASSERT(0 && "NIY IR_IF_FP");
-				break;
+				ir_emit_load_ex(ctx, type, def_reg, op3, def);
 			}
-		case IR_COND:
-			if (!IR_IS_CONST_REF(insn->op1) && (ctx->use_lists[insn->op1].count == 1 || all_usages_are_fusable(ctx, insn->op1))) {
-				ir_insn *op1_insn = &ctx->ir_base[insn->op1];
+			op3_reg = op2_reg;
+			eq = 1; // reverse
+		}

-				if (op1_insn->op >= IR_EQ && op1_insn->op <= IR_UNORDERED) {
-					if (IR_IS_TYPE_INT(ctx->ir_base[op1_insn->op1].type)) {
-						if (ctx->use_lists[insn->op1].count == 1) {
-							ir_match_fuse_load_cmp_int(ctx, op1_insn, ref);
-						}
-						ctx->rules[insn->op1] = IR_FUSED | IR_CMP_INT;
-						return IR_COND_CMP_INT;
-					} else {
-						if (ctx->use_lists[insn->op1].count == 1) {
-							ir_match_fuse_load_cmp_fp_br(ctx, op1_insn, ref);
-						}
-						ctx->rules[insn->op1] = IR_FUSED | IR_CMP_FP;
-						return IR_COND_CMP_FP;
-					}
-				} else if (op1_insn->op == IR_AND) {
-					/* c = AND(_, _) ... IF(c) => SKIP_TEST ... TEST_AND_BRANCH */
-					ir_match_fuse_load_test_int(ctx, op1_insn, ref);
-					ctx->rules[insn->op1] = IR_FUSED | IR_TEST_INT;
-					return IR_COND_TEST_INT;
+		if (eq) {
+			|	ASM_REG_REG_OP2 cmovne, type, def_reg, op3_reg
+		} else {
+			|	ASM_REG_REG_OP2 cmove, type, def_reg, op3_reg
+		}
+	} else {
+		|	jne >2
+		|1:
+
+		if (op2_reg != IR_REG_NONE) {
+			if (def_reg != op2_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, op2_reg);
+				} else {
+					ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
 				}
 			}
-			if (IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
-				ir_match_fuse_load(ctx, insn->op1, ref);
+		} else {
+			ir_emit_load_ex(ctx, type, def_reg, op2, def);
+		}
+		|	jmp >3
+		|2:
+		if (op3_reg != IR_REG_NONE) {
+			if (def_reg != op3_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, op3_reg);
+				} else {
+					ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
+				}
 			}
-			return IR_COND;
-		case IR_GUARD:
-		case IR_GUARD_NOT:
-			if (!IR_IS_CONST_REF(insn->op2) && (ctx->use_lists[insn->op2].count == 1 || all_usages_are_fusable(ctx, insn->op2))) {
-				op2_insn = &ctx->ir_base[insn->op2];
-				if (op2_insn->op >= IR_EQ && op2_insn->op <= IR_UNORDERED) {
-					if (IR_IS_TYPE_INT(ctx->ir_base[op2_insn->op1].type)) {
-						if (IR_IS_CONST_REF(op2_insn->op2)
-						 && !IR_IS_SYM_CONST(ctx->ir_base[op2_insn->op2].op)
-						 && ctx->ir_base[op2_insn->op2].val.i64 == 0) {
-							if (op2_insn->op1 == insn->op2 - 1 /* previous instruction */
-							 && ir_in_same_block(ctx, op2_insn->op1)
-							 && !ir_match_has_flags_deps(ctx, insn->op2, ref)) {
-								ir_insn *op1_insn = &ctx->ir_base[op2_insn->op1];
+		} else {
+			ir_emit_load_ex(ctx, type, def_reg, op3, def);
+		}
+		|3:
+	}

-								if ((op1_insn->op == IR_OR || op1_insn->op == IR_AND || op1_insn->op == IR_XOR) ||
-										/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
-										((op1_insn->op == IR_ADD || op1_insn->op == IR_SUB) &&
-											(op2_insn->op == IR_EQ || op2_insn->op == IR_NE ||
-												op2_insn->op == IR_LT || op2_insn->op == IR_GE))) {
-									if (ir_op_flags[op1_insn->op] & IR_OP_FLAG_COMMUTATIVE) {
-										if (ctx->use_lists[insn->op2].count == 1) {
-											ir_match_fuse_load_commutative_int(ctx, op1_insn, ref);
-										}
-										ctx->rules[op2_insn->op1] = IR_BINOP_INT | IR_MAY_SWAP;
-									} else {
-										if (ctx->use_lists[insn->op2].count == 1) {
-											ir_match_fuse_load(ctx, op1_insn->op2, ref);
-										}
-										ctx->rules[op2_insn->op1] = IR_BINOP_INT;
-									}
-									/* v = BINOP(_, _); c = CMP(v, 0) ... IF(c) => BINOP; SKIP_CMP ... GUARD_JCC */
-									ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
-									return IR_GUARD_JCC_INT;
-								}
-							} else if ((ctx->flags & IR_OPT_CODEGEN)
-							 && ctx->use_lists[insn->op2].count == 1
-							 && op2_insn->op1 == insn->op2 - 2 /* before previous instruction */
-							 && ir_in_same_block(ctx, op2_insn->op1)
-							 && ctx->use_lists[op2_insn->op1].count == 2) {
-								ir_insn *store_insn = &ctx->ir_base[insn->op2 - 1];
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}

-								if (store_insn->op == IR_STORE && store_insn->op3 == op2_insn->op1) {
-									ir_insn *op_insn = &ctx->ir_base[op2_insn->op1];
+static void ir_emit_cond_test_bit(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];

-									if ((op_insn->op == IR_OR || op_insn->op == IR_AND || op_insn->op == IR_XOR) ||
-											/* GT(ADD(_, _), 0) can't be optimized because ADD may overflow */
-											((op_insn->op == IR_ADD || op_insn->op == IR_SUB) &&
-												(op2_insn->op == IR_EQ || op2_insn->op == IR_NE ||
-													op2_insn->op == IR_LT || op2_insn->op == IR_GE))) {
-										if (ctx->ir_base[op_insn->op1].op == IR_LOAD
-										 && ctx->ir_base[op_insn->op1].op2 == store_insn->op2) {
-											if (ir_in_same_block(ctx, op_insn->op1)
-											 && ctx->use_lists[op_insn->op1].count == 2
-											 && store_insn->op1 == op_insn->op1) {
-												/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; GUARD_JCC */
-												ctx->rules[op2_insn->op1] = IR_FUSED | IR_BINOP_INT;
-												ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
-												ir_match_fuse_addr(ctx, store_insn->op2);
-												ctx->rules[insn->op2 - 1] = IR_MEM_BINOP_INT;
-												ctx->rules[insn->op2] = IR_SKIPPED | IR_NOP;
-												return IR_GUARD_JCC_INT;
-											}
-										} else if ((ir_op_flags[op_insn->op] & IR_OP_FLAG_COMMUTATIVE)
-										 && ctx->ir_base[op_insn->op2].op == IR_LOAD
-										 && ctx->ir_base[op_insn->op2].op2 == store_insn->op2) {
-											if (ir_in_same_block(ctx, op_insn->op2)
-											 && ctx->use_lists[op_insn->op2].count == 2
-											 && store_insn->op1 == op_insn->op2) {
-												/* v = MEM_BINOP(_, _); IF(v) => MEM_BINOP; JCC */
-												ir_swap_ops(op_insn);
-												ctx->rules[op2_insn->op1] = IR_FUSED | IR_BINOP_INT;
-												ctx->rules[op_insn->op1] = IR_SKIPPED | IR_LOAD;
-												ir_match_fuse_addr(ctx, store_insn->op2);
-												ctx->rules[insn->op2 - 1] = IR_MEM_BINOP_INT;
-												ctx->rules[insn->op2] = IR_SKIPPED | IR_NOP;
-												return IR_GUARD_JCC_INT;
-											}
-										}
-									}
-								}
-							}
-						}
-						/* c = CMP(_, _) ... GUARD(c) => SKIP_CMP ... GUARD_CMP */
-						if (ctx->use_lists[insn->op2].count == 1) {
-							ir_match_fuse_load_cmp_int(ctx, op2_insn, ref);
-						}
-						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_INT;
-						return IR_GUARD_CMP_INT;
-					} else {
-						/* c = CMP(_, _) ... GUARD(c) => SKIP_CMP ... GUARD_CMP */
-						if (ctx->use_lists[insn->op2].count == 1) {
-							ir_match_fuse_load_cmp_fp_br(ctx, op2_insn, ref);
-						}
-						ctx->rules[insn->op2] = IR_FUSED | IR_CMP_FP;
-						return IR_GUARD_CMP_FP;
-					}
-				} else if (op2_insn->op == IR_AND) { // TODO: OR, XOR. etc
-					/* c = AND(_, _) ... GUARD(c) => SKIP_TEST ... GUARD_TEST */
-					ir_match_fuse_load_test_int(ctx, op2_insn, ref);
-					ctx->rules[insn->op2] = IR_FUSED | IR_TEST_INT;
-					return IR_GUARD_TEST_INT;
-				} else if (op2_insn->op == IR_OVERFLOW && ir_in_same_block(ctx, insn->op2)) {
-					/* c = OVERFLOW(_) ... GUARD(c) => SKIP_OVERFLOW ... GUARD_OVERFLOW */
-					ctx->rules[insn->op2] = IR_FUSED | IR_SIMPLE | IR_OVERFLOW;
-					return IR_GUARD_OVERFLOW;
-				}
-			}
-			ir_match_fuse_load(ctx, insn->op2, ref);
-			return insn->op;
-		case IR_INT2FP:
-			if (ir_type_size[ctx->ir_base[insn->op1].type] > (IR_IS_TYPE_SIGNED(ctx->ir_base[insn->op1].type) ? 2 : 4)) {
-				ir_match_fuse_load(ctx, insn->op1, ref);
-			}
-			return insn->op;
-		case IR_SEXT:
-		case IR_ZEXT:
-		case IR_FP2INT:
-		case IR_FP2FP:
-			ir_match_fuse_load(ctx, insn->op1, ref);
-			return insn->op;
-		case IR_TRUNC:
-		case IR_PROTO:
-			ir_match_fuse_load(ctx, insn->op1, ref);
-			return insn->op | IR_MAY_REUSE;
-		case IR_BITCAST:
-			ir_match_fuse_load(ctx, insn->op1, ref);
-			if (IR_IS_TYPE_INT(insn->type) && IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type)) {
-				return insn->op | IR_MAY_REUSE;
-			} else {
-				return insn->op;
-			}
-		case IR_CTLZ:
-		case IR_CTTZ:
-			ir_match_fuse_load(ctx, insn->op1, ref);
-			return IR_BIT_COUNT;
-		case IR_CTPOP:
-			ir_match_fuse_load(ctx, insn->op1, ref);
-			return (ctx->mflags & IR_X86_BMI1) ? IR_BIT_COUNT : IR_CTPOP;
-		case IR_VA_START:
-			ctx->flags2 |= IR_HAS_VA_START;
-			if ((ctx->ir_base[insn->op2].op == IR_ALLOCA) || (ctx->ir_base[insn->op2].op == IR_VADDR)) {
-				ir_use_list *use_list = &ctx->use_lists[insn->op2];
-				ir_ref *p, n = use_list->count;
-				for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
-					ir_insn *use_insn = &ctx->ir_base[*p];
-					if (use_insn->op == IR_VA_START || use_insn->op == IR_VA_END) {
-					} else if (use_insn->op == IR_VA_COPY) {
-						if (use_insn->op3 == insn->op2) {
-							ctx->flags2 |= IR_HAS_VA_COPY;
-						}
-					} else if (use_insn->op == IR_VA_ARG) {
-						if (use_insn->op2 == insn->op2) {
-							if (IR_IS_TYPE_INT(use_insn->type)) {
-								ctx->flags2 |= IR_HAS_VA_ARG_GP;
-							} else {
-								IR_ASSERT(IR_IS_TYPE_FP(use_insn->type));
-								ctx->flags2 |= IR_HAS_VA_ARG_FP;
-							}
-						}
-					} else if (*p > ref) {
-						/* diriect va_list access */
-						ctx->flags2 |= IR_HAS_VA_ARG_GP|IR_HAS_VA_ARG_FP;
+	if (op2 != op3) {
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, op3);
+		}
+	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+		op3_reg = op2_reg;
+	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, op3);
+		op2_reg = op3_reg;
+	}
+
+	ir_emit_test_bit_common(ctx, def, insn->op1);
+
+	if (IR_IS_TYPE_INT(type)) {
+		bool eq = 0;
+
+		if (op3_reg != IR_REG_NONE) {
+			if (op3_reg == def_reg) {
+				IR_ASSERT(op2_reg != IR_REG_NONE);
+				op3_reg = op2_reg;
+				eq = 1; // reverse
+			} else {
+				if (op2_reg != IR_REG_NONE) {
+					if (def_reg != op2_reg) {
+//						if (IR_IS_TYPE_INT(type)) {
+							ir_emit_mov(ctx, type, def_reg, op2_reg);
+//						} else {
+//							ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+//						}
 					}
+				} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
+					/* prevent "xor" and flags clobbering */
+					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
+				} else {
+					ir_emit_load_ex(ctx, type, def_reg, op2, def);
 				}
+			}
+		} else {
+			IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
+			if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
+				/* prevent "xor" and flags clobbering */
+				ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
 			} else {
-				/* va_list may escape */
-				ctx->flags2 |= IR_HAS_VA_ARG_GP|IR_HAS_VA_ARG_FP;
+				ir_emit_load_ex(ctx, type, def_reg, op3, def);
 			}
-			return IR_VA_START;
-		case IR_VA_END:
-			return IR_SKIPPED | IR_NOP;
-		case IR_VADDR:
-			if (ctx->use_lists[ref].count > 0) {
-				ir_use_list *use_list = &ctx->use_lists[ref];
-				ir_ref *p, n = use_list->count;
+			op3_reg = op2_reg;
+			eq = 1; // reverse
+		}

-				for (p = &ctx->use_edges[use_list->refs]; n > 0; p++, n--) {
-					if (ctx->ir_base[*p].op != IR_VA_END) {
-						return IR_STATIC_ALLOCA;
-					}
+		if (eq) {
+			|	ASM_REG_REG_OP2 cmovc, type, def_reg, op3_reg
+		} else {
+			|	ASM_REG_REG_OP2 cmovnc, type, def_reg, op3_reg
+		}
+	} else {
+		|	jc >2
+		|1:
+
+		if (op2_reg != IR_REG_NONE) {
+			if (def_reg != op2_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, op2_reg);
+				} else {
+					ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
 				}
 			}
-			return IR_SKIPPED | IR_NOP;
-		case IR_ARGVAL:
-			return IR_FUSED | IR_ARGVAL;
-		case IR_NOP:
-			return IR_SKIPPED | IR_NOP;
-		case IR_ASM:
-		case IR_ASM_OUT:
-		case IR_ASM_GOTO:
-			fprintf(stderr, "ERROR: IR_ASM is not implemented yet\n");
-			exit(1);
-			return IR_SKIPPED | IR_NOP;
-		default:
-			break;
+		} else {
+			ir_emit_load_ex(ctx, type, def_reg, op2, def);
+		}
+		|	jmp >3
+		|2:
+		if (op3_reg != IR_REG_NONE) {
+			if (def_reg != op3_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, op3_reg);
+				} else {
+					ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
+				}
+			}
+		} else {
+			ir_emit_load_ex(ctx, type, def_reg, op3, def);
+		}
+		|3:
 	}

-	return insn->op;
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
 }

-static void ir_match_insn2(ir_ctx *ctx, ir_ref ref, uint32_t rule)
+static void ir_emit_cond_cmp_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	if (rule == IR_LEA_IB) {
-		ir_match_try_revert_lea_to_add(ctx, ref);
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_op op;
+
+	if (op2 != op3) {
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, op3);
+		}
+	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+		op3_reg = op2_reg;
+	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, op3);
+		op2_reg = op3_reg;
+	}
+
+	ir_emit_cmp_int_common2(ctx, def, insn->op1, &ctx->ir_base[insn->op1]);
+	op = ctx->ir_base[insn->op1].op;
+
+	if (IR_IS_TYPE_INT(type)) {
+		if (op3_reg != IR_REG_NONE) {
+			if (op3_reg == def_reg) {
+				IR_ASSERT(op2_reg != IR_REG_NONE);
+				op3_reg = op2_reg;
+				op ^= 1; // reverse
+			} else {
+				if (op2_reg != IR_REG_NONE) {
+					if (def_reg != op2_reg) {
+						ir_emit_mov(ctx, type, def_reg, op2_reg);
+					}
+				} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
+					/* prevent "xor" and flags clobbering */
+					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
+				} else {
+					ir_emit_load_ex(ctx, type, def_reg, op2, def);
+				}
+			}
+		} else {
+			IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
+			if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
+				/* prevent "xor" and flags clobbering */
+				ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
+			} else {
+				ir_emit_load_ex(ctx, type, def_reg, op3, def);
+			}
+			op3_reg = op2_reg;
+			op ^= 1; // reverse
+		}
+
+		switch (op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_EQ:
+				|	ASM_REG_REG_OP2 cmovne, type, def_reg, op3_reg
+				break;
+			case IR_NE:
+				|	ASM_REG_REG_OP2 cmove, type, def_reg, op3_reg
+				break;
+			case IR_LT:
+				|	ASM_REG_REG_OP2 cmovge, type, def_reg, op3_reg
+				break;
+			case IR_GE:
+				|	ASM_REG_REG_OP2 cmovl, type, def_reg, op3_reg
+				break;
+			case IR_LE:
+				|	ASM_REG_REG_OP2 cmovg, type, def_reg, op3_reg
+				break;
+			case IR_GT:
+				|	ASM_REG_REG_OP2 cmovle, type, def_reg, op3_reg
+				break;
+			case IR_ULT:
+				|	ASM_REG_REG_OP2 cmovae, type, def_reg, op3_reg
+				break;
+			case IR_UGE:
+				|	ASM_REG_REG_OP2 cmovb, type, def_reg, op3_reg
+				break;
+			case IR_ULE:
+				|	ASM_REG_REG_OP2 cmova, type, def_reg, op3_reg
+				break;
+			case IR_UGT:
+				|	ASM_REG_REG_OP2 cmovbe, type, def_reg, op3_reg
+				break;
+		}
+	} else {
+		switch (op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_EQ:
+				|	jne >2
+				break;
+			case IR_NE:
+				|	je >2
+				break;
+			case IR_LT:
+				|	jge >2
+				break;
+			case IR_GE:
+				|	jl >2
+				break;
+			case IR_LE:
+				|	jg >2
+				break;
+			case IR_GT:
+				|	jle >2
+				break;
+			case IR_ULT:
+				|	jae >2
+				break;
+			case IR_UGE:
+				|	jb >2
+				break;
+			case IR_ULE:
+				|	ja >2
+				break;
+			case IR_UGT:
+				|	jbe >2
+				break;
+		}
+		|1:
+
+		if (op2_reg != IR_REG_NONE) {
+			if (def_reg != op2_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, op2_reg);
+				} else {
+					ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+				}
+			}
+		} else {
+			ir_emit_load_ex(ctx, type, def_reg, op2, def);
+		}
+		|	jmp >3
+		|2:
+		if (op3_reg != IR_REG_NONE) {
+			if (def_reg != op3_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, op3_reg);
+				} else {
+					ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
+				}
+			}
+		} else {
+			ir_emit_load_ex(ctx, type, def_reg, op3, def);
+		}
+		|3:
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
 	}
 }

-/* code generation */
-static int32_t ir_ref_spill_slot_offset(ir_ctx *ctx, ir_ref ref, ir_reg *reg)
+static void ir_emit_cond_cmp_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	int32_t offset;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_op op;

-	IR_ASSERT(ref >= 0 && ctx->vregs[ref] && ctx->live_intervals[ctx->vregs[ref]]);
-	offset = ctx->live_intervals[ctx->vregs[ref]]->stack_spill_pos;
-	IR_ASSERT(offset != -1);
-	if (ctx->live_intervals[ctx->vregs[ref]]->flags & IR_LIVE_INTERVAL_SPILL_SPECIAL) {
-		IR_ASSERT(ctx->spill_base != IR_REG_NONE);
-		*reg = ctx->spill_base;
-		return offset;
+	if (op2 != op3) {
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, op3);
+		}
+	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+		op3_reg = op2_reg;
+	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, op3);
+		op2_reg = op3_reg;
 	}
-	*reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	return IR_SPILL_POS_TO_OFFSET(offset);
-}

-static ir_mem ir_vreg_spill_slot(ir_ctx *ctx, ir_ref v)
-{
-	int32_t offset;
-	ir_reg base;
+	op = ir_emit_cmp_fp_common(ctx, def, insn->op1, &ctx->ir_base[insn->op1]);

-	IR_ASSERT(v > 0 && v <= ctx->vregs_count && ctx->live_intervals[v]);
-	offset = ctx->live_intervals[v]->stack_spill_pos;
-	IR_ASSERT(offset != -1);
-	if (ctx->live_intervals[v]->flags & IR_LIVE_INTERVAL_SPILL_SPECIAL) {
-		IR_ASSERT(ctx->spill_base != IR_REG_NONE);
-		return IR_MEM_BO(ctx->spill_base, offset);
+	switch (op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	jne >2
+			|	jp >2
+			break;
+		case IR_NE:
+			|	jp >1
+			|	je >2
+			break;
+		case IR_LT:
+			|	jp >2
+			|	jae >2
+			break;
+		case IR_GE:
+			|	jb >2
+			break;
+		case IR_LE:
+			|	jp >2
+			|	ja >2
+			break;
+		case IR_GT:
+			|	jbe >2
+			break;
+		case IR_ULT:
+			|	jae >2
+			break;
+		case IR_UGE:
+			|	jp >1
+			|	jb >2
+			break;
+		case IR_ULE:
+			|	ja >2
+			break;
+		case IR_UGT:
+			|	jp >1
+			|	jbe >2
+			break;
+		case IR_ORDERED:
+			|	jp >2
+			break;
+		case IR_UNORDERED:
+			|	jnp >2
+			break;
 	}
-	base = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	offset = IR_SPILL_POS_TO_OFFSET(offset);
-	return IR_MEM_BO(base, offset);
-}
+	|1:

-static ir_mem ir_ref_spill_slot(ir_ctx *ctx, ir_ref ref)
-{
-	IR_ASSERT(!IR_IS_CONST_REF(ref));
-	return ir_vreg_spill_slot(ctx, ctx->vregs[ref]);
-}
+	if (op2_reg != IR_REG_NONE) {
+		if (def_reg != op2_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, op2_reg);
+			} else {
+				ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+			}
+		}
+	} else {
+		ir_emit_load_ex(ctx, type, def_reg, op2, def);
+	}
+	|	jmp >3
+	|2:
+	if (op3_reg != IR_REG_NONE) {
+		if (def_reg != op3_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, op3_reg);
+			} else {
+				ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
+			}
+		}
+	} else {
+		ir_emit_load_ex(ctx, type, def_reg, op3, def);
+	}
+	|3:

-static bool ir_is_same_spill_slot(ir_ctx *ctx, ir_ref ref, ir_mem mem)
-{
-	ir_mem m = ir_ref_spill_slot(ctx, ref);
-	return IR_MEM_VAL(m) == IR_MEM_VAL(mem);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
 }

-static ir_mem ir_var_spill_slot(ir_ctx *ctx, ir_ref ref)
+static void ir_emit_return_void(ir_ctx *ctx)
 {
-	ir_insn *var_insn = &ctx->ir_base[ref];
-	ir_reg reg;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;

-	IR_ASSERT(var_insn->op == IR_VAR);
-	reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	return IR_MEM_BO(reg, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
+	ir_emit_epilogue(ctx);
+
+	if (data->ra_data.cc->cleanup_stack_by_callee && ctx->param_stack_size) {
+		|	ret ctx->param_stack_size
+	} else {
+		|	ret
+	}
 }

-static bool ir_may_avoid_spill_load(ir_ctx *ctx, ir_ref ref, ir_ref use)
+static void ir_emit_return_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
-	ir_live_interval *ival;
+	ir_backend_data *data = ctx->data;
+	ir_reg ret_reg = data->ra_data.cc->int_ret_reg;
+	ir_reg op2_reg = ctx->regs[ref][2];

-	IR_ASSERT(ctx->vregs[ref] && ctx->live_intervals[ctx->vregs[ref]]);
-	ival = ctx->live_intervals[ctx->vregs[ref]];
-	while (ival) {
-		ir_use_pos *use_pos = ival->use_pos;
-		while (use_pos) {
-			if (IR_LIVE_POS_TO_REF(use_pos->pos) == use) {
-				return !use_pos->next || use_pos->next->op_num == 0;
-			}
-			use_pos = use_pos->next;
+	if (op2_reg != ret_reg) {
+		ir_type type = ctx->ir_base[insn->op2].type;
+
+		if (op2_reg != IR_REG_NONE && !IR_REG_SPILLED(op2_reg)) {
+			ir_emit_mov(ctx, type, ret_reg, op2_reg);
+		} else {
+			ir_emit_load(ctx, type, ret_reg, insn->op2);
 		}
-		ival = ival->next;
 	}
-	return 0;
+	ir_emit_return_void(ctx);
 }

-static void ir_emit_mov_imm_int(ir_ctx *ctx, ir_type type, ir_reg reg, int64_t val)
+static void ir_emit_return_fp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	ir_reg op2_reg = ctx->regs[ref][2];
+	ir_type type = ctx->ir_base[insn->op2].type;
+	ir_reg ret_reg = data->ra_data.cc->fp_ret_reg;

-	if (ir_type_size[type] == 8) {
-		IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-		if (IR_IS_UNSIGNED_32BIT(val)) {
-			|	mov Rd(reg), (uint32_t)val // zero extended load
-		} else if (IR_IS_SIGNED_32BIT(val)) {
-			|	mov Rq(reg), (int32_t)val // sign extended load
-		} else if (type == IR_ADDR && IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, (intptr_t)val)) {
-			|	lea Ra(reg), [&val]
+#if IR_SIMD && defined(IR_TARGET_X86)
+	if (IR_IS_TYPE_VECTOR(type)) {
+		ret_reg = data->ra_data.cc->vector_ret_reg;
+	}
+#endif
+
+	if (op2_reg != ret_reg && ret_reg != IR_REG_NONE) {
+		if (op2_reg != IR_REG_NONE && !IR_REG_SPILLED(op2_reg)) {
+			ir_emit_fp_mov(ctx, type, ret_reg, op2_reg);
 		} else {
-			|	mov64 Ra(reg), val
+			ir_emit_load(ctx, type, ret_reg, insn->op2);
 		}
-|.endif
-	} else {
-		|	ASM_REG_IMM_OP mov, type, reg, (int32_t)val // sign extended load
 	}
-}

-static void ir_emit_load_imm_int(ir_ctx *ctx, ir_type type, ir_reg reg, int64_t val)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+#ifdef IR_TARGET_X86
+	if (ret_reg == IR_REG_NONE) {
+		dasm_State **Dst = &data->dasm_state;

-	IR_ASSERT(IR_IS_TYPE_INT(type));
-	if (val == 0) {
-		|	ASM_REG_REG_OP xor, type, reg, reg
-	} else {
-		ir_emit_mov_imm_int(ctx, type, reg, val);
+		if (IR_IS_CONST_REF(insn->op2)) {
+			ir_insn *value = &ctx->ir_base[insn->op2];
+
+			if ((type == IR_FLOAT && value->val.f == 0.0) || (type == IR_DOUBLE && value->val.d == 0.0)) {
+				|	fldz
+			} else if ((type == IR_FLOAT && value->val.f == 1.0) || (type == IR_DOUBLE && value->val.d == 1.0)) {
+				|	fld1
+			} else {
+				int label = ir_get_const_label(ctx, insn->op2);
+
+				if (type == IR_DOUBLE) {
+					|	fld qword [=>label]
+				} else {
+					IR_ASSERT(type == IR_FLOAT);
+					|	fld dword [=>label]
+				}
+			}
+		} else if (op2_reg == IR_REG_NONE || IR_REG_SPILLED(op2_reg)) {
+			ir_reg fp;
+			int32_t offset = ir_ref_spill_slot_offset(ctx, insn->op2, &fp);
+
+			if (type == IR_DOUBLE) {
+				|	fld qword [Ra(fp)+offset]
+			} else {
+				IR_ASSERT(type == IR_FLOAT);
+				|	fld dword [Ra(fp)+offset]
+			}
+		} else {
+			int32_t offset = ctx->ret_slot;
+			ir_reg fp;
+
+			IR_ASSERT(offset != -1);
+			offset = IR_SPILL_POS_TO_OFFSET(offset);
+			fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), op2_reg);
+			if (type == IR_DOUBLE) {
+				|	fld qword [Ra(fp)+offset]
+			} else {
+				IR_ASSERT(type == IR_FLOAT);
+				|	fld dword [Ra(fp)+offset]
+			}
+		}
 	}
-}
-
-static void ir_emit_load_mem_int(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem mem)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+#endif

-	|	ASM_REG_MEM_OP mov, type, reg, mem
+	ir_emit_return_void(ctx);
 }

-static void ir_emit_load_imm_fp(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref src)
+static void ir_emit_sext_common(ir_ctx *ctx, ir_ref def, ir_insn *insn, ir_type dst_type, ir_reg def_reg)
 {
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *insn = &ctx->ir_base[src];
-	int label;
+	ir_reg op1_reg = ctx->regs[def][1];

-	if (type == IR_FLOAT && insn->val.u32 == 0) {
-		if (ctx->mflags & IR_X86_AVX) {
-			|	vxorps xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+	IR_ASSERT(IR_IS_TYPE_INT(src_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
+	IR_ASSERT(ir_type_size[dst_type] > ir_type_size[src_type]);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (ir_type_size[src_type] == 1) {
+			if (ir_type_size[dst_type] == 2) {
+				if (def_reg == IR_REG_RAX && op1_reg == IR_REG_RAX) {
+					|	cbw
+				} else {
+					|	movsx Rw(def_reg), Rb(op1_reg)
+				}
+			} else if (ir_type_size[dst_type] == 4) {
+				|	movsx Rd(def_reg), Rb(op1_reg)
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	movsx Rq(def_reg), Rb(op1_reg)
+|.endif
+			}
+		} else if (ir_type_size[src_type] == 2) {
+			if (ir_type_size[dst_type] == 4) {
+				if (def_reg == IR_REG_RAX && op1_reg == IR_REG_RAX) {
+					|	cwde
+				} else {
+					|	movsx Rd(def_reg), Rw(op1_reg)
+				}
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	movsx Rq(def_reg), Rw(op1_reg)
+|.endif
+			}
 		} else {
-			|	xorps xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+			IR_ASSERT(ir_type_size[src_type] == 4);
+			IR_ASSERT(ir_type_size[dst_type] == 8);
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			if (def_reg == IR_REG_RAX && op1_reg == IR_REG_RAX) {
+				|	cdqe
+			} else {
+				|	movsxd Rq(def_reg), Rd(op1_reg)
+			}
+|.endif
 		}
-	} else if (type == IR_DOUBLE && insn->val.u64 == 0) {
-		if (ctx->mflags & IR_X86_AVX) {
-			|	vxorpd xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		int64_t val;
+
+		if (ir_type_size[src_type] == 1) {
+			val = ctx->ir_base[insn->op1].val.i8;
+		} else if (ir_type_size[src_type] == 2) {
+			val = ctx->ir_base[insn->op1].val.i16;
+		} else if (ir_type_size[src_type] == 4) {
+			val = ctx->ir_base[insn->op1].val.i32;
 		} else {
-			|	xorpd xmm(reg-IR_REG_FP_FIRST), xmm(reg-IR_REG_FP_FIRST)
+			IR_ASSERT(ir_type_size[src_type] == 8);
+			val = ctx->ir_base[insn->op1].val.i64;
 		}
+		ir_emit_mov_imm_int(ctx, dst_type, def_reg, val);
 	} else {
-		label = ir_get_const_label(ctx, src);
-		|	ASM_FP_REG_TXT_OP movs, type, reg, [=>label]
-	}
-}
-
-static void ir_emit_load_mem_fp(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem mem)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+		ir_mem mem;

-	|	ASM_FP_REG_MEM_OP movs, type, reg, mem
-}
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}

-static void ir_emit_load_mem(ir_ctx *ctx, ir_type type, ir_reg reg, ir_mem mem)
-{
-	if (IR_IS_TYPE_INT(type)) {
-		ir_emit_load_mem_int(ctx, type, reg, mem);
-	} else {
-		ir_emit_load_mem_fp(ctx, type, reg, mem);
+		if (ir_type_size[src_type] == 1) {
+			if (ir_type_size[dst_type] == 2) {
+				|	ASM_TXT_TMEM_OP movsx, Rw(def_reg), byte, mem
+			} else if (ir_type_size[dst_type] == 4) {
+				|	ASM_TXT_TMEM_OP movsx, Rd(def_reg), byte, mem
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	ASM_TXT_TMEM_OP movsx, Rq(def_reg), byte, mem
+|.endif
+			}
+		} else if (ir_type_size[src_type] == 2) {
+			if (ir_type_size[dst_type] == 4) {
+				|	ASM_TXT_TMEM_OP movsx, Rd(def_reg), word, mem
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	ASM_TXT_TMEM_OP movsx, Rq(def_reg), word, mem
+|.endif
+			}
+		} else {
+			IR_ASSERT(ir_type_size[src_type] == 4);
+			IR_ASSERT(ir_type_size[dst_type] == 8);
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			|	ASM_TXT_TMEM_OP movsxd, Rq(def_reg), dword, mem
+|.endif
+		}
 	}
 }

-static int32_t ir_local_offset(ir_ctx *ctx, ir_insn *insn)
+static void ir_emit_sext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	if (insn->op != IR_PARAM) {
-		IR_ASSERT(insn->op == IR_VAR || insn->op == IR_ALLOCA || insn->op == IR_VADDR);
-		return IR_SPILL_POS_TO_OFFSET(insn->op3);
-	} else {
-		IR_ASSERT(ctx->value_params && ctx->value_params[insn->op3 - 1].align);
-		return IR_SPILL_POS_TO_OFFSET(ctx->value_params[insn->op3 - 1].offset);
+	ir_emit_sext_common(ctx, def, insn, insn->type, IR_REG_NUM(ctx->regs[def][0]));
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, IR_REG_NUM(ctx->regs[def][0]));
 	}
 }

-static void ir_load_local_addr(ir_ctx *ctx, ir_reg reg, ir_ref src)
+static void ir_emit_zext_common(ir_ctx *ctx, ir_ref def, ir_insn *insn, ir_type dst_type, ir_type src_type, uint8_t src_size, ir_reg def_reg, bool force)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg base = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	ir_insn *var_insn;
-	int32_t offset;
+	ir_reg op1_reg = ctx->regs[def][1];

-	IR_ASSERT(ir_rule(ctx, src) == IR_STATIC_ALLOCA);
-	var_insn = &ctx->ir_base[src];
-	if (var_insn->op == IR_VADDR) {
-		var_insn = &ctx->ir_base[var_insn->op1];
-	}
-	offset = ir_local_offset(ctx, var_insn);
-	if (offset == 0) {
-		| mov Ra(reg), Ra(base)
-	} else {
-		| lea Ra(reg), [Ra(base)+offset]
-	}
-}
+	IR_ASSERT(IR_IS_TYPE_INT(src_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
+	IR_ASSERT(ir_type_size[dst_type] > src_size);
+	IR_ASSERT(def_reg != IR_REG_NONE);

-static void ir_resolve_label_syms(ir_ctx *ctx)
-{
-	uint32_t b;
-	ir_block *bb;
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (src_size == 1) {
+			if (ir_type_size[dst_type] == 2) {
+				|	movzx Rw(def_reg), Rb(op1_reg)
+			} else if (ir_type_size[dst_type] == 4) {
+				|	movzx Rd(def_reg), Rb(op1_reg)
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	movzx Rq(def_reg), Rb(op1_reg)
+|.endif
+			}
+		} else if (src_size == 2) {
+			if (ir_type_size[dst_type] == 4) {
+				|	movzx Rd(def_reg), Rw(op1_reg)
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	movzx Rq(def_reg), Rw(op1_reg)
+|.endif
+			}
+		} else {
+			IR_ASSERT(src_size == 4);
+			IR_ASSERT(ir_type_size[dst_type] == 8);
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			/* Avoid zero extension to the same register. This may be not always safe ??? */
+			if (force || op1_reg != def_reg) {
+				|	mov Rd(def_reg), Rd(op1_reg)
+			}
+|.endif
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		uint64_t val;

-	for (b = 1, bb = &ctx->cfg_blocks[b]; b <= ctx->cfg_blocks_count; bb++, b++) {
-		ir_insn *insn = &ctx->ir_base[bb->start];
+		if (src_size == 1)  {
+			val = ctx->ir_base[insn->op1].val.u8;
+		} else if (src_size == 2) {
+			val = ctx->ir_base[insn->op1].val.u16;
+		} else if (src_size == 4) {
+			val = ctx->ir_base[insn->op1].val.u32;
+		} else {
+			IR_ASSERT(src_size == 8);
+			val = ctx->ir_base[insn->op1].val.u64;
+		}
+		ir_emit_mov_imm_int(ctx, dst_type, def_reg, val);
+	} else {
+		ir_mem mem;

-		if (insn->op == IR_BEGIN && insn->op2) {
-			IR_ASSERT(ctx->ir_base[insn->op2].op == IR_LABEL);
-			ctx->ir_base[insn->op2].val.u32_hi = b;
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}
+
+		if (src_size == 1) {
+			if (ir_type_size[dst_type] == 2) {
+				|	ASM_TXT_TMEM_OP movzx, Rw(def_reg), byte, mem
+			} else if (ir_type_size[dst_type] == 4) {
+				|	ASM_TXT_TMEM_OP movzx, Rd(def_reg), byte, mem
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	ASM_TXT_TMEM_OP movzx, Rq(def_reg), byte, mem
+|.endif
+			}
+		} else if (src_size == 2) {
+			if (ir_type_size[dst_type] == 4) {
+				|	ASM_TXT_TMEM_OP movzx, Rd(def_reg), word, mem
+			} else {
+				IR_ASSERT(ir_type_size[dst_type] == 8);
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	ASM_TXT_TMEM_OP movzx, Rq(def_reg), word, mem
+|.endif
+			}
+		} else {
+			IR_ASSERT(src_size == 4);
+			IR_ASSERT(ir_type_size[dst_type] == 8);
+|.if X64
+			|	ASM_TXT_TMEM_OP mov, Rd(def_reg), dword, mem
+|.endif
 		}
 	}
 }

-static void ir_emit_load_label_addr(ir_ctx *ctx, ir_reg reg, ir_insn *label)
+static void ir_emit_zext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	ir_type src_type = ctx->ir_base[insn->op1].type;

-	if (!data->resolved_label_syms) {
-		data->resolved_label_syms = 1;
-		ir_resolve_label_syms(ctx);
+	ir_emit_zext_common(ctx, def, insn, insn->type, src_type, ir_type_size[src_type], IR_REG_NUM(ctx->regs[def][0]), 0);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, IR_REG_NUM(ctx->regs[def][0]));
 	}
+}

-	IR_ASSERT(label->op == IR_LABEL);
-	int b = label->val.u32_hi;
+static void ir_emit_and_zext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	uint8_t src_size;

-	b = ir_skip_empty_target_blocks(ctx, b);
-	|	lea Ra(reg), aword [=>b]
+	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+	if (ctx->ir_base[insn->op2].val.u64 == 0xff) {
+		src_size = 1;
+	} else if (ctx->ir_base[insn->op2].val.u64 == 0xffff) {
+		src_size = 2;
+	} else {
+		IR_ASSERT(ctx->ir_base[insn->op2].val.u64 == 0xffffffff);
+		src_size = 4;
+	}
+	ir_emit_zext_common(ctx, def, insn, insn->type, src_type, src_size, IR_REG_NUM(ctx->regs[def][0]), 1);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, IR_REG_NUM(ctx->regs[def][0]));
+	}
 }

-static void ir_emit_load(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref src)
+static void ir_emit_trunc(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	if (IR_IS_CONST_REF(src)) {
-		if (IR_IS_TYPE_INT(type)) {
-			ir_insn *insn = &ctx->ir_base[src];
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];

-			if (insn->op == IR_SYM || insn->op == IR_FUNC) {
-				void *addr = ir_sym_val(ctx, insn);
-				ir_emit_load_imm_int(ctx, type, reg, (intptr_t)addr);
-			} else if (insn->op == IR_STR) {
+	IR_ASSERT(IR_IS_TYPE_INT(src_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
+	IR_ASSERT(ir_type_size[dst_type] < ir_type_size[src_type]);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE) {
+#if IR_X86_I64
+		if (src_type == IR_I64 || src_type == IR_U64) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				ir_reg op1_reg_hi;
+
+				op1_reg = IR_REG_NUM(op1_reg);
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
+				ir_emit_load_i64_lo(ctx, op1_reg, insn->op1);
+				ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
+			} else {
+				op1_reg = IR_REG_I64_LO(op1_reg);
+			}
+		} else
+#endif
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (op1_reg != def_reg) {
+#ifdef IR_TARGET_X86
+			if (ir_type_size[dst_type] == 1
+			 && (op1_reg == IR_REG_RBP || op1_reg == IR_REG_RSI || op1_reg == IR_REG_RDI)) {
 				ir_backend_data *data = ctx->data;
 				dasm_State **Dst = &data->dasm_state;
-				int label = ir_get_const_label(ctx, src);

-				|	lea Ra(reg), aword [=>label]
-			} else if (insn->op == IR_LABEL) {
-				ir_emit_load_label_addr(ctx, reg, insn);
+#if IR_X86_I64
+				if (src_type == IR_I64 || src_type == IR_U64) {
+					src_type = IR_U32;
+				}
+#endif
+				ir_emit_mov(ctx, src_type, def_reg, op1_reg);
+				|	and	Rb(def_reg), 0xff
 			} else {
-				ir_emit_load_imm_int(ctx, type, reg, insn->val.i64);
+				ir_emit_mov(ctx, dst_type, def_reg, op1_reg);
 			}
-		} else {
-			ir_emit_load_imm_fp(ctx, type, reg, src);
+#else
+			ir_emit_mov(ctx, dst_type, def_reg, op1_reg);
+#endif
 		}
-	} else if (ctx->vregs[src]) {
-		ir_emit_load_mem(ctx, type, reg, ir_ref_spill_slot(ctx, src));
+	} else if (!IR_IS_CONST_REF(insn->op1) && ctx->vregs[insn->op1] == ctx->vregs[def]) {
+		/* If source and destination share the same vreg, we load the whole source */
+		ir_emit_load_ex(ctx, src_type, def_reg, insn->op1, def);
 	} else {
-		ir_load_local_addr(ctx, reg, src);
+		ir_emit_load_ex(ctx, dst_type, def_reg, insn->op1, def);
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
 	}
 }

-static void ir_emit_store_mem_int(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg reg)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-
-	|	ASM_MEM_REG_OP mov, type, mem, reg
-}
-
-static void ir_emit_store_mem_fp(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg reg)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-
-	|	ASM_FP_MEM_REG_OP movs, type, mem, reg
-}
-
-static void ir_emit_store_mem_imm(ir_ctx *ctx, ir_type type, ir_mem mem, int32_t imm)
+static void ir_emit_bitcast(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];

-	|	ASM_MEM_IMM_OP mov, type, mem, imm
-}
+	IR_ASSERT(ir_get_type_size(dst_type) == ir_get_type_size(src_type));
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (IR_IS_TYPE_INT(src_type) && IR_IS_TYPE_INT(dst_type)) {
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+			}
+			if (op1_reg != def_reg) {
+				ir_emit_mov(ctx, dst_type, def_reg, op1_reg);
+			}
+		} else {
+			ir_emit_load_ex(ctx, dst_type, def_reg, insn->op1, def);
+		}
+	} else if ((IR_IS_TYPE_FP(src_type) || IR_IS_TYPE_VECTOR(src_type))
+			&& (IR_IS_TYPE_FP(dst_type) || IR_IS_TYPE_VECTOR(dst_type))) {
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+			}
+			if (op1_reg != def_reg) {
+				ir_emit_fp_mov(ctx, dst_type, def_reg, op1_reg);
+			}
+		} else {
+			ir_emit_load_ex(ctx, dst_type, def_reg, insn->op1, def);
+		}
+	} else if (IR_IS_TYPE_FP(src_type) || IR_IS_TYPE_VECTOR(src_type)) {
+		IR_ASSERT(IR_IS_TYPE_INT(dst_type));
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+			}
+			if (src_type == IR_DOUBLE) {
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vmovq Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	movq Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+			} else if (src_type == IR_FLOAT) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vmovd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	movd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+				if (ir_type_size[dst_type] == 8) {
+|.if X64
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovq Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movq Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+|.endif
+				} else {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (ir_type_size[dst_type] == 2) {
+						|	and Rd(def_reg), 0xffff
+					} else if (ir_type_size[dst_type] == 1) {
+						|	and Rd(def_reg), 0xff
+					}
+				}
+			}
+		} else if (IR_IS_CONST_REF(insn->op1)) {
+			ir_insn *_insn = &ctx->ir_base[insn->op1];
+			IR_ASSERT(!IR_IS_SYM_CONST(_insn->op));
+			if (src_type == IR_DOUBLE) {
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	mov64 Rq(def_reg), _insn->val.i64
+|.endif
+			} else if (src_type == IR_FLOAT) {
+				|	mov Rd(def_reg), _insn->val.i32
+			} else {
+				void *p;

-static void ir_emit_store_mem_int_const(ir_ctx *ctx, ir_type type, ir_mem mem, ir_ref src, ir_reg tmp_reg, bool is_arg)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_insn *val_insn = &ctx->ir_base[src];
+				IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+				p = ir_long_const_ptr(ctx, insn->op1);
+				if (ir_type_size[dst_type] == 8) {
+|.if X64
+					|	mov64 Rq(def_reg), *(int64_t*)p
+|.endif
+				} else if (ir_type_size[dst_type] == 4) {
+					|	mov Rd(def_reg), *(int32_t*)p
+				} else if (ir_type_size[dst_type] == 2) {
+					|	mov Rw(def_reg), *(int16_t*)p
+				} else if (ir_type_size[dst_type] == 1) {
+					|	mov Rb(def_reg), *(int8_t*)p
+				} else {
+					IR_ASSERT(0);
+				}
+			}
+		} else {
+			ir_mem mem;

-	IR_ASSERT(IR_IS_CONST_REF(src));
-	if (val_insn->op == IR_STR) {
-		int label = ir_get_const_label(ctx, src);
+			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, insn->op1);
+			} else {
+				mem = ir_ref_spill_slot(ctx, insn->op1);
+			}

-		IR_ASSERT(tmp_reg != IR_REG_NONE);
+			if (src_type == IR_DOUBLE) {
+				IR_ASSERT(sizeof(void*) == 8);
 |.if X64
-		|	lea Ra(tmp_reg), aword [=>label]
-||		ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
-|.else
-		|	ASM_TMEM_TXT_OP mov, aword, mem, =>label
+				|	ASM_TXT_TMEM_OP mov, Rq(def_reg), qword, mem
 |.endif
-	} else if (val_insn->op == IR_LABEL) {
-		IR_ASSERT(tmp_reg != IR_REG_NONE);
-		tmp_reg = IR_REG_NUM(tmp_reg);
-		ir_emit_load_label_addr(ctx, tmp_reg, val_insn);
-		ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
-	} else {
-		int64_t val = val_insn->val.i64;
-
-		if (val_insn->op == IR_FUNC || val_insn->op == IR_SYM) {
-			val = (int64_t)(intptr_t)ir_sym_val(ctx, val_insn);
+			} else if (src_type == IR_FLOAT) {
+				|	ASM_TXT_TMEM_OP mov, Rd(def_reg), dword, mem
+			} else {
+				IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+				if (ir_type_size[dst_type] == 8) {
+|.if X64
+					|	ASM_TXT_TMEM_OP mov, Rq(def_reg), qword, mem
+|.endif
+				} else if (ir_type_size[dst_type] == 4) {
+					|	ASM_TXT_TMEM_OP mov, Rd(def_reg), dword, mem
+				} else if (ir_type_size[dst_type] == 2) {
+					|	ASM_TXT_TMEM_OP mov, Rw(def_reg), word, mem
+				} else if (ir_type_size[dst_type] == 1) {
+					|	ASM_TXT_TMEM_OP mov, Rb(def_reg), byte, mem
+				} else {
+					IR_ASSERT(0);
+				}
+			}
 		}
-
-		if (ir_type_size[val_insn->type] <= 4 || IR_IS_SIGNED_32BIT(val)) {
-			if (is_arg && ir_type_size[type] < 4) {
-				type = IR_U32;
+	} else if (IR_IS_TYPE_FP(dst_type) || IR_IS_TYPE_VECTOR(dst_type)) {
+		IR_ASSERT(IR_IS_TYPE_INT(src_type));
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+			}
+			if (dst_type == IR_DOUBLE) {
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+				} else {
+					|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+				}
+|.endif
+			} else if (dst_type == IR_FLOAT) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				} else {
+					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				}
+			} else {
+				IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+				if (ir_type_size[src_type] == 8) {
+|.if X64
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+					} else {
+						|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+					}
+|.endif
+				} else {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					} else {
+						|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					}
+				}
 			}
-			ir_emit_store_mem_imm(ctx, type, mem, val);
+		} else if (IR_IS_CONST_REF(insn->op1)) {
+			ir_emit_load_imm_fp(ctx, dst_type, def_reg, insn->op1);
 		} else {
-			IR_ASSERT(tmp_reg != IR_REG_NONE);
-			tmp_reg = IR_REG_NUM(tmp_reg);
-			ir_emit_load_imm_int(ctx, type, tmp_reg, val);
-			ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
-		}
-	}
-}
-
-static void ir_emit_store_mem_fp_const(ir_ctx *ctx, ir_type type, ir_mem mem, ir_ref src, ir_reg tmp_reg, ir_reg tmp_fp_reg)
-{
-	ir_val *val = &ctx->ir_base[src].val;
+			ir_mem mem;

-	if (type == IR_FLOAT) {
-		ir_emit_store_mem_imm(ctx, IR_U32, mem, val->i32);
-	} else if (sizeof(void*) == 8 && val->i64 == 0) {
-		ir_emit_store_mem_imm(ctx, IR_U64, mem, 0);
-	} else if (sizeof(void*) == 8 && tmp_reg != IR_REG_NONE) {
-		ir_emit_load_imm_int(ctx, IR_U64, tmp_reg, val->i64);
-		ir_emit_store_mem_int(ctx, IR_U64, mem, tmp_reg);
-	} else {
-		tmp_fp_reg = IR_REG_NUM(tmp_fp_reg);
-		ir_emit_load(ctx, type, tmp_fp_reg, src);
-		ir_emit_store_mem_fp(ctx, IR_DOUBLE, mem, tmp_fp_reg);
-	}
-}
+			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, insn->op1);
+			} else {
+				mem = ir_ref_spill_slot(ctx, insn->op1);
+			}

-static void ir_emit_store_mem(ir_ctx *ctx, ir_type type, ir_mem mem, ir_reg reg)
-{
-	if (IR_IS_TYPE_INT(type)) {
-		ir_emit_store_mem_int(ctx, type, mem, reg);
+			ir_emit_load_mem_fp(ctx, dst_type, def_reg, mem);
+		}
 	} else {
-		ir_emit_store_mem_fp(ctx, type, mem, reg);
+		IR_ASSERT(0);
 	}
-}

-static void ir_emit_store(ir_ctx *ctx, ir_type type, ir_ref dst, ir_reg reg)
-{
-	IR_ASSERT(dst >= 0);
-	ir_emit_store_mem(ctx, type, ir_ref_spill_slot(ctx, dst), reg);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
 }

-static void ir_emit_mov(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
+static void ir_emit_int2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];

-	|	ASM_REG_REG_OP mov, type, dst, src
-}
+	IR_ASSERT(IR_IS_TYPE_INT(src_type));
+	IR_ASSERT(IR_IS_TYPE_FP(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE);

-#define IR_HAVE_SWAP_INT
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	}

-static void ir_emit_swap(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	if (IR_IS_TYPE_UNSIGNED(src_type) && ir_type_size[src_type] >= sizeof(void*)) {
+		ir_reg tmp_reg = ctx->regs[def][2];

-	|	ASM_REG_REG_OP xchg, type, dst, src
-}
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		if (op1_reg == IR_REG_NONE) {
+			if (IR_IS_CONST_REF(insn->op1)) {
+				IR_ASSERT(0);
+			} else {
+				ir_mem mem;

-static void ir_emit_mov_ext(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+				if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, def, insn->op1);
+				} else {
+					mem = ir_ref_spill_slot(ctx, insn->op1);
+				}
+				ir_emit_load_mem_int(ctx, src_type, tmp_reg, mem);
+				op1_reg = tmp_reg;
+			}
+		}
+		if (sizeof(void*) == 4) {
+			if (tmp_reg == op1_reg) {
+				| add Rd(op1_reg), 0x80000000
+			} else {
+				| lea Rd(tmp_reg), dword [Rd(op1_reg)+0x80000000]
+				op1_reg = tmp_reg;
+			}
+		} else {
+|.if X64
+			|	test Rq(op1_reg), Rq(op1_reg)
+			|	js >1
+			|.cold_code
+			|1:
+			if (tmp_reg != op1_reg) {
+				| mov Rq(tmp_reg), Rq(op1_reg)
+			}
+			// TODO: we might replace "jnc" by "and", but this would require an extra temporary register ???
+			|	shr	Rq(tmp_reg), 1
+			|	jnc >3
+			|	or Rq(tmp_reg), 1
+			|3:
+			if (dst_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
+					|	vaddsd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
+					|	addsd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(dst_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
+					|	vaddss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
+					|	addss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			}
+			|	jmp >2
+			|.code
+|.endif
+		}
+	}

-	if (ir_type_size[type] > 2) {
-		|	ASM_REG_REG_OP mov, type, dst, src
-	} else if (ir_type_size[type] == 2) {
-		if (IR_IS_TYPE_SIGNED(type)) {
-			if (dst == IR_REG_RAX && src == IR_REG_RAX) {
-				|	cwde
+	if (op1_reg != IR_REG_NONE) {
+		bool src64 = 0;
+
+		if (IR_IS_TYPE_SIGNED(src_type)) {
+			if (ir_type_size[src_type] < 4) {
+|.if X64
+||				if (ir_type_size[src_type] == 1) {
+					| movsx Rq(op1_reg), Rb(op1_reg)
+||				} else {
+					| movsx Rq(op1_reg), Rw(op1_reg)
+||				}
+||				src64 = 1;
+|.else
+||				if (ir_type_size[src_type] == 1) {
+					| movsx Rd(op1_reg), Rb(op1_reg)
+||				} else if (op1_reg == IR_REG_RAX) {
+					| cwde
+||				} else {
+					| movsx Rd(op1_reg), Rw(op1_reg)
+||				}
+|.endif
+			} else if (ir_type_size[src_type] > 4) {
+				src64 = 1;
+			}
+		} else {
+			if (ir_type_size[src_type] < 8) {
+|.if X64
+||				if (ir_type_size[src_type] == 1) {
+					| movzx Rq(op1_reg), Rb(op1_reg)
+||				} else if (ir_type_size[src_type] == 2) {
+					| movzx Rq(op1_reg), Rw(op1_reg)
+||				} else if (ctx->ir_base[insn->op1].op == IR_TRUNC && IR_REG_NUM(ctx->regs[insn->op1][1]) == op1_reg) {
+					/* clear high bits (see: gcc/testsuite/gcc.dg/pr37544.c) */
+					| mov Rd(op1_reg), Rd(op1_reg)
+||				}
+||				src64 = 1;
+|.else
+||				if (ir_type_size[src_type] == 1) {
+					| movzx Rd(op1_reg), Rb(op1_reg)
+||				} else if (ir_type_size[src_type] == 2) {
+					| movzx Rd(op1_reg), Rw(op1_reg)
+||				}
+|.endif
+			} else {
+				src64 = 1;
+			}
+		}
+		if (!src64) {
+			if (dst_type == IR_DOUBLE || (sizeof(void*) == 4 && IR_IS_TYPE_UNSIGNED(src_type) && ir_type_size[src_type] == 4)) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				}
+			} else {
+				IR_ASSERT(dst_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				}
+			}
+		} else {
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			if (dst_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+				}
+			} else {
+				IR_ASSERT(dst_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+				}
+			}
+|.endif
+		}
+		|2:
+		if (sizeof(void*) == 4 && IR_IS_TYPE_UNSIGNED(src_type) && ir_type_size[src_type] >= sizeof(void*)) {
+			if (!data->u2d_const) {
+				data->u2d_const = 1;
+				ir_rodata(ctx);
+				|.align 8
+				|->u2d_const:
+				|.dword 0, 0x41e00000
+				|.code
+			}
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vaddsd	xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword [->u2d_const]
+			} else {
+				|	addsd	xmm(def_reg-IR_REG_FP_FIRST), qword [->u2d_const]
+			}
+			if (dst_type == IR_FLOAT) {
+				|	cvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+			}
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;
+		bool src64 = ir_type_size[src_type] == 8;
+
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}
+
+		if (!src64) {
+			if (dst_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TXT_TMEM_OP vcvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TMEM_OP cvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+				}
+			} else {
+				IR_ASSERT(dst_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TXT_TMEM_OP vcvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TMEM_OP cvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+				}
+			}
+		} else {
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			if (dst_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TXT_TMEM_OP vcvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TMEM_OP cvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				}
 			} else {
-				|	movsx Rd(dst), Rw(src)
+				IR_ASSERT(dst_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TXT_TMEM_OP vcvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				} else {
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TMEM_OP cvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				}
 			}
-		} else {
-			|	movzx Rd(dst), Rw(src)
-		}
-	} else /* if (ir_type_size[type] == 1) */ {
-		if (IR_IS_TYPE_SIGNED(type)) {
-			|	movsx Rd(dst), Rb(src)
-		} else {
-			|	movzx Rd(dst), Rb(src)
+|.endif
 		}
 	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
 }

-static void ir_emit_fp_mov(ir_ctx *ctx, ir_type type, ir_reg dst, ir_reg src)
+static void ir_emit_fp2int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][2];
+	bool dst64 = 0;

-	|	ASM_FP_REG_REG_OP movap, type, dst, src
-}
-
-static ir_mem ir_fuse_addr_const(ir_ctx *ctx, ir_ref ref)
-{
-	ir_mem mem;
-	ir_insn *addr_insn = &ctx->ir_base[ref];
-
-	IR_ASSERT(IR_IS_CONST_REF(ref));
-	if (IR_IS_SYM_CONST(addr_insn->op)) {
-		void *addr = ir_sym_val(ctx, addr_insn);
-		IR_ASSERT(sizeof(void*) == 4 || IR_IS_SIGNED_32BIT((intptr_t)addr));
-		mem = IR_MEM_O((int32_t)(intptr_t)addr);
-	} else {
-		IR_ASSERT(sizeof(void*) == 4 || IR_IS_SIGNED_32BIT(addr_insn->val.i64));
-		mem = IR_MEM_O(addr_insn->val.i32);
+	IR_ASSERT(IR_IS_TYPE_FP(src_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (IR_IS_TYPE_SIGNED(dst_type) ? ir_type_size[dst_type] == 8 : ir_type_size[dst_type] >= 4) {
+		// TODO: we might need to perform truncation from 32/64 bit integer
+		dst64 = 1;
 	}
-	return mem;
-}
-
-static ir_mem ir_fuse_addr(ir_ctx *ctx, ir_ref root, ir_ref ref)
-{
-	uint32_t rule = ctx->rules[ref];
-	ir_insn *insn = &ctx->ir_base[ref];
-	ir_insn *op1_insn, *op2_insn, *offset_insn;
-	ir_ref base_reg_ref, index_reg_ref;
-	ir_reg base_reg = IR_REG_NONE, index_reg;
-	int32_t offset = 0, scale;
-
-	IR_ASSERT(((rule & IR_RULE_MASK) >= IR_LEA_FIRST &&
-			(rule & IR_RULE_MASK) <= IR_LEA_LAST) ||
-		rule == IR_STATIC_ALLOCA);
-	switch (rule & IR_RULE_MASK) {
-		default:
-			IR_ASSERT(0);
-		case IR_LEA_OB:
-			offset_insn = insn;
-			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-			} else {
-				base_reg_ref = ref * sizeof(ir_ref) + 1;
-			}
-			index_reg_ref = IR_UNUSED;
-			scale = 1;
-			break;
-		case IR_LEA_SI:
-			scale = ctx->ir_base[insn->op2].val.i32;
-			index_reg_ref = ref * sizeof(ir_ref) + 1;
-			base_reg_ref = IR_UNUSED;
-			offset_insn = NULL;
-			break;
-		case IR_LEA_SIB:
-			base_reg_ref = index_reg_ref = ref * sizeof(ir_ref) + 1;
-			scale = ctx->ir_base[insn->op2].val.i32 - 1;
-			offset_insn = NULL;
-			break;
-		case IR_LEA_IB:
-			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = ref * sizeof(ir_ref) + 2;
-			} else if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = ref * sizeof(ir_ref) + 1;
-			} else {
-				base_reg_ref = ref * sizeof(ir_ref) + 1;
-				index_reg_ref = ref * sizeof(ir_ref) + 2;
-			}
-			offset_insn = NULL;
-			scale = 1;
-			break;
-		case IR_LEA_OB_I:
-			op1_insn = &ctx->ir_base[insn->op1];
-			offset_insn = op1_insn;
-			scale = 1;
-			if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-			} else if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = ref * sizeof(ir_ref) + 2;
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (!dst64) {
+			if (src_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttsd2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	cvttsd2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
 			} else {
-				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-				index_reg_ref = ref * sizeof(ir_ref) + 2;
+				IR_ASSERT(src_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttss2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	cvttss2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
 			}
-			break;
-		case IR_LEA_I_OB:
-			op2_insn = &ctx->ir_base[insn->op2];
-			offset_insn = op2_insn;
-			scale = 1;
-			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
-			} else if (ir_rule(ctx, op2_insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op2_insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = ref * sizeof(ir_ref) + 1;
+#ifdef IR_TARGET_X86
+|.if not X64
+		} else if (sizeof(void*) == 4 && dst_type == IR_U32) {
+			ir_reg fp;
+			int32_t offset;
+
+			offset = ctx->ret_slot;
+			IR_ASSERT(offset != -1);
+			offset = IR_SPILL_POS_TO_OFFSET(offset);
+			fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+
+			if (src_type == IR_DOUBLE) {
+				ir_emit_store_mem_fp(ctx, src_type, IR_MEM_BO(fp, offset), op1_reg);
+				|	fld qword [Ra(fp)+offset]
+				|	fisttp qword [Ra(fp)+offset]
+				|	mov Rd(def_reg), dword [Ra(fp)+offset]
+				|2:
 			} else {
-				base_reg_ref = ref * sizeof(ir_ref) + 1;
-				index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+				IR_ASSERT(src_type == IR_FLOAT);
+				ir_emit_store_mem_fp(ctx, src_type, IR_MEM_BO(fp, offset), op1_reg);
+				|	fld dword [Ra(fp)+offset]
+				|	fisttp qword [Ra(fp)+offset]
+				|	mov Rd(def_reg), dword [Ra(fp)+offset]
+				|2:
 			}
-			break;
-		case IR_LEA_SI_O:
-			index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-			op1_insn = &ctx->ir_base[insn->op1];
-			scale = ctx->ir_base[op1_insn->op2].val.i32;
-			offset_insn = insn;
-			base_reg_ref = IR_UNUSED;
-			break;
-		case IR_LEA_SIB_O:
-			base_reg_ref = index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-			op1_insn = &ctx->ir_base[insn->op1];
-			scale = ctx->ir_base[op1_insn->op2].val.i32 - 1;
-			offset_insn = insn;
-			break;
-		case IR_LEA_IB_O:
-			op1_insn = &ctx->ir_base[insn->op1];
-			offset_insn = insn;
-			scale = 1;
-			if (ir_rule(ctx, op1_insn->op2) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op2]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-			} else if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-				index_reg_ref = insn->op1 * sizeof(ir_ref) + 2;
-			} else {
-				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-				index_reg_ref = insn->op1 * sizeof(ir_ref) + 2;
+|.endif
+#endif
+		} else {
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			if (dst_type == IR_U64) {
+				if (src_type == IR_DOUBLE) {
+					if (!data->ull2d_const) {
+						data->ull2d_const = 1;
+						ir_rodata(ctx);
+						|.align 8
+						|->ull2d_const:
+						|.dword 0, 0x43e00000
+						|.code
+					}
+					|	ASM_FP_REG_TXT_OP ucomis, src_type, op1_reg, [->ull2d_const]
+					|	jnb >1
+				} else {
+					if (!data->ull2f_const) {
+						data->ull2f_const = 1;
+						ir_rodata(ctx);
+						|.align 4
+						|->ull2f_const:
+						|.dword 0x5f000000
+						|.code
+					}
+					IR_ASSERT(src_type == IR_FLOAT);
+					|	ASM_FP_REG_TXT_OP ucomis, src_type, op1_reg, [->ull2f_const]
+					|	jnb >1
+				}
 			}
-			break;
-		case IR_LEA_OB_SI:
-			index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
-			op1_insn = &ctx->ir_base[insn->op1];
-			offset_insn = op1_insn;
-			op2_insn = &ctx->ir_base[insn->op2];
-			scale = ctx->ir_base[op2_insn->op2].val.i32;
-			if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
+			if (src_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttsd2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	cvttsd2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
 			} else {
-				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+				IR_ASSERT(src_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttss2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	cvttss2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
 			}
-			break;
-		case IR_LEA_SI_OB:
-			index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-			op1_insn = &ctx->ir_base[insn->op1];
-			scale = ctx->ir_base[op1_insn->op2].val.i32;
-			op2_insn = &ctx->ir_base[insn->op2];
-			offset_insn = op2_insn;
-			if (ir_rule(ctx, op2_insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op2_insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
-			} else {
-				base_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
+			if (dst_type == IR_U64) {
+				IR_ASSERT(tmp_reg != IR_REG_NONE);
+				|	jmp >2
+				|1:
+				if (src_type == IR_DOUBLE) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	ASM_AVX_REG_REG_TXT_OP vsubs, src_type, tmp_reg, op1_reg, [->ull2d_const]
+						|	vcvttsd2si Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						if (tmp_reg != op1_reg) {
+							|	movsd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	ASM_SSE2_REG_TXT_OP subs, src_type, tmp_reg, [->ull2d_const]
+						|	cvttsd2si Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	ASM_AVX_REG_REG_TXT_OP vsubs, src_type, tmp_reg, op1_reg, [->ull2f_const]
+						|	vcvttss2si Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						if (tmp_reg != op1_reg) {
+							|	movss xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	ASM_SSE2_REG_TXT_OP subs, src_type, tmp_reg, [->ull2f_const]
+						|	cvttss2si Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				}
+				|	btc Rq(def_reg), 63
+				|2:
 			}
-			break;
-		case IR_LEA_B_SI:
-			if (ir_rule(ctx, insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
+|.endif
+		}
+
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		int label = ir_get_const_label(ctx, insn->op1);
+
+		if (!dst64) {
+			if (src_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttsd2si Rd(def_reg), qword [=>label]
+				} else {
+					|	cvttsd2si Rd(def_reg), qword [=>label]
+				}
 			} else {
-				base_reg_ref = ref * sizeof(ir_ref) + 1;
+				IR_ASSERT(src_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttss2si Rd(def_reg), dword [=>label]
+				} else {
+					|	cvttss2si Rd(def_reg), dword [=>label]
+				}
 			}
-			index_reg_ref = insn->op2 * sizeof(ir_ref) + 1;
-			op2_insn = &ctx->ir_base[insn->op2];
-			scale = ctx->ir_base[op2_insn->op2].val.i32;
-			offset_insn = NULL;
-			break;
-		case IR_LEA_SI_B:
-			index_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
-			if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
+		} else {
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			if (src_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttsd2si Rq(def_reg), qword [=>label]
+				} else {
+					|	cvttsd2si Rq(def_reg), qword [=>label]
+				}
 			} else {
-				base_reg_ref = ref * sizeof(ir_ref) + 2;
+				IR_ASSERT(src_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vcvttss2si Rq(def_reg), dword [=>label]
+				} else {
+					|	cvttss2si Rq(def_reg), dword [=>label]
+				}
 			}
-			op1_insn = &ctx->ir_base[insn->op1];
-			scale = ctx->ir_base[op1_insn->op2].val.i32;
-			offset_insn = NULL;
-			break;
-		case IR_LEA_B_SI_O:
-			offset_insn = insn;
-			op1_insn = &ctx->ir_base[insn->op1];
-			if (ir_rule(ctx, op1_insn->op1) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op1]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
+|.endif
+		}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}
+
+		if (!dst64) {
+			if (src_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	ASM_TXT_TMEM_OP vcvttsd2si, Rd(def_reg), qword, mem
+				} else {
+					|	ASM_TXT_TMEM_OP cvttsd2si, Rd(def_reg), qword, mem
+				}
 			} else {
-				base_reg_ref = insn->op1 * sizeof(ir_ref) + 1;
+				IR_ASSERT(src_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	ASM_TXT_TMEM_OP vcvttss2si, Rd(def_reg), dword, mem
+				} else {
+					|	ASM_TXT_TMEM_OP cvttss2si, Rd(def_reg), dword, mem
+				}
 			}
-			index_reg_ref = op1_insn->op2 * sizeof(ir_ref) + 1;
-			op2_insn = &ctx->ir_base[op1_insn->op2];
-			scale = ctx->ir_base[op2_insn->op2].val.i32;
-			break;
-		case IR_LEA_SI_B_O:
-			offset_insn = insn;
-			op1_insn = &ctx->ir_base[insn->op1];
-			index_reg_ref = op1_insn->op1 * sizeof(ir_ref) + 1;
-			if (ir_rule(ctx, op1_insn->op2) == IR_STATIC_ALLOCA) {
-				offset = ir_local_offset(ctx, &ctx->ir_base[op1_insn->op2]);
-				base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-				base_reg_ref = IR_UNUSED;
+		} else {
+			IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+			if (src_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	ASM_TXT_TMEM_OP vcvttsd2si, Rq(def_reg), qword, mem
+				} else {
+					|	ASM_TXT_TMEM_OP cvttsd2si, Rq(def_reg), qword, mem
+				}
 			} else {
-				base_reg_ref = insn->op1 * sizeof(ir_ref) + 2;
+				IR_ASSERT(src_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					|	ASM_TXT_TMEM_OP vcvttss2si, Rq(def_reg), dword, mem
+				} else {
+					|	ASM_TXT_TMEM_OP cvttss2si, Rq(def_reg), dword, mem
+				}
 			}
-			op1_insn = &ctx->ir_base[op1_insn->op1];
-			scale = ctx->ir_base[op1_insn->op2].val.i32;
-			break;
-		case IR_LEA_SYM_O:
-			op1_insn = &ctx->ir_base[insn->op1];
-			op2_insn = &ctx->ir_base[insn->op2];
-			offset = (intptr_t)ir_sym_val(ctx, op1_insn) + (intptr_t)op2_insn->val.i64;
-			base_reg_ref = index_reg_ref = IR_UNUSED;
-			scale = 1;
-			offset_insn = NULL;
-			break;
-		case IR_LEA_O_SYM:
-			op1_insn = &ctx->ir_base[insn->op1];
-			op2_insn = &ctx->ir_base[insn->op2];
-			offset = (intptr_t)ir_sym_val(ctx, op2_insn) + (intptr_t)op1_insn->val.i64;
-			base_reg_ref = index_reg_ref = IR_UNUSED;
-			scale = 1;
-			offset_insn = NULL;
-			break;
-		case IR_ALLOCA:
-			offset = ir_local_offset(ctx, insn);
-			base_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			base_reg_ref = index_reg_ref = IR_UNUSED;
-			scale = 1;
-			offset_insn = NULL;
-			break;
+|.endif
+		}
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
 	}
+}

-	if (offset_insn) {
-		ir_insn *addr_insn = &ctx->ir_base[offset_insn->op2];
+static void ir_emit_fp2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];

-		if (IR_IS_SYM_CONST(addr_insn->op)) {
-			void *addr = ir_sym_val(ctx, addr_insn);
-			IR_ASSERT(sizeof(void*) != 8 || IR_IS_SIGNED_32BIT((intptr_t)addr));
-			offset += (int64_t)(intptr_t)(addr);
+	IR_ASSERT(IR_IS_TYPE_FP(src_type));
+	IR_ASSERT(IR_IS_TYPE_FP(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (src_type == dst_type) {
+			if (op1_reg != def_reg) {
+				ir_emit_fp_mov(ctx, dst_type, def_reg, op1_reg);
+			}
+		} else if (src_type == IR_DOUBLE) {
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vcvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+			} else {
+				|	cvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+			}
 		} else {
-			if (offset_insn->op == IR_SUB) {
-				offset -= addr_insn->val.i32;
+			IR_ASSERT(src_type == IR_FLOAT);
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vcvtss2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
 			} else {
-				offset += addr_insn->val.i32;
+				|	cvtss2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
 			}
 		}
-	}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		int label = ir_get_const_label(ctx, insn->op1);

-	if (base_reg_ref) {
-		if (UNEXPECTED(ctx->rules[base_reg_ref / sizeof(ir_ref)] & IR_FUSED_REG)) {
-			base_reg = ir_get_fused_reg(ctx, root, base_reg_ref);
+		if (src_type == IR_DOUBLE) {
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vcvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword [=>label]
+			} else {
+				|	cvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), qword [=>label]
+			}
 		} else {
-			base_reg = ((int8_t*)ctx->regs)[base_reg_ref];
-		}
-		IR_ASSERT(base_reg != IR_REG_NONE);
-		if (IR_REG_SPILLED(base_reg)) {
-			base_reg = IR_REG_NUM(base_reg);
-			ir_emit_load(ctx, insn->type, base_reg, ((ir_ref*)ctx->ir_base)[base_reg_ref]);
+			IR_ASSERT(src_type == IR_FLOAT);
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vcvtss2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword [=>label]
+			} else {
+				|	cvtss2sd xmm(def_reg-IR_REG_FP_FIRST), dword [=>label]
+			}
 		}
-	}
+	} else {
+		ir_mem mem;

-	index_reg = IR_REG_NONE;
-	if (index_reg_ref) {
-		if (base_reg_ref
-			&& ((ir_ref*)ctx->ir_base)[index_reg_ref]
-				== ((ir_ref*)ctx->ir_base)[base_reg_ref]) {
-			index_reg = base_reg;
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
 		} else {
-			if (UNEXPECTED(ctx->rules[index_reg_ref / sizeof(ir_ref)] & IR_FUSED_REG)) {
-				index_reg = ir_get_fused_reg(ctx, root, index_reg_ref);
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}
+
+		if (src_type == IR_DOUBLE) {
+			if (ctx->mflags & IR_X86_AVX) {
+				|	ASM_TXT_TXT_TMEM_OP vcvtsd2ss, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword, mem
 			} else {
-				index_reg = ((int8_t*)ctx->regs)[index_reg_ref];
+				|	ASM_TXT_TMEM_OP cvtsd2ss, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
 			}
-			IR_ASSERT(index_reg != IR_REG_NONE);
-			if (IR_REG_SPILLED(index_reg)) {
-				index_reg = IR_REG_NUM(index_reg);
-				ir_emit_load(ctx, insn->type, index_reg, ((ir_ref*)ctx->ir_base)[index_reg_ref]);
+		} else {
+			IR_ASSERT(src_type == IR_FLOAT);
+			if (ctx->mflags & IR_X86_AVX) {
+				|	ASM_TXT_TXT_TMEM_OP vcvtss2sd, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+			} else {
+				|	ASM_TXT_TMEM_OP cvtss2sd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
 			}
 		}
 	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
+}
+
+static void ir_emit_copy_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_ref type = insn->type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];

-	return IR_MEM(base_reg, offset, index_reg, scale);
+	IR_ASSERT(def_reg != IR_REG_NONE || op1_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+	if (def_reg == op1_reg) {
+		/* same reg */
+	} else if (def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE) {
+		ir_emit_mov(ctx, type, def_reg, op1_reg);
+	} else if (def_reg != IR_REG_NONE) {
+		ir_emit_load(ctx, type, def_reg, insn->op1);
+	} else if (op1_reg != IR_REG_NONE) {
+		ir_emit_store(ctx, type, def, op1_reg);
+	} else {
+		IR_ASSERT(0);
+	}
+	if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
 }

-static ir_mem ir_fuse_mem(ir_ctx *ctx, ir_ref root, ir_ref ref, ir_insn *mem_insn, ir_reg reg)
+static void ir_emit_copy_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	if (reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(reg)) {
-			reg = IR_REG_NUM(reg);
-			ir_emit_load(ctx, IR_ADDR, reg, mem_insn->op2);
-		}
-		return IR_MEM_B(reg);
-	} else if (IR_IS_CONST_REF(mem_insn->op2)) {
-		return ir_fuse_addr_const(ctx, mem_insn->op2);
+	ir_type type = insn->type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(def_reg != IR_REG_NONE || op1_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+	if (def_reg == op1_reg) {
+		/* same reg */
+	} else if (def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE) {
+		ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
+	} else if (def_reg != IR_REG_NONE) {
+		ir_emit_load(ctx, type, def_reg, insn->op1);
+	} else if (op1_reg != IR_REG_NONE) {
+		ir_emit_store(ctx, type, def, op1_reg);
 	} else {
-		return ir_fuse_addr(ctx, root, mem_insn->op2);
+		IR_ASSERT(0);
+	}
+	if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
 	}
 }

-static ir_mem ir_fuse_load(ir_ctx *ctx, ir_ref root, ir_ref ref)
+static void ir_emit_vaddr(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_insn *load_insn = &ctx->ir_base[ref];
-	ir_reg reg;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_ref type = insn->type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_mem mem;
+	int32_t offset;
+	ir_reg fp;

-	IR_ASSERT(load_insn->op == IR_LOAD || load_insn->op == IR_LOAD_v ||
-		load_insn->op == IR_VLOAD || load_insn->op == IR_VLOAD_v);
-	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
-		reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 2);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	mem = ir_var_spill_slot(ctx, insn->op1);
+	fp = IR_MEM_BASE(mem);
+	offset = IR_MEM_OFFSET(mem);
+	if (offset == 0) {
+		|	mov Ra(def_reg), Ra(fp)
 	} else {
-		reg = ctx->regs[ref][2];
+		|	lea Ra(def_reg), aword [Ra(fp)+offset]
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
 	}
-	return ir_fuse_mem(ctx, root, ref, load_insn, reg);
 }

-static int32_t ir_fuse_imm(ir_ctx *ctx, ir_ref ref)
+static void ir_emit_vload(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_insn *val_insn = &ctx->ir_base[ref];
+	ir_insn *var_insn = &ctx->ir_base[insn->op2];
+	ir_ref type = insn->type;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg fp;
+	ir_mem mem;

-	IR_ASSERT(IR_IS_CONST_REF(ref));
-	if (IR_IS_SYM_CONST(val_insn->op)) {
-		void *addr = ir_sym_val(ctx, val_insn);
-		IR_ASSERT(IR_IS_SIGNED_32BIT((intptr_t)addr));
-		return (int32_t)(intptr_t)addr;
-	} else {
-		IR_ASSERT(ir_type_size[val_insn->type] == 4 || IR_IS_SIGNED_32BIT(val_insn->val.i64));
-		return val_insn->val.i32;
+	if (ctx->use_lists[def].count == 1) {
+		/* dead load */
+		return;
+	}
+	IR_ASSERT(var_insn->op == IR_VAR);
+	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	mem = IR_MEM_BO(fp, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
+	if (def_reg == IR_REG_NONE && ir_is_same_mem_var(ctx, def, var_insn->op3)) {
+		return; // fake load
+	}
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+#if IR_X86_I64
+	if (type == IR_I64 || type == IR_U64) {
+		ir_mem mem_hi = IR_MEM_I64_HI(mem);
+		ir_reg def_reg_hi = IR_REG_I64_HI(def_reg);
+
+		def_reg = IR_REG_I64_LO(def_reg);
+		ir_emit_load_mem_int(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, mem_hi);
+
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, def_reg);
+			ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+		}
+		return;
+	}
+#endif
+
+	ir_emit_load_mem(ctx, type, def_reg, mem);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
 	}
 }

-static void ir_emit_load_ex(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref src, ir_ref root)
+static void ir_emit_vstore_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
-	if (IR_IS_CONST_REF(src)) {
-		if (IR_IS_TYPE_INT(type)) {
-			ir_insn *insn = &ctx->ir_base[src];
+	ir_insn *var_insn = &ctx->ir_base[insn->op2];
+	ir_insn *val_insn = &ctx->ir_base[insn->op3];
+	ir_ref type = val_insn->type;
+	ir_reg op3_reg = ctx->regs[ref][3];
+	ir_reg fp;
+	ir_mem mem;

-			if (insn->op == IR_SYM || insn->op == IR_FUNC) {
-				void *addr = ir_sym_val(ctx, insn);
-				ir_emit_load_imm_int(ctx, type, reg, (intptr_t)addr);
-			} else if (insn->op == IR_STR) {
-				ir_backend_data *data = ctx->data;
-				dasm_State **Dst = &data->dasm_state;
-				int label = ir_get_const_label(ctx, src);
+	IR_ASSERT(var_insn->op == IR_VAR);
+	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	mem = IR_MEM_BO(fp, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
+	if ((op3_reg == IR_REG_NONE || IR_REG_SPILLED(op3_reg))
+	 && !IR_IS_CONST_REF(insn->op3)
+	 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
+	 && ir_is_same_mem_var(ctx, insn->op3, var_insn->op3)) {
+		return; // fake store
+	}
+	if (IR_IS_CONST_REF(insn->op3)) {
+		ir_emit_store_mem_int_const(ctx, type, mem, insn->op3, op3_reg, 0);
+	} else {
+		IR_ASSERT(op3_reg != IR_REG_NONE);
+#if IR_X86_I64
+		if (type == IR_I64 || type == IR_U64) {
+			ir_reg op3_reg_hi;
+			ir_mem mem_hi = IR_MEM_I64_HI(mem);

-				|	lea Ra(reg), aword [=>label]
-			} else if (insn->op == IR_LABEL) {
-				ir_emit_load_label_addr(ctx, reg, insn);
+			if (IR_REG_SPILLED(op3_reg)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				op3_reg_hi = IR_REG_I64_HI(op3_reg);
+				op3_reg = IR_REG_I64_LO(op3_reg);
+				ir_emit_load_i64_lo(ctx, op3_reg, insn->op3);
+				ir_emit_load_i64_hi(ctx, op3_reg_hi, insn->op3);
 			} else {
-				ir_emit_load_imm_int(ctx, type, reg, insn->val.i64);
+				op3_reg_hi = IR_REG_I64_HI(op3_reg);
+				op3_reg = IR_REG_I64_LO(op3_reg);
 			}
-		} else {
-			ir_emit_load_imm_fp(ctx, type, reg, src);
+			ir_emit_store_mem_int(ctx, IR_U32, mem, op3_reg);
+			ir_emit_store_mem_int(ctx, IR_U32, mem_hi, op3_reg_hi);
+			return;
 		}
-	} else if (ir_rule(ctx, src) == IR_STATIC_ALLOCA) {
-		ir_load_local_addr(ctx, reg, src);
-	} else {
-		ir_mem mem;
-
-		if (ir_rule(ctx, src) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, root, src);
-		} else {
-			mem = ir_ref_spill_slot(ctx, src);
+#endif
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, insn->op3);
 		}
-		ir_emit_load_mem(ctx, type, reg, mem);
+		ir_emit_store_mem_int(ctx, type, mem, op3_reg);
 	}
 }

-static void ir_emit_prologue(ir_ctx *ctx)
+static void ir_emit_vstore_fp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	int offset = ctx->stack_frame_size + ctx->call_stack_size;
+	ir_insn *var_insn = &ctx->ir_base[insn->op2];
+	ir_ref type = ctx->ir_base[insn->op3].type;
+	ir_reg op3_reg = ctx->regs[ref][3];
+	ir_reg fp;
+	ir_mem mem;

-	if (ctx->flags & IR_USE_FRAME_POINTER) {
-		|	push Ra(IR_REG_RBP)
-		|	mov Ra(IR_REG_RBP), Ra(IR_REG_RSP)
+	IR_ASSERT(var_insn->op == IR_VAR);
+	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	mem = IR_MEM_BO(fp, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
+	if ((op3_reg == IR_REG_NONE || IR_REG_SPILLED(op3_reg))
+	 && !IR_IS_CONST_REF(insn->op3)
+	 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
+	 && ir_is_same_mem_var(ctx, insn->op3, var_insn->op3)) {
+		return; // fake store
 	}
-	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP)) {
-		int i;
-		ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP);
-
-		for (i = IR_REG_GP_FIRST; i <= IR_REG_GP_LAST; i++) {
-			if (IR_REGSET_IN(used_preserved_regs, i)) {
-				offset -= sizeof(void*);
-				|	push Ra(i)
-			}
+	if (IR_IS_CONST_REF(insn->op3)) {
+		ir_emit_store_mem_fp_const(ctx, type, mem, insn->op3, IR_REG_NONE, op3_reg);
+	} else {
+		IR_ASSERT(op3_reg != IR_REG_NONE);
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, insn->op3);
 		}
+		ir_emit_store_mem_fp(ctx, type, mem, op3_reg);
 	}
-	if (ctx->stack_frame_size + ctx->call_stack_size) {
-		if (ctx->fixed_stack_red_zone) {
-			IR_ASSERT(ctx->stack_frame_size + ctx->call_stack_size <= ctx->fixed_stack_red_zone);
-		} else if (offset) {
-			|	sub Ra(IR_REG_RSP), offset
+}
+
+static void ir_emit_load_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_ref type = insn->type;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_mem mem;
+
+	if (ctx->use_lists[def].count == 1) {
+		/* dead load */
+		return;
+	}
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
+			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+		}
+		mem = IR_MEM_B(op2_reg);
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		mem = ir_fuse_addr_const(ctx, insn->op2);
+	} else {
+		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
+		mem = ir_fuse_addr(ctx, def, insn->op2);
+		if (IR_REG_SPILLED(ctx->regs[def][0]) && ir_is_same_spill_slot(ctx, def, mem)) {
+			if (!ir_may_avoid_spill_load(ctx, def, def)) {
+				ir_emit_load_mem_int(ctx, type, def_reg, mem);
+			}
+			/* avoid load to the same location (valid only when register is not reused) */
+			return;
 		}
 	}
-	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_FP)) {
-		ir_reg fp;
-		int i;
-		ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_FP);

-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			fp = IR_REG_FRAME_POINTER;
-			offset -= ctx->stack_frame_size + ctx->call_stack_size;
+#if IR_X86_I64
+	if (type == IR_I64 || type == IR_U64) {
+		ir_mem mem_hi = IR_MEM_I64_HI(mem);
+		ir_reg def_reg_hi = IR_REG_I64_HI(def_reg);
+
+		def_reg = IR_REG_I64_LO(def_reg);
+		if (IR_MEM_BASE(mem) != def_reg && IR_MEM_INDEX(mem) != def_reg) {
+			ir_emit_load_mem_int(ctx, IR_U32, def_reg, mem);
+			ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, mem_hi);
 		} else {
-			fp = IR_REG_STACK_POINTER;
+			IR_ASSERT(IR_MEM_BASE(mem) != def_reg_hi && IR_MEM_INDEX(mem) != def_reg_hi);
+			ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, mem_hi);
+			ir_emit_load_mem_int(ctx, IR_U32, def_reg, mem);
 		}
-		for (i = IR_REG_FP_FIRST; i <= IR_REG_FP_LAST; i++) {
-			if (IR_REGSET_IN(used_preserved_regs, i)) {
-				offset -= sizeof(void*);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vmovsd qword [Ra(fp)+offset], xmm(i-IR_REG_FP_FIRST)
-				} else {
-					|	movsd qword [Ra(fp)+offset], xmm(i-IR_REG_FP_FIRST)
-				}
-			}
+
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, def_reg);
+			ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 		}
+		return;
 	}
-	if ((ctx->flags & IR_VARARG_FUNC) && (ctx->flags2 & IR_HAS_VA_START)) {
-		const ir_call_conv_dsc *cc = data->ra_data.cc;
+#endif

-		if (cc->shadow_store_size) {
-			ir_reg fp;
-			int shadow_store;
-			int offset = 0;
-			int n = 0;
+	ir_emit_load_mem_int(ctx, type, def_reg, mem);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}

-			if (ctx->flags & IR_USE_FRAME_POINTER) {
-				fp = IR_REG_FRAME_POINTER;
-				shadow_store = sizeof(void*) * 2;
-			} else {
-				fp = IR_REG_STACK_POINTER;
-				shadow_store = ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*);
-			}
+static void ir_emit_load_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_ref type = insn->type;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_mem mem;

-			while (offset < cc->shadow_store_size && n < cc->int_param_regs_count) {
-				|	mov [Ra(fp)+shadow_store+offset], Ra(cc->int_param_regs[n])
-				n++;
-				offset += sizeof(void*);
+	if (ctx->use_lists[def].count == 1) {
+		/* dead load */
+		return;
+	}
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
+			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+		}
+		mem = IR_MEM_B(op2_reg);
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		mem = ir_fuse_addr_const(ctx, insn->op2);
+	} else {
+		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
+		mem = ir_fuse_addr(ctx, def, insn->op2);
+		if (IR_REG_SPILLED(ctx->regs[def][0]) && ir_is_same_spill_slot(ctx, def, mem)) {
+			if (!ir_may_avoid_spill_load(ctx, def, def)) {
+				ir_emit_load_mem_fp(ctx, type, def_reg, mem);
 			}
+			/* avoid load to the same location (valid only when register is not reused) */
+			return;
 		}
+	}

-		if (cc->sysv_varargs) {
-			IR_ASSERT(sizeof(void*) == 8);
-#ifdef IR_TARGET_X64
-|.if X64
-			int32_t i;
-			ir_reg fp;
-			int offset;
+	ir_emit_load_mem_fp(ctx, type, def_reg, mem);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
+}

-			if (ctx->flags & IR_USE_FRAME_POINTER) {
-				fp = IR_REG_FRAME_POINTER;
+static void ir_emit_store_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+{
+	ir_insn *val_insn = &ctx->ir_base[insn->op3];
+	ir_ref type = val_insn->type;
+	ir_reg op2_reg = ctx->regs[ref][2];
+	ir_reg op3_reg = ctx->regs[ref][3];
+	ir_mem mem;

-				offset = -(ctx->stack_frame_size - ctx->locals_area_size);
-			} else {
-				fp = IR_REG_STACK_POINTER;
-				offset = ctx->locals_area_size + ctx->call_stack_size;
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
+			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+		}
+		mem = IR_MEM_B(op2_reg);
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		mem = ir_fuse_addr_const(ctx, insn->op2);
+	} else {
+		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
+		mem = ir_fuse_addr(ctx, ref, insn->op2);
+		if (!IR_IS_CONST_REF(insn->op3)
+		 && IR_REG_SPILLED(op3_reg)
+		 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
+		 && ir_is_same_spill_slot(ctx, insn->op3, mem)) {
+			if (!ir_may_avoid_spill_load(ctx, insn->op3, ref)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				ir_emit_load(ctx, type, op3_reg, insn->op3);
 			}
+			/* avoid store to the same location */
+			return;
+		}
+	}

-			if ((ctx->flags2 & (IR_HAS_VA_ARG_GP|IR_HAS_VA_COPY)) && ctx->gp_reg_params < cc->int_param_regs_count) {
-				/* skip named args */
-				offset += sizeof(void*) * ctx->gp_reg_params;
-				for (i = ctx->gp_reg_params; i < cc->int_param_regs_count; i++) {
-					|	mov qword [Ra(fp)+offset], Rq(cc->int_param_regs[i])
-					offset += sizeof(void*);
-				}
-			}
-			if ((ctx->flags2 & (IR_HAS_VA_ARG_FP|IR_HAS_VA_COPY)) && ctx->fp_reg_params < cc->fp_param_regs_count) {
-				|	test al, al
-				|	je	>1
-				/* skip named args */
-				offset += 16 * ctx->fp_reg_params;
-				for (i = ctx->fp_reg_params; i < cc->fp_param_regs_count; i++) {
-					|	movaps [Ra(fp)+offset], xmm(cc->fp_param_regs[i]-IR_REG_FP_FIRST)
-					offset += 16;
-				}
-				|1:
+	if (IR_IS_CONST_REF(insn->op3)) {
+		ir_emit_store_mem_int_const(ctx, type, mem, insn->op3, op3_reg, 0);
+	} else {
+		IR_ASSERT(op3_reg != IR_REG_NONE);
+#if IR_X86_I64
+		if (type == IR_I64 || type == IR_U64) {
+			ir_reg op3_reg_hi;
+			ir_mem mem_hi = IR_MEM_I64_HI(mem);
+
+			if (IR_REG_SPILLED(op3_reg)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				op3_reg_hi = IR_REG_I64_HI(op3_reg);
+				op3_reg = IR_REG_I64_LO(op3_reg);
+				ir_emit_load_i64_lo(ctx, op3_reg, insn->op3);
+				ir_emit_load_i64_hi(ctx, op3_reg_hi, insn->op3);
+			} else {
+				op3_reg_hi = IR_REG_I64_HI(op3_reg);
+				op3_reg = IR_REG_I64_LO(op3_reg);
 			}
-|.endif
+			ir_emit_store_mem_int(ctx, IR_U32, mem, op3_reg);
+			ir_emit_store_mem_int(ctx, IR_U32, mem_hi, op3_reg_hi);
+			return;
+		}
 #endif
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, insn->op3);
 		}
+		ir_emit_store_mem_int(ctx, type, mem, op3_reg);
 	}
 }

-static void ir_emit_epilogue(ir_ctx *ctx)
+static void ir_emit_cmp_and_store_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-
-	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_FP)) {
-		int i;
-		int offset;
-		ir_reg fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-		ir_regset used_preserved_regs = (ir_regset)ctx->used_preserved_regs;
+	ir_reg addr_reg = ctx->regs[ref][2];
+	ir_mem mem;
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op3];
+	ir_op op = cmp_insn->op;
+	ir_type type = ctx->ir_base[cmp_insn->op1].type;
+	ir_ref op1 = cmp_insn->op1;
+	ir_ref op2 = cmp_insn->op2;
+	ir_reg op1_reg = ctx->regs[insn->op3][1];
+	ir_reg op2_reg = ctx->regs[insn->op3][2];

-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			fp = IR_REG_FRAME_POINTER;
-			offset = 0;
-		} else {
-			fp = IR_REG_STACK_POINTER;
-			offset = ctx->stack_frame_size + ctx->call_stack_size;
+	if (addr_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(addr_reg)) {
+			addr_reg = IR_REG_NUM(addr_reg);
+			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
+			ir_emit_load(ctx, IR_ADDR, addr_reg, insn->op2);
 		}
-		for (i = 0; i < IR_REG_NUM; i++) {
-			if (IR_REGSET_IN(used_preserved_regs, i)) {
-				if (i < IR_REG_FP_FIRST) {
-					offset -= sizeof(void*);
-				} else {
-					offset -= sizeof(void*);
-					if (ctx->mflags & IR_X86_AVX) {
-						|	vmovsd xmm(i-IR_REG_FP_FIRST), qword [Ra(fp)+offset]
-					} else {
-						|	movsd xmm(i-IR_REG_FP_FIRST), qword [Ra(fp)+offset]
-					}
-				}
-			}
+		mem = IR_MEM_B(addr_reg);
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		mem = ir_fuse_addr_const(ctx, insn->op2);
+	} else {
+		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
+		mem = ir_fuse_addr(ctx, ref, insn->op2);
+	}
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+	}
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		if (op1 != op2) {
+			ir_emit_load(ctx, type, op2_reg, op2);
 		}
 	}

-	if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP)) {
-		int i;
-		ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP);
-		int offset;
+	ir_emit_cmp_int_common(ctx, type, ref, cmp_insn, op1_reg, op1, op2_reg, op2);
+	_ir_emit_setcc_int_mem(ctx, op, mem);
+}

-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			offset = 0;
-		} else {
-			offset = ctx->stack_frame_size + ctx->call_stack_size;
-		}
-		if (IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP)) {
-			int i;
-			ir_regset used_preserved_regs = IR_REGSET_INTERSECTION((ir_regset)ctx->used_preserved_regs, IR_REGSET_GP);
+static void ir_emit_store_fp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+{
+	ir_ref type = ctx->ir_base[insn->op3].type;
+	ir_reg op2_reg = ctx->regs[ref][2];
+	ir_reg op3_reg = ctx->regs[ref][3];
+	ir_mem mem;

-			for (i = IR_REG_GP_LAST; i >= IR_REG_GP_FIRST; i--) {
-				if (IR_REGSET_IN(used_preserved_regs, i)) {
-					offset -= sizeof(void*);
-				}
-			}
-		}
-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			|	lea Ra(IR_REG_RSP), [Ra(IR_REG_RBP)+offset]
-		} else if (offset) {
-			|	add Ra(IR_REG_RSP), offset
+	IR_ASSERT(op3_reg != IR_REG_NONE);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
+			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
 		}
-		for (i = IR_REG_GP_LAST; i >= IR_REG_GP_FIRST; i--) {
-			if (IR_REGSET_IN(used_preserved_regs, i)) {
-				|	pop Ra(i)
+		mem = IR_MEM_B(op2_reg);
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		mem = ir_fuse_addr_const(ctx, insn->op2);
+	} else {
+		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
+		mem = ir_fuse_addr(ctx, ref, insn->op2);
+		if (!IR_IS_CONST_REF(insn->op3)
+		 && IR_REG_SPILLED(op3_reg)
+		 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
+		 && ir_is_same_spill_slot(ctx, insn->op3, mem)) {
+			if (!ir_may_avoid_spill_load(ctx, insn->op3, ref)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				ir_emit_load(ctx, type, op3_reg, insn->op3);
 			}
+			/* avoid store to the same location */
+			return;
 		}
-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			|	pop Ra(IR_REG_RBP)
-		}
-	} else if (ctx->flags & IR_USE_FRAME_POINTER) {
-		|	mov Ra(IR_REG_RSP), Ra(IR_REG_RBP)
-		|	pop Ra(IR_REG_RBP)
-	} else if (ctx->stack_frame_size + ctx->call_stack_size) {
-		if (ctx->fixed_stack_red_zone) {
-			IR_ASSERT(ctx->stack_frame_size + ctx->call_stack_size <= ctx->fixed_stack_red_zone);
-		} else {
-			|	add Ra(IR_REG_RSP), (ctx->stack_frame_size + ctx->call_stack_size)
+	}
+
+	if (IR_IS_CONST_REF(insn->op3)) {
+		ir_emit_store_mem_fp_const(ctx, type, mem, insn->op3, IR_REG_NONE, op3_reg);
+	} else {
+		IR_ASSERT(op3_reg != IR_REG_NONE);
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, insn->op3);
 		}
+		ir_emit_store_mem_fp(ctx, type, mem, op3_reg);
 	}
 }

-static void ir_emit_binop_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_rload(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	ir_reg src_reg = insn->op2;
 	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];

-	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (IR_REGSET_IN(IR_REGSET_UNION((ir_regset)ctx->fixed_regset, IR_REGSET_FIXED), src_reg)) {
+		if (ctx->vregs[def]
+		 && ctx->live_intervals[ctx->vregs[def]]
+		 && ctx->live_intervals[ctx->vregs[def]]->stack_spill_pos != -1) {
+			ir_emit_store(ctx, type, def, src_reg);
+		}
+	} else {
+		ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		if (def_reg == IR_REG_NONE) {
+			/* op3 is used as a flag that the value is already stored in memory.
+			 * If op3 is set we don't have to store the value once again (in case of spilling)
+			 */
+			if (!insn->op3 || !ir_is_same_spill_slot(ctx, def, IR_MEM_BO(ctx->spill_base, insn->op3))) {
+				ir_emit_store(ctx, type, def, src_reg);
+			}
 		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
-		}
-		if (op1 == op2) {
-			op2_reg = def_reg;
+			if (src_reg != def_reg) {
+				if (IR_IS_TYPE_INT(type)) {
+					ir_emit_mov(ctx, type, def_reg, src_reg);
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(type));
+					ir_emit_fp_mov(ctx, type, def_reg, src_reg);
+				}
+			}
+			if (IR_REG_SPILLED(ctx->regs[def][0])
+			 && (!insn->op3 || !ir_is_same_spill_slot(ctx, def,  IR_MEM_BO(ctx->spill_base, insn->op3)))) {
+				ir_emit_store(ctx, type, def, def_reg);
+			}
 		}
 	}
+}
+
+static void ir_emit_rstore(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+{
+	ir_ref type = ctx->ir_base[insn->op2].type;
+	ir_reg op2_reg = ctx->regs[ref][2];
+	ir_reg dst_reg = insn->op3;

 	if (op2_reg != IR_REG_NONE) {
 		if (IR_REG_SPILLED(op2_reg)) {
 			op2_reg = IR_REG_NUM(op2_reg);
-			if (op1 != op2) {
-				ir_emit_load(ctx, type, op2_reg, op2);
-			}
-		}
-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-			case IR_ADD_OV:
-				|	ASM_REG_REG_OP add, type, def_reg, op2_reg
-				break;
-			case IR_SUB:
-			case IR_SUB_OV:
-				|	ASM_REG_REG_OP sub, type, def_reg, op2_reg
-				break;
-			case IR_MUL:
-			case IR_MUL_OV:
-				|	ASM_REG_REG_MUL imul, type, def_reg, op2_reg
-				break;
-			case IR_OR:
-				|	ASM_REG_REG_OP or, type, def_reg, op2_reg
-				break;
-			case IR_AND:
-				|	ASM_REG_REG_OP and, type, def_reg, op2_reg
-				break;
-			case IR_XOR:
-				|	ASM_REG_REG_OP xor, type, def_reg, op2_reg
-				break;
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
 		}
-	} else if (IR_IS_CONST_REF(op2)) {
-		int32_t val = ir_fuse_imm(ctx, op2);
-
-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-			case IR_ADD_OV:
-				|	ASM_REG_IMM_OP add, type, def_reg, val
-				break;
-			case IR_SUB:
-			case IR_SUB_OV:
-				|	ASM_REG_IMM_OP sub, type, def_reg, val
-				break;
-			case IR_MUL:
-			case IR_MUL_OV:
-				|	ASM_REG_IMM_MUL imul, type, def_reg, val
-				break;
-			case IR_OR:
-				|	ASM_REG_IMM_OP or, type, def_reg, val
-				break;
-			case IR_AND:
-				|	ASM_REG_IMM_OP and, type, def_reg, val
-				break;
-			case IR_XOR:
-				|	ASM_REG_IMM_OP xor, type, def_reg, val
-				break;
+		if (op2_reg != dst_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, dst_reg, op2_reg);
+			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(type));
+				ir_emit_fp_mov(ctx, type, dst_reg, op2_reg);
+			}
 		}
 	} else {
-		ir_mem mem;
-
-		if (ir_rule(ctx, op2) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, op2);
-		} else {
-			mem = ir_ref_spill_slot(ctx, op2);
-		}
-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-			case IR_ADD_OV:
-				|	ASM_REG_MEM_OP add, type, def_reg, mem
-				break;
-			case IR_SUB:
-			case IR_SUB_OV:
-				|	ASM_REG_MEM_OP sub, type, def_reg, mem
-				break;
-			case IR_MUL:
-			case IR_MUL_OV:
-				|	ASM_REG_MEM_MUL imul, type, def_reg, mem
-				break;
-			case IR_OR:
-				|	ASM_REG_MEM_OP or, type, def_reg, mem
-				break;
-			case IR_AND:
-				|	ASM_REG_MEM_OP and, type, def_reg, mem
-				break;
-			case IR_XOR:
-				|	ASM_REG_MEM_OP xor, type, def_reg, mem
-				break;
-		}
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_load_ex(ctx, type, dst_reg, insn->op2, ref);
 	}
 }

-static void ir_emit_imul3(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_alloca(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	int32_t val = ir_fuse_imm(ctx, op2);

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	IR_ASSERT(!IR_IS_CONST_REF(op1));
+	if (ctx->use_lists[def].count == 1) {
+		/* dead alloca */
+		return;
+	}
+	if (IR_IS_CONST_REF(insn->op2)) {
+		ir_insn *val = &ctx->ir_base[insn->op2];
+		int32_t size = val->val.i32;

-	if (op1_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, type, op1_reg, op1);
-		}
-		switch (ir_type_size[type]) {
-			default:
-				IR_ASSERT(0);
-			case 2:
-				|	imul Rw(def_reg), Rw(op1_reg), val
-				break;
-			case 4:
-				|	imul Rd(def_reg), Rd(op1_reg), val
-				break;
-|.if X64
-||			case 8:
-|				imul Rq(def_reg), Rq(op1_reg), val
-||				break;
-|.endif
+		IR_ASSERT(IR_IS_TYPE_INT(val->type));
+		IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+		IR_ASSERT(IR_IS_TYPE_UNSIGNED(val->type) || val->val.i64 >= 0);
+		IR_ASSERT(IR_IS_SIGNED_32BIT(val->val.i64));
+
+		/* Stack must be 16 byte aligned */
+		size = IR_ALIGNED_SIZE(size, 16);
+		ir_stack_alloca(ctx, size);
+		if (!(ctx->flags & IR_USE_FRAME_POINTER)) {
+			ctx->call_stack_size += size;
 		}
 	} else {
-		ir_mem mem;
+		int32_t alignment = 16;
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_type type = ctx->ir_base[insn->op2].type;

-		if (ir_rule(ctx, op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, op1);
-		} else {
-			mem = ir_ref_spill_slot(ctx, op1);
+		IR_ASSERT(ctx->flags & IR_FUNCTION);
+		IR_ASSERT(ctx->flags & IR_USE_FRAME_POINTER);
+		IR_ASSERT(def_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
+		}
+		if (def_reg != op2_reg) {
+			if (op2_reg != IR_REG_NONE) {
+				ir_emit_mov(ctx, type, def_reg, op2_reg);
+			} else {
+				ir_emit_load(ctx, type, def_reg, insn->op2);
+			}
+		}
+
+		|	ASM_REG_IMM_OP add, IR_ADDR, def_reg, (alignment-1)
+		|	ASM_REG_IMM_OP and, IR_ADDR, def_reg, ~(alignment-1)
+		|	ASM_REG_REG_OP sub, IR_ADDR, IR_REG_RSP, def_reg
+	}
+	if (def_reg != IR_REG_NONE) {
+		|	mov Ra(def_reg), Ra(IR_REG_RSP)
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store(ctx, insn->type, def, def_reg);
 		}
-		|	ASM_REG_MEM_TXT_MUL imul, type, def_reg, mem, val
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+	} else {
+		ir_emit_store(ctx, IR_ADDR, def, IR_REG_STACK_POINTER);
 	}
 }

-static void ir_emit_min_max_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_afree(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];
-
-	IR_ASSERT(def_reg != IR_REG_NONE && op2_reg != IR_REG_NONE);
-
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
-		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
-		}
-	}

-	if (IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		if (op1 != op2) {
-			ir_emit_load(ctx, type, op2_reg, op2);
-		}
-	}
+	if (IR_IS_CONST_REF(insn->op2)) {
+		ir_insn *val = &ctx->ir_base[insn->op2];
+		int32_t size = val->val.i32;

-	if (op1 == op2) {
-		return;
-	}
+		IR_ASSERT(IR_IS_TYPE_INT(val->type));
+		IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+		IR_ASSERT(IR_IS_TYPE_UNSIGNED(val->type) || val->val.i64 > 0);
+		IR_ASSERT(IR_IS_SIGNED_32BIT(val->val.i64));

-	|	ASM_REG_REG_OP cmp, type, def_reg, op2_reg
-	if (insn->op == IR_MIN) {
-		if (IR_IS_TYPE_SIGNED(type)) {
-			|	ASM_REG_REG_OP2 cmovg, type, def_reg, op2_reg
-		} else {
-			|	ASM_REG_REG_OP2 cmova, type, def_reg, op2_reg
+		/* Stack must be 16 byte aligned */
+		size = IR_ALIGNED_SIZE(size, 16);
+		|	ASM_REG_IMM_OP add, IR_ADDR, IR_REG_RSP, size
+		if (!(ctx->flags & IR_USE_FRAME_POINTER)) {
+			ctx->call_stack_size -= size;
 		}
 	} else {
-		IR_ASSERT(insn->op == IR_MAX);
-		if (IR_IS_TYPE_SIGNED(type)) {
-			|	ASM_REG_REG_OP2 cmovl, type, def_reg, op2_reg
-		} else {
-			|	ASM_REG_REG_OP2 cmovb, type, def_reg, op2_reg
+//		int32_t alignment = 16;
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_type type = ctx->ir_base[insn->op2].type;
+
+		IR_ASSERT(ctx->flags & IR_FUNCTION);
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
 		}
-	}

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		// TODO: alignment ???
+
+		|	ASM_REG_REG_OP add, IR_ADDR, IR_REG_RSP, op2_reg
 	}
 }

-static void ir_emit_overflow(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_block_begin(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_type type = ctx->ir_base[insn->op1].type;

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	IR_ASSERT(IR_IS_TYPE_INT(type));
-	if (IR_IS_TYPE_SIGNED(type)) {
-		|	seto Rb(def_reg)
-	} else {
-		|	setc Rb(def_reg)
+	if (ctx->use_lists[def].count == 1) {
+		/* dead load */
+		return;
 	}
+	|	mov Ra(def_reg), Ra(IR_REG_RSP)
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
+		ir_emit_store(ctx, IR_ADDR, def, def_reg);
 	}
 }

-static void ir_emit_overflow_and_branch(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+static void ir_emit_block_end(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *overflow_insn = &ctx->ir_base[insn->op2];
-	ir_type type = ctx->ir_base[overflow_insn->op1].type;
-	uint32_t true_block, false_block;
-	bool reverse = 0;
+	ir_reg op2_reg = ctx->regs[def][2];

-	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
-	if (true_block == next_block) {
-		reverse = 1;
-		true_block = false_block;
-		false_block = 0;
-	} else if (false_block == next_block) {
-		false_block = 0;
+	IR_ASSERT(op2_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
 	}

-	if (IR_IS_TYPE_SIGNED(type)) {
-		if (reverse) {
-			|	jno =>true_block
-		} else {
-			|	jo =>true_block
-		}
+	|	mov Ra(IR_REG_RSP), Ra(op2_reg)
+}
+
+static void ir_emit_frame_addr(ir_ctx *ctx, ir_ref def)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+
+	if (ctx->flags & IR_USE_FRAME_POINTER) {
+		|	mov Ra(def_reg), Ra(IR_REG_RBP)
 	} else {
-		if (reverse) {
-			|	jnc =>true_block
-		} else {
-			|	jc =>true_block
-		}
+		|	lea Ra(def_reg), [Ra(IR_REG_RSP)+(ctx->stack_frame_size + ctx->call_stack_size)]
 	}
-	if (false_block) {
-		|	jmp =>false_block
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, IR_ADDR, def, def_reg);
 	}
 }

-static void ir_emit_mem_binop_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_va_start(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
+	const ir_call_conv_dsc *cc = data->ra_data.cc;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *op_insn = &ctx->ir_base[insn->op3];
-	ir_type type = op_insn->type;
-	ir_ref op2 = op_insn->op2;
-	ir_reg op2_reg = ctx->regs[insn->op3][2];
-	ir_mem mem;

-	if (insn->op == IR_STORE) {
-		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
-	} else {
-		IR_ASSERT(insn->op == IR_VSTORE);
-		mem = ir_var_spill_slot(ctx, insn->op2);
-	}
+	if (!cc->sysv_varargs) {
+		ir_reg fp;
+		int arg_area_offset;
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_reg tmp_reg = ctx->regs[def][3];
+		int32_t offset;

-	if (op2_reg == IR_REG_NONE) {
-		int32_t val = ir_fuse_imm(ctx, op2);
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+			}
+			offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
+			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+		}

-		switch (op_insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-			case IR_ADD_OV:
-				|	ASM_MEM_IMM_OP add, type, mem, val
-				break;
-			case IR_SUB:
-			case IR_SUB_OV:
-				|	ASM_MEM_IMM_OP sub, type, mem, val
-				break;
-			case IR_OR:
-				|	ASM_MEM_IMM_OP or, type, mem, val
-				break;
-			case IR_AND:
-				|	ASM_MEM_IMM_OP and, type, mem, val
-				break;
-			case IR_XOR:
-				|	ASM_MEM_IMM_OP xor, type, mem, val
-				break;
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			fp = IR_REG_FRAME_POINTER;
+			arg_area_offset = sizeof(void*) * 2 + ctx->param_stack_size;
+		} else {
+			fp = IR_REG_STACK_POINTER;
+			arg_area_offset = ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*) + ctx->param_stack_size;
 		}
+		|	lea Ra(tmp_reg), aword [Ra(fp)+arg_area_offset]
+		|	mov aword [Ra(op2_reg)+offset], Ra(tmp_reg)
 	} else {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, op2);
+		IR_ASSERT(sizeof(void*) == 8);
+#ifdef IR_TARGET_X64
+|.if X64
+		ir_reg fp;
+		int reg_save_area_offset;
+		int overflow_arg_area_offset;
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_reg tmp_reg = ctx->regs[def][3];
+		bool have_reg_save_area = 0;
+		int32_t offset;
+
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+			}
+			offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
+			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
 		}
-		switch (op_insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-			case IR_ADD_OV:
-				|	ASM_MEM_REG_OP add, type, mem, op2_reg
-				break;
-			case IR_SUB:
-			case IR_SUB_OV:
-				|	ASM_MEM_REG_OP sub, type, mem, op2_reg
-				break;
-			case IR_OR:
-				|	ASM_MEM_REG_OP or, type, mem, op2_reg
-				break;
-			case IR_AND:
-				|	ASM_MEM_REG_OP and, type, mem, op2_reg
-				break;
-			case IR_XOR:
-				|	ASM_MEM_REG_OP xor, type, mem, op2_reg
-				break;
+
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			fp = IR_REG_FRAME_POINTER;
+			reg_save_area_offset = -(int32_t)(ctx->stack_frame_size - ctx->locals_area_size);
+			overflow_arg_area_offset = sizeof(void*) * 2 + ctx->param_stack_size;
+		} else {
+			fp = IR_REG_STACK_POINTER;
+			reg_save_area_offset = ctx->locals_area_size + ctx->call_stack_size;
+			overflow_arg_area_offset = ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*) + ctx->param_stack_size;
+		}
+
+		if ((ctx->flags2 & (IR_HAS_VA_ARG_GP|IR_HAS_VA_COPY)) && ctx->gp_reg_params < cc->int_param_regs_count) {
+			|	lea Ra(tmp_reg), aword [Ra(fp)+reg_save_area_offset]
+			have_reg_save_area = 1;
+			/* Set va_list.gp_offset */
+			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], sizeof(void*) * ctx->gp_reg_params
+		} else {
+			reg_save_area_offset -= sizeof(void*) * cc->int_param_regs_count;
+			/* Set va_list.gp_offset */
+			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], sizeof(void*) * cc->int_param_regs_count
+		}
+		if ((ctx->flags2 & (IR_HAS_VA_ARG_FP|IR_HAS_VA_COPY)) && ctx->fp_reg_params < cc->fp_param_regs_count) {
+			if (!have_reg_save_area) {
+				|	lea Ra(tmp_reg), aword [Ra(fp)+reg_save_area_offset]
+				have_reg_save_area = 1;
+			}
+			/* Set va_list.fp_offset */
+			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], sizeof(void*) * cc->int_param_regs_count + 16 * ctx->fp_reg_params
+		} else {
+			/* Set va_list.fp_offset */
+			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], sizeof(void*) * cc->int_param_regs_count + 16 * cc->fp_param_regs_count
+		}
+		if (have_reg_save_area) {
+			/* Set va_list.reg_save_area */
+			|	mov qword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))], Ra(tmp_reg)
 		}
+		|	lea Ra(tmp_reg), aword [Ra(fp)+overflow_arg_area_offset]
+		/* Set va_list.overflow_arg_area */
+		|	mov qword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
+|.endif
+#endif
 	}
 }

-static void ir_emit_reg_binop_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_va_copy(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
+	const ir_call_conv_dsc *cc = data->ra_data.cc;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *op_insn = &ctx->ir_base[insn->op2];
-	ir_type type = op_insn->type;
-	ir_ref op2 = op_insn->op2;
-	ir_reg op2_reg = ctx->regs[insn->op2][2];
-	ir_reg reg;

-	IR_ASSERT(insn->op == IR_RSTORE);
-	reg = insn->op3;
-
-	if (op2_reg == IR_REG_NONE) {
-		int32_t val = ir_fuse_imm(ctx, op2);
+	if (!cc->sysv_varargs) {
+		ir_reg tmp_reg = ctx->regs[def][1];
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_reg op3_reg = ctx->regs[def][3];
+		int32_t op2_offset, op3_offset;

-		switch (op_insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_REG_IMM_OP add, type, reg, val
-				break;
-			case IR_SUB:
-				|	ASM_REG_IMM_OP sub, type, reg, val
-				break;
-			case IR_OR:
-				|	ASM_REG_IMM_OP or, type, reg, val
-				break;
-			case IR_AND:
-				|	ASM_REG_IMM_OP and, type, reg, val
-				break;
-			case IR_XOR:
-				|	ASM_REG_IMM_OP xor, type, reg, val
-				break;
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+			}
+			op2_offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
+			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			op2_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+		}
+		if (op3_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op3_reg)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				ir_emit_load(ctx, IR_ADDR, op3_reg, insn->op3);
+			}
+			op3_offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op3) == IR_STATIC_ALLOCA);
+			op3_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			op3_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op3]);
 		}
+		|	mov Ra(tmp_reg), aword [Ra(op3_reg)+op3_offset]
+		|	mov aword [Ra(op2_reg)+op2_offset], Ra(tmp_reg)
 	} else {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, op2);
+		IR_ASSERT(sizeof(void*) == 8);
+#ifdef IR_TARGET_X64
+|.if X64
+		ir_reg tmp_reg = ctx->regs[def][1];
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_reg op3_reg = ctx->regs[def][3];
+		int32_t op2_offset, op3_offset;
+
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+			}
+			op2_offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
+			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			op2_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
 		}
-		switch (op_insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_REG_REG_OP add, type, reg, op2_reg
-				break;
-			case IR_SUB:
-				|	ASM_REG_REG_OP sub, type, reg, op2_reg
-				break;
-			case IR_OR:
-				|	ASM_REG_REG_OP or, type, reg, op2_reg
-				break;
-			case IR_AND:
-				|	ASM_REG_REG_OP and, type, reg, op2_reg
-				break;
-			case IR_XOR:
-				|	ASM_REG_REG_OP xor, type, reg, op2_reg
-				break;
+		if (op3_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op3_reg)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				ir_emit_load(ctx, IR_ADDR, op3_reg, insn->op3);
+			}
+			op3_offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op3) == IR_STATIC_ALLOCA);
+			op3_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			op3_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op3]);
 		}
+		|	mov Rd(tmp_reg), dword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))]
+		|	mov dword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], Rd(tmp_reg)
+		|	mov Rd(tmp_reg), dword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))]
+		|	mov aword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], Ra(tmp_reg)
+		|	mov Ra(tmp_reg), aword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
+		|	mov aword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
+		|	mov Ra(tmp_reg), aword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))]
+		|	mov aword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))], Ra(tmp_reg)
+|.endif
+#endif
 	}
 }

-static void ir_emit_mul_div_mod_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_va_arg(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
+	const ir_call_conv_dsc *cc = data->ra_data.cc;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];

-	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
-	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (!cc->sysv_varargs) {
+		ir_type type = insn->type;
+		ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_reg tmp_reg = ctx->regs[def][3];
+		int32_t offset;

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
+		IR_ASSERT((def_reg != IR_REG_NONE || ctx->use_lists[def].count == 1) && tmp_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+			}
+			offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
+			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+		}
+		|	mov Ra(tmp_reg), aword [Ra(op2_reg)+offset]
+#if IR_X86_I64
+		if (type == IR_I64 || type == IR_U64) {
+			if (def_reg != IR_REG_NONE) {
+				ir_reg def_reg_hi = IR_REG_I64_HI(def_reg);
+				def_reg = IR_REG_I64_LO(def_reg);
+
+				|	mov Rd(def_reg), dword [Ra(tmp_reg)]
+				|	mov Rd(def_reg_hi), dword [Ra(tmp_reg)+4]
+
+				|	add Ra(tmp_reg), sizeof(uint64_t)
+				|	mov aword [Ra(op2_reg)+offset], Ra(tmp_reg)
+
+				if (IR_REG_SPILLED(ctx->regs[def][0])) {
+					ir_emit_store_i64_lo(ctx, def, def_reg);
+					ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+				}
+			} else {
+				|	add Ra(tmp_reg), sizeof(uint64_t)
+				|	mov aword [Ra(op2_reg)+offset], Ra(tmp_reg)
+			}
+			return;
+		} else
+#endif
+		if (!cc->pass_struct_by_val || !insn->op3) {
+			if (def_reg != IR_REG_NONE) {
+				ir_emit_load_mem(ctx, type, def_reg, IR_MEM_B(tmp_reg));
+			}
+			|	add Ra(tmp_reg), IR_MAX(ir_type_size[type], sizeof(void*))
+		} else {
+			int size = IR_VA_ARG_SIZE(insn->op3);
+
+			if (def_reg != IR_REG_NONE) {
+				IR_ASSERT(type == IR_ADDR);
+				int align = IR_VA_ARG_ALIGN(insn->op3);
+
+				if (align > (int)sizeof(void*)) {
+					|	add Ra(tmp_reg), (align-1)
+					|	and Ra(tmp_reg), ~(align-1)
+				}
+				|	mov Ra(def_reg), Ra(tmp_reg)
+			}
+			|	add Ra(tmp_reg), IR_ALIGNED_SIZE(size, sizeof(void*))
+		}
+		|	mov aword [Ra(op2_reg)+offset], Ra(tmp_reg)
+		if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store(ctx, type, def, def_reg);
+		}
+	} else {
+		IR_ASSERT(sizeof(void*) == 8);
+#ifdef IR_TARGET_X64
+|.if X64
+		ir_type type = insn->type;
+		ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+		ir_reg op2_reg = ctx->regs[def][2];
+		ir_reg tmp_reg = ctx->regs[def][3];
+		int32_t offset;
+
+		IR_ASSERT((def_reg != IR_REG_NONE || ctx->use_lists[def].count == 1) && tmp_reg != IR_REG_NONE);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+			}
+			offset = 0;
+		} else {
+			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
+			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+		}
+		if (insn->op3) {
+			/* long struct arguemnt */
+			IR_ASSERT(type == IR_ADDR);
+			int align = IR_VA_ARG_ALIGN(insn->op3);
+			int size = IR_VA_ARG_SIZE(insn->op3);
+
+			|	mov Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
+			if (align > (int)sizeof(void*)) {
+				|	add Ra(tmp_reg), (align-1)
+				|	and Ra(tmp_reg), ~(align-1)
+			}
+			if (def_reg != IR_REG_NONE) {
+				|	mov Ra(def_reg), Ra(tmp_reg)
+			}
+			|	add Ra(tmp_reg), IR_ALIGNED_SIZE(size, sizeof(void*))
+			|	mov aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
+		} else if (IR_IS_TYPE_INT(type)) {
+			|	mov Rd(tmp_reg), dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))]
+			|	cmp Rd(tmp_reg), sizeof(void*) * cc->int_param_regs_count
+			|	jge >1
+			|	add Rd(tmp_reg), sizeof(void*)
+			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], Rd(tmp_reg)
+			|	add Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))]
+			|	jmp >2
+			|1:
+			|	mov Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
+			|	add Ra(tmp_reg), sizeof(void*)
+			|	mov aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
+			|2:
+			if (def_reg != IR_REG_NONE) {
+				if (ir_type_size[type] == 8) {
+					|	mov Rq(def_reg), qword [Ra(tmp_reg)-sizeof(void*)]
+				} else {
+					|	mov Rd(def_reg), dword [Ra(tmp_reg)-sizeof(void*)]
+				}
+			}
 		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
+			|	mov Rd(tmp_reg), dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))]
+			|	cmp Rd(tmp_reg), sizeof(void*) * cc->int_param_regs_count + 16 * cc->fp_param_regs_count
+			|	jge >1
+			|	add Rd(tmp_reg), 16
+			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], Rd(tmp_reg)
+			|	add Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))]
+			if (def_reg != IR_REG_NONE) {
+				ir_emit_load_mem_fp(ctx, type, def_reg, IR_MEM_BO(tmp_reg, -16));
+			}
+			|	jmp >2
+			|1:
+			|	mov Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
+			if (def_reg != IR_REG_NONE) {
+				ir_emit_load_mem_fp(ctx, type, def_reg, IR_MEM_BO(tmp_reg, 0));
+			}
+			|	add Ra(tmp_reg), 8
+			|	mov aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
+			|2:
 		}
-	}
-	if (insn->op == IR_MUL) {
-		uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
-
-		if (shift == 1) {
-			|	ASM_REG_REG_OP add, type, def_reg, def_reg
-		} else {
-			|	ASM_REG_IMM_OP shl, type, def_reg, shift
+		if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store(ctx, type, def, def_reg);
 		}
-	} else if (insn->op == IR_DIV) {
-		uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
-
-		|	ASM_REG_IMM_OP shr, type, def_reg, shift
-	} else {
-		IR_ASSERT(insn->op == IR_MOD);
-		uint64_t mask = ctx->ir_base[insn->op2].val.u64 - 1;
-
-|.if X64
-||		if (ir_type_size[type] == 8 && ctx->regs[def][2] != IR_REG_NONE) {
-||			ir_reg op2_reg = ctx->regs[def][2];
-||
-||			op2_reg = IR_REG_NUM(op2_reg);
-||			ir_emit_load_imm_int(ctx, type, op2_reg, mask);
-			|	ASM_REG_REG_OP and, type, def_reg, op2_reg
-||		} else {
-|.endif
-			|	ASM_REG_IMM_OP and, type, def_reg, mask
-|.if X64
-||		}
 |.endif
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+#endif
 	}
 }

-static void ir_emit_bit_op(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_switch(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-
-	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
-	IR_ASSERT(def_reg != IR_REG_NONE);
+	ir_type type;
+	ir_block *bb;
+	ir_insn *use_insn, *val;
+	uint32_t n, *p, use_block;
+	int i;
+	int label, default_label = 0;
+	int count = 0;
+	ir_val min, max;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	bool has_case_range = 0;
+#if IR_X86_I64
+	ir_reg op2_reg_hi = IR_REG_NONE;
+#endif

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+	type = ctx->ir_base[insn->op2].type;
+	IR_ASSERT(tmp_reg != IR_REG_NONE);
+	if (IR_IS_TYPE_SIGNED(type)) {
+		min.u64 = 0x7fffffffffffffff;
+		max.u64 = 0x8000000000000000;
+	} else {
+		min.u64 = 0xffffffffffffffff;
+		max.u64 = 0x0;
 	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
+
+	bb = &ctx->cfg_blocks[b];
+	p = &ctx->cfg_edges[bb->successors];
+	for (n = bb->successors_count; n != 0; p++, n--) {
+		use_block = *p;
+		use_insn = &ctx->ir_base[ctx->cfg_blocks[use_block].start];
+		if (use_insn->op == IR_CASE_VAL) {
+			val = &ctx->ir_base[use_insn->op2];
+			IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+			if (IR_IS_TYPE_SIGNED(type)) {
+				IR_ASSERT(IR_IS_TYPE_SIGNED(val->type));
+				min.i64 = IR_MIN(min.i64, val->val.i64);
+				max.i64 = IR_MAX(max.i64, val->val.i64);
+			} else {
+				IR_ASSERT(!IR_IS_TYPE_SIGNED(val->type));
+				min.u64 = (int64_t)IR_MIN(min.u64, val->val.u64);
+				max.u64 = (int64_t)IR_MAX(max.u64, val->val.u64);
+			}
+			count++;
+		} else if (use_insn->op == IR_CASE_RANGE) {
+			has_case_range = 1;
+			val = &ctx->ir_base[use_insn->op2];
+			IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+			ir_insn *val2 = &ctx->ir_base[use_insn->op3];
+			IR_ASSERT(!IR_IS_SYM_CONST(val2->op));
+			if (IR_IS_TYPE_SIGNED(type)) {
+				IR_ASSERT(IR_IS_TYPE_SIGNED(val->type));
+				min.i64 = IR_MIN(min.i64, val->val.i64);
+				max.i64 = IR_MAX(max.i64, val2->val.i64);
+			} else {
+				IR_ASSERT(!IR_IS_TYPE_SIGNED(val->type));
+				min.u64 = (int64_t)IR_MIN(min.u64, val->val.u64);
+				max.u64 = (int64_t)IR_MAX(max.u64, val2->val.u64);
+			}
 		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
+			IR_ASSERT(use_insn->op == IR_CASE_DEFAULT);
+			default_label = ir_skip_empty_target_blocks(ctx, use_block);
 		}
 	}
-	if (insn->op == IR_OR) {
-		uint32_t bit = IR_LOG2(ctx->ir_base[insn->op2].val.u64);

-		|	ASM_REG16_IMM_OP, bts, type, def_reg, bit
-	} else {
-		IR_ASSERT(insn->op == IR_AND);
-		uint32_t bit = IR_LOG2(~ctx->ir_base[insn->op2].val.u64);
-
-		|	ASM_REG16_IMM_OP, btr, type, def_reg, bit
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+	IR_ASSERT(op2_reg != IR_REG_NONE);
+#if IR_X86_I64
+	if (type == IR_I64 || type == IR_U64) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, insn->op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, insn->op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+		}
+	} else
+#endif
+	if (IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, insn->op2);
 	}
-}
-
-static void ir_emit_sdiv_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
-	int64_t offset = ctx->ir_base[insn->op2].val.u64 - 1;

-	IR_ASSERT(shift != 0);
-	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
-	IR_ASSERT(op1_reg != IR_REG_NONE && def_reg != IR_REG_NONE && op1_reg != def_reg);
+	/* Generate a table jmp or a seqence of calls */
+	if (!has_case_range && count > 2 && (max.i64-min.i64) < count * 8) {
+		int *labels = ir_mem_malloc(sizeof(int) * (size_t)(max.i64 - min.i64 + 1));

-	if (IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
+		for (i = 0; i <= (max.i64 - min.i64); i++) {
+			labels[i] = default_label;
+		}
+		p = &ctx->cfg_edges[bb->successors];
+		for (n = bb->successors_count; n != 0; p++, n--) {
+			use_block = *p;
+			use_insn = &ctx->ir_base[ctx->cfg_blocks[use_block].start];
+			if (use_insn->op == IR_CASE_VAL) {
+				val = &ctx->ir_base[use_insn->op2];
+				IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+				label = ir_skip_empty_target_blocks(ctx, use_block);
+				labels[val->val.i64 - min.i64] = label;
+			}
+		}

-	if (shift == 1) {
-|.if X64
-||		if (ir_type_size[type] == 8) {
-			|	mov Rq(def_reg), Rq(op1_reg)
-			|	ASM_REG_IMM_OP shr, type, def_reg, 63
-			|	add Rq(def_reg), Rq(op1_reg)
-||		} else {
-|.endif
-			|	mov Rd(def_reg), Rd(op1_reg)
-			|	ASM_REG_IMM_OP shr, type, def_reg, (ir_type_size[type]*8-1)
-			|	add Rd(def_reg), Rd(op1_reg)
-|.if X64
-||		}
-|.endif
-	} else {
-|.if X64
-||		if (ir_type_size[type] == 8) {
-||			ir_reg op2_reg = ctx->regs[def][2];
-||
-||			if (op2_reg != IR_REG_NONE) {
-||				op2_reg =  IR_REG_NUM(op2_reg);
-||				ir_emit_load_imm_int(ctx, type, op2_reg, offset);
-				|	lea Rq(def_reg), [Rq(op1_reg)+Rq(op2_reg)]
-||			} else {
-				|	lea Rq(def_reg), [Rq(op1_reg)+(int32_t)offset]
-||			}
-||		} else {
-|.endif
-			|	lea Rd(def_reg), [Rd(op1_reg)+(int32_t)offset]
+		switch (ir_type_size[type]) {
+			default:
+				IR_ASSERT(0 && "Unsupported type size");
+			case 1:
+				if (IR_IS_TYPE_SIGNED(type)) {
+					|	movsx Ra(op2_reg), Rb(op2_reg)
+				} else {
+					|	movzx Ra(op2_reg), Rb(op2_reg)
+				}
+				break;
+			case 2:
+				if (IR_IS_TYPE_SIGNED(type)) {
+					|	movsx Ra(op2_reg), Rw(op2_reg)
+				} else {
+					|	movzx Ra(op2_reg), Rw(op2_reg)
+				}
+				break;
+			case 4:
 |.if X64
-||		}
+				if (IR_IS_TYPE_SIGNED(type)) {
+					if (op2_reg == IR_REG_RAX) {
+						|	cdqe
+					} else {
+						|	movsxd Ra(op2_reg), Rd(op2_reg)
+					}
+				} else if (ctx->ir_base[insn->op2].op == IR_TRUNC && IR_REG_NUM(ctx->regs[insn->op2][1]) == op2_reg) {
+					/* Explicit zero extnsion need in very rare case. */
+					|	mov Rd(op2_reg), Rd(op2_reg)
+				}
+				break;
+||			case 8:
 |.endif
-		|	ASM_REG_REG_OP test, type, op1_reg, op1_reg
-		|	ASM_REG_REG_OP2 cmovns, type, def_reg, op1_reg
-	}
-	|	ASM_REG_IMM_OP sar, type, def_reg, shift
+				break;
+		}

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
-}
+		if (min.i64 != 0) {
+			int64_t offset = -min.i64;

-static void ir_emit_smod_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg tmp_reg = ctx->regs[def][3];
-	uint32_t shift = IR_LOG2(ctx->ir_base[insn->op2].val.u64);
-	uint64_t mask = ctx->ir_base[insn->op2].val.u64 - 1;
+			if (IR_IS_SIGNED_32BIT(offset)) {
+				|	lea Ra(tmp_reg), [Ra(op2_reg)+(int32_t)offset]
+			} else {
+				IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+				|	mov64 Rq(tmp_reg), offset
+				|	add Ra(tmp_reg), Ra(op2_reg)
+|.endif
+			}
+			if (default_label) {
+				offset = max.i64 - min.i64;

-	IR_ASSERT(shift != 0);
-	IR_ASSERT(IR_IS_CONST_REF(insn->op2));
-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
-	IR_ASSERT(def_reg != IR_REG_NONE && tmp_reg != IR_REG_NONE && def_reg != tmp_reg);
+				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
+				|	cmp Ra(tmp_reg), (int32_t)offset
+				|	ja =>default_label
+			}
+|.if X64
+			if (ctx->code_buffer
+			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->start)
+			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->end)) {
+				|	jmp aword [Ra(tmp_reg)*8+>1]
+			} else {
+				int64_t offset = -min.i64;

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
+				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
+				offset *= 8;
+				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
+				|	lea Ra(tmp_reg), aword [>1]
+				|	jmp aword [Ra(tmp_reg)+Ra(op2_reg)*8+offset]
+			}
+|.else
+			|	jmp aword [Ra(tmp_reg)*4+>1]
+|.endif
 		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
+			if (default_label) {
+				int64_t offset = max.i64;
+
+				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
+				|	cmp Ra(op2_reg), (int32_t)offset
+				|	ja =>default_label
+			}
+|.if X64
+			if (ctx->code_buffer
+			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->start)
+			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->end)) {
+				|	jmp aword [Ra(op2_reg)*8+>1]
+			} else {
+				|	lea Ra(tmp_reg), aword [>1]
+				|	jmp aword [Ra(tmp_reg)+Ra(op2_reg)*8]
+			}
+|.else
+			|	jmp aword [Ra(op2_reg)*4+>1]
+|.endif
 		}
-	}
-	if (tmp_reg != op1_reg) {
-		ir_emit_mov(ctx, type, tmp_reg, def_reg);
-	}

+		|.jmp_table
+		if (!data->jmp_table_label) {
+			data->jmp_table_label = ctx->cfg_blocks_count + ctx->consts_count + 3;
+			|=>data->jmp_table_label:
+		}
+		|.align aword
+		|1:
+		for (i = 0; i <= (max.i64 - min.i64); i++) {
+			int b = labels[i];
+			if (b) {
+				ir_block *bb = &ctx->cfg_blocks[b];
+				ir_insn *insn = &ctx->ir_base[bb->end];

-	if (shift == 1) {
-		|	ASM_REG_IMM_OP shr, type, tmp_reg, (ir_type_size[type]*8-1)
-	} else {
-		|	ASM_REG_IMM_OP sar, type, tmp_reg, (ir_type_size[type]*8-1)
-		|	ASM_REG_IMM_OP shr, type, tmp_reg, (ir_type_size[type]*8-shift)
-	}
-	|	ASM_REG_REG_OP add, type, def_reg, tmp_reg
+				if (insn->op == IR_IJMP && IR_IS_CONST_REF(insn->op2)) {
+					ir_ref prev = ctx->prev_ref[bb->end];
+					if (prev != bb->start && ctx->ir_base[prev].op == IR_SNAPSHOT) {
+						prev = ctx->prev_ref[prev];
+					}
+					if (prev == bb->start) {
+						void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op2]);

+						|	.aword &addr
+						if (ctx->ir_base[bb->start].op1 == def
+						 && ctx->ir_base[bb->start].op != IR_CASE_DEFAULT) {
+							bb->flags |= IR_BB_EMPTY;
+						}
+						continue;
+					}
+				}
+				|	.aword =>b
+			} else {
+				|	.aword 0
+			}
+		}
+		|.code
+		ir_mem_free(labels);
+	} else {
+		p = &ctx->cfg_edges[bb->successors];
+		for (n = bb->successors_count; n != 0; p++, n--) {
+			use_block = *p;
+			use_insn = &ctx->ir_base[ctx->cfg_blocks[use_block].start];
+			if (use_insn->op == IR_CASE_VAL) {
+				val = &ctx->ir_base[use_insn->op2];
+				IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+				label = ir_skip_empty_target_blocks(ctx, use_block);
+#if IR_X86_I64
+				if (type == IR_I64 || type == IR_U64) {
+					if (val->val.u64 == 0) {
+						|	mov Rd(tmp_reg), Rd(op2_reg)
+						|	or Rd(tmp_reg), Rd(op2_reg_hi)
+					} else {
+						|	cmp Rd(op2_reg_hi), val->val.u32_hi
+						|   jne >1
+						|	cmp Rd(op2_reg), val->val.i32
+					}
+				} else
+#endif
+				if (val->val.u64 == 0) {
+					|	ASM_REG_REG_OP test, type, op2_reg, op2_reg
+				} else if (IR_IS_32BIT(type, val->val)) {
+					|	ASM_REG_IMM_OP cmp, type, op2_reg, val->val.i32
+				} else {
+					IR_ASSERT(sizeof(void*) == 8);
 |.if X64
-||	if (ir_type_size[type] == 8 && ctx->regs[def][2] != IR_REG_NONE) {
-||		ir_reg op2_reg = ctx->regs[def][2];
-||
-||		op2_reg = IR_REG_NUM(op2_reg);
-||		ir_emit_load_imm_int(ctx, type, op2_reg, mask);
-		|	ASM_REG_REG_OP and, type, def_reg, op2_reg
-||	} else {
+					|	mov64 Ra(tmp_reg), val->val.i64
+					|	ASM_REG_REG_OP cmp, type, op2_reg, tmp_reg
 |.endif
-		|	ASM_REG_IMM_OP and, type, def_reg, mask
+				}
+				|	je =>label
+				|1:
+			} else if (use_insn->op == IR_CASE_RANGE) {
+				val = &ctx->ir_base[use_insn->op2];
+				IR_ASSERT(!IR_IS_SYM_CONST(val->op));
+				label = ir_skip_empty_target_blocks(ctx, use_block);
+#if IR_X86_I64
+				if (type == IR_I64 || type == IR_U64) {
+					|	cmp Rd(op2_reg), val->val.u32
+					|	mov Rd(tmp_reg), Rd(op2_reg_hi)
+					|	sbb Rd(tmp_reg), val->val.u32_hi
+					if (IR_IS_TYPE_SIGNED(type)) {
+						|	jl >1
+					} else {
+						|	jc >1
+					}
+				} else
+#endif
+				if (IR_IS_32BIT(type, val->val)) {
+					|	ASM_REG_IMM_OP cmp, type, op2_reg, val->val.i32
+					if (IR_IS_TYPE_SIGNED(type)) {
+						|	jl >1
+					} else {
+						|	jb >1
+					}
+				} else {
+					IR_ASSERT(sizeof(void*) == 8);
 |.if X64
-||	}
+					|	mov64 Ra(tmp_reg), val->val.i64
+					|	ASM_REG_REG_OP cmp, type, op2_reg, tmp_reg
+					if (IR_IS_TYPE_SIGNED(type)) {
+						|	jl >1
+					} else {
+						|	jb >1
+					}
 |.endif
+				}

-	|	ASM_REG_REG_OP sub, type, def_reg, tmp_reg
-
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
-}
-
-static void ir_emit_mem_mul_div_mod_pwr2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_insn *op_insn = &ctx->ir_base[insn->op3];
-	ir_type type = op_insn->type;
-	ir_mem mem;
-
-	IR_ASSERT(IR_IS_CONST_REF(op_insn->op2));
-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[op_insn->op2].op));
-
-	if (insn->op == IR_STORE) {
-		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
-	} else {
-		IR_ASSERT(insn->op == IR_VSTORE);
-		mem = ir_var_spill_slot(ctx, insn->op2);
-	}
-
-	if (op_insn->op == IR_MUL) {
-		uint32_t shift = IR_LOG2(ctx->ir_base[op_insn->op2].val.u64);
-		|	ASM_MEM_IMM_OP shl, type, mem, shift
-	} else if (op_insn->op == IR_DIV) {
-		uint32_t shift = IR_LOG2(ctx->ir_base[op_insn->op2].val.u64);
-		|	ASM_MEM_IMM_OP shr, type, mem, shift
-	} else {
-		IR_ASSERT(op_insn->op == IR_MOD);
-		uint64_t mask = ctx->ir_base[op_insn->op2].val.u64 - 1;
-		IR_ASSERT(IR_IS_UNSIGNED_32BIT(mask));
-		|	ASM_MEM_IMM_OP and, type, mem, mask
-	}
-}
-
-static void ir_emit_shift(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];
-
-	IR_ASSERT(def_reg != IR_REG_NONE && def_reg != IR_REG_RCX);
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, insn->op1);
-	}
-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, insn->op2);
-	}
-	if (op2_reg != IR_REG_RCX) {
-		if (op1_reg == IR_REG_RCX) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
-			op1_reg = def_reg;
-		}
-		if (op2_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, IR_REG_RCX, op2_reg);
-		} else {
-			ir_emit_load(ctx, type, IR_REG_RCX, insn->op2);
+				val = &ctx->ir_base[use_insn->op3];
+				IR_ASSERT(!IR_IS_SYM_CONST(val->op3));
+				label = ir_skip_empty_target_blocks(ctx, use_block);
+#if IR_X86_I64
+				if (type == IR_I64 || type == IR_U64) {
+					|	mov Rd(tmp_reg), val->val.u32
+					|	cmp Rd(tmp_reg), Rd(op2_reg)
+					|	mov Rd(tmp_reg), val->val.u32_hi
+					|	sbb Rd(tmp_reg), Rd(op2_reg_hi)
+					if (IR_IS_TYPE_SIGNED(type)) {
+						|	jl >1
+					} else {
+						|	jc >1
+					}
+					|	jmp =>label
+				} else
+#endif
+				if (IR_IS_32BIT(type, val->val)) {
+					|	ASM_REG_IMM_OP cmp, type, op2_reg, val->val.i32
+					if (IR_IS_TYPE_SIGNED(type)) {
+						|	jle =>label
+					} else {
+						|	jbe =>label
+					}
+				} else {
+					IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+					|	mov64 Ra(tmp_reg), val->val.i64
+					|	ASM_REG_REG_OP cmp, type, op2_reg, tmp_reg
+					if (IR_IS_TYPE_SIGNED(type)) {
+						|	jle =>label
+					} else {
+						|	jbe =>label
+					}
+|.endif
+				}
+				|1:
+			}
 		}
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
-		} else {
-			ir_emit_load(ctx, type, def_reg, insn->op1);
+		if (default_label) {
+			|	jmp =>default_label
 		}
 	}
-	switch (insn->op) {
-		default:
-			IR_ASSERT(0);
-		case IR_SHL:
-			|	ASM_REG_TXT_OP shl, insn->type, def_reg, cl
-			break;
-		case IR_SHR:
-			|	ASM_REG_TXT_OP shr, insn->type, def_reg, cl
-			break;
-		case IR_SAR:
-			|	ASM_REG_TXT_OP sar, insn->type, def_reg, cl
-			break;
-		case IR_ROL:
-			|	ASM_REG_TXT_OP rol, insn->type, def_reg, cl
-			break;
-		case IR_ROR:
-			|	ASM_REG_TXT_OP ror, insn->type, def_reg, cl
-			break;
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
 }

-static void ir_emit_mem_shift(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static int32_t ir_call_used_stack(ir_ctx *ctx, ir_insn *insn, const ir_call_conv_dsc *cc, int *copy_stack_ptr)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_insn *op_insn = &ctx->ir_base[insn->op3];
-	ir_type type = op_insn->type;
-	ir_ref op2 = op_insn->op2;
-	ir_reg op2_reg = ctx->regs[insn->op3][2];
-	ir_mem mem;
+	int j, n;
+	ir_type type;
+	int int_param = 0;
+	int fp_param = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param = 0;
+#endif
+	int32_t used_stack = 0;
+	int32_t copy_stack = 0;

-	if (insn->op == IR_STORE) {
-		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
-	} else {
-		IR_ASSERT(insn->op == IR_VSTORE);
-		mem = ir_var_spill_slot(ctx, insn->op2);
-	}
+	n = insn->inputs_count;
+	for (j = 3; j <= n; j++) {
+		ir_insn *arg = &ctx->ir_base[ir_insn_op(insn, j)];
+		type = arg->type;
+		if (IR_IS_TYPE_INT(type)) {
+			if (arg->op == IR_ARGVAL) {
+				int size = arg->op2;
+				int align = arg->op3;

-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, op2);
-	}
-	if (op2_reg != IR_REG_RCX) {
-		if (op2_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, IR_REG_RCX, op2_reg);
+				if (!cc->pass_struct_by_val) {
+					copy_stack += size;
+					align = IR_MAX((int)sizeof(void*), align);
+					copy_stack = IR_ALIGNED_SIZE(copy_stack, align);
+					type = IR_ADDR;
+				} else {
+					align = IR_MAX((int)sizeof(void*), align);
+					used_stack = IR_ALIGNED_SIZE(used_stack, align);
+					used_stack += size;
+					used_stack = IR_ALIGNED_SIZE(used_stack, sizeof(void*));
+					continue;
+				}
+			}
+#if IR_X86_I64
+			if (type == IR_I64 || type == IR_U64) {
+				used_stack += IR_MAX(sizeof(void*), ir_type_size[type]);
+				int_param++;
+				if (cc->shadow_param_regs) {
+					fp_param++;
+				}
+			} else
+#endif
+			if (int_param >= cc->int_param_regs_count) {
+				used_stack += IR_MAX(sizeof(void*), ir_type_size[type]);
+			}
+			int_param++;
+			if (cc->shadow_param_regs) {
+				fp_param++;
+			}
+#if IR_SIMD && defined(IR_TARGET_X86)
+		} else if (IR_IS_TYPE_VECTOR(type)) {
+			if (vector_param >= cc->vector_param_regs_count) {
+				used_stack += IR_MAX(sizeof(void*), ir_get_type_size(type));
+			}
+			vector_param++;
+#endif
 		} else {
-			ir_emit_load(ctx, type, IR_REG_RCX, op2);
+			IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
+			if (fp_param >= cc->fp_param_regs_count) {
+				used_stack += IR_MAX(sizeof(void*), ir_get_type_size(type));
+			}
+			fp_param++;
+			if (cc->shadow_param_regs) {
+				int_param++;
+			}
 		}
 	}
-	switch (op_insn->op) {
-		default:
-			IR_ASSERT(0);
-		case IR_SHL:
-			|	ASM_MEM_TXT_OP shl, type, mem, cl
-			break;
-		case IR_SHR:
-			|	ASM_MEM_TXT_OP shr, type, mem, cl
-			break;
-		case IR_SAR:
-			|	ASM_MEM_TXT_OP sar, type, mem, cl
-			break;
-		case IR_ROL:
-			|	ASM_MEM_TXT_OP rol, type, mem, cl
-			break;
-		case IR_ROR:
-			|	ASM_MEM_TXT_OP ror, type, mem, cl
-			break;
-	}
-}

-static void ir_emit_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	int32_t shift;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	/* Reserved "home space" or "shadow store" for register arguments (used in Windows64 ABI) */
+	used_stack += cc->shadow_store_size;

-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
-	IR_ASSERT(IR_IS_SIGNED_32BIT(ctx->ir_base[insn->op2].val.i64));
-	shift = ctx->ir_base[insn->op2].val.i32;
-	IR_ASSERT(def_reg != IR_REG_NONE);
+	copy_stack = IR_ALIGNED_SIZE(copy_stack, 16);
+	used_stack += copy_stack;
+	*copy_stack_ptr = copy_stack;

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
-		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
-		}
-	}
-	switch (insn->op) {
-		default:
-			IR_ASSERT(0);
-		case IR_SHL:
-			|	ASM_REG_IMM_OP shl, insn->type, def_reg, shift
-			break;
-		case IR_SHR:
-			|	ASM_REG_IMM_OP shr, insn->type, def_reg, shift
-			break;
-		case IR_SAR:
-			|	ASM_REG_IMM_OP sar, insn->type, def_reg, shift
-			break;
-		case IR_ROL:
-			|	ASM_REG_IMM_OP rol, insn->type, def_reg, shift
-			break;
-		case IR_ROR:
-			|	ASM_REG_IMM_OP ror, insn->type, def_reg, shift
-			break;
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
+	return used_stack;
 }

-static void ir_emit_mem_shift_const(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_proto_t *proto, const ir_call_conv_dsc *cc, ir_op op, ir_reg tmp_reg)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *op_insn = &ctx->ir_base[insn->op3];
-	ir_type type = op_insn->type;
-	int32_t shift;
-	ir_mem mem;
+	int j, n;
+	ir_ref arg;
+	ir_insn *arg_insn;
+	uint8_t type;
+	ir_reg src_reg, dst_reg;
+	int int_param = 0;
+	int fp_param = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param = 0;
+#endif
+	int count = 0;
+	int32_t used_stack, copy_stack = 0, stack_offset = cc->shadow_store_size;
+	ir_copy *copies;
+	bool do_pass3 = 0;
+	/* For temporaries we may use any scratch registers except for registers used for parameters */
+	ir_reg tmp_fp_reg = IR_REG_FP_LAST; /* Temporary register for FP loads and swap */

-	IR_ASSERT(IR_IS_CONST_REF(op_insn->op2));
-	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[op_insn->op2].op));
-	IR_ASSERT(IR_IS_SIGNED_32BIT(ctx->ir_base[op_insn->op2].val.i64));
-	shift = ctx->ir_base[op_insn->op2].val.i32;
-	if (insn->op == IR_STORE) {
-		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
-	} else {
-		IR_ASSERT(insn->op == IR_VSTORE);
-		mem = ir_var_spill_slot(ctx, insn->op2);
+	n = insn->inputs_count;
+	if (n < 3) {
+		return 0;
 	}

-	switch (op_insn->op) {
-		default:
-			IR_ASSERT(0);
-		case IR_SHL:
-			|	ASM_MEM_IMM_OP shl, type, mem, shift
-			break;
-		case IR_SHR:
-			|	ASM_MEM_IMM_OP shr, type, mem, shift
-			break;
-		case IR_SAR:
-			|	ASM_MEM_IMM_OP sar, type, mem, shift
-			break;
-		case IR_ROL:
-			|	ASM_MEM_IMM_OP rol, type, mem, shift
-			break;
-		case IR_ROR:
-			|	ASM_MEM_IMM_OP ror, type, mem, shift
-			break;
+	if (tmp_reg == IR_REG_NONE) {
+		tmp_reg = IR_REG_RAX;
 	}
-}
-
-static void ir_emit_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-
-	IR_ASSERT(def_reg != IR_REG_NONE);

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, def_reg, op1_reg);
+	if (op == IR_CALL
+	 && (ctx->flags2 & IR_PREALLOCATED_STACK)
+	 && !cc->cleanup_stack_by_callee) {
+		if (!cc->pass_struct_by_val) {
+			used_stack = ir_call_used_stack(ctx, insn, cc, &copy_stack);
 		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
+			used_stack = 0;
 		}
-	}
-	if (rule == IR_INC) {
-		|	ASM_REG_OP inc, insn->type, def_reg
-	} else if (rule == IR_DEC) {
-		|	ASM_REG_OP dec, insn->type, def_reg
-	} else if (insn->op == IR_NOT) {
-		|	ASM_REG_OP not, insn->type, def_reg
-	} else if (insn->op == IR_NEG) {
-		|	ASM_REG_OP neg, insn->type, def_reg
 	} else {
-		IR_ASSERT(insn->op == IR_BSWAP);
-		switch (ir_type_size[insn->type]) {
-			default:
-				IR_ASSERT(0);
-			case 4:
-				|	bswap Rd(def_reg)
-				break;
-			case 8:
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	bswap Rq(def_reg)
-|.endif
-				break;
+		used_stack = ir_call_used_stack(ctx, insn, cc, &copy_stack);
+		if (cc->shadow_store_size
+		 && op == IR_TAILCALL
+		 && used_stack == cc->shadow_store_size) {
+			used_stack = 0;
+		}
+		if (ctx->fixed_call_stack_size
+		 && used_stack <= ctx->fixed_call_stack_size
+		 && !cc->cleanup_stack_by_callee) {
+			used_stack = 0;
+		} else {
+			/* Stack must be 16 byte aligned */
+			int32_t aligned_stack = IR_ALIGNED_SIZE(used_stack, 16);
+			ctx->call_stack_size += aligned_stack;
+			if (aligned_stack) {
+				ir_stack_alloca(ctx, aligned_stack);
+			}
 		}
 	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
-}

-static void ir_emit_bit_count(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	if (copy_stack) {
+		/* Copy struct arguments */
+		IR_ASSERT(sizeof(void*) == 8);
+|.if X64
+		int copy_stack_offset = 0;

-	IR_ASSERT(def_reg != IR_REG_NONE);
+		for (j = 3; j <= n; j++) {
+			arg = ir_insn_op(insn, j);
+			src_reg = ir_get_alocated_reg(ctx, def, j);
+			arg_insn = &ctx->ir_base[arg];
+			type = arg_insn->type;

-	if (op1_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, type, op1_reg, op1);
+			if (arg_insn->op == IR_ARGVAL) {
+				/* make a stack copy */
+				int size = arg_insn->op2;
+				int align = arg_insn->op3;
+
+				copy_stack_offset += size;
+				align = IR_MAX((int)sizeof(void*), align);
+				copy_stack_offset = IR_ALIGNED_SIZE(copy_stack_offset, align);
+				src_reg = ctx->regs[arg][1];
+
+				|	lea	rdi, [rsp + (used_stack - copy_stack_offset)]
+				if (src_reg != IR_REG_NONE) {
+					if (IR_REG_SPILLED(src_reg)) {
+						src_reg = IR_REG_NUM(src_reg);
+						ir_emit_load(ctx, IR_ADDR, src_reg, arg_insn->op1);
+					}
+					|	mov rsi, Ra(src_reg)
+				} else {
+					ir_emit_load(ctx, IR_ADDR, IR_REG_RSI, arg_insn->op1);
+				}
+				ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_RCX, size);
+				|	rep; movsb
+			}
 		}
-		switch (ir_type_size[insn->type]) {
-			default:
-				IR_ASSERT(0);
-			case 2:
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	lzcnt Rw(def_reg), Rw(op1_reg)
+|.endif
+	}
+
+	/* 1. move all register arguments that should be passed through stack
+	 *    and collect arguments that should be passed through registers */
+	copies = ir_mem_malloc((n - 2) * sizeof(ir_copy));
+	for (j = 3; j <= n; j++) {
+		arg = ir_insn_op(insn, j);
+		src_reg = ir_get_alocated_reg(ctx, def, j);
+		arg_insn = &ctx->ir_base[arg];
+		type = arg_insn->type;
+		if (IR_IS_TYPE_INT(type)) {
+			if (arg_insn->op == IR_ARGVAL && cc->pass_struct_by_val) {
+				int size = arg_insn->op2;
+				int align = arg_insn->op3;
+				align = IR_MAX((int)sizeof(void*), align);
+				stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
+				if (size) {
+					src_reg = ctx->regs[arg][1];
+					if (src_reg != IR_REG_NONE) {
+						if (IR_REG_SPILLED(src_reg)) {
+							src_reg = IR_REG_NUM(src_reg);
+							ir_emit_load(ctx, IR_ADDR, src_reg, arg_insn->op1);
+						}
+						if (src_reg != IR_REG_RSI) {
+							|.if X64
+							|	mov rsi, Ra(src_reg)
+							|.else
+							|	mov	esi, Ra(src_reg)
+							|.endif
+						}
 					} else {
-						|	bsr Rw(def_reg), Rw(op1_reg)
-						|	xor Rw(def_reg), 0xf
+						ir_emit_load(ctx, IR_ADDR, IR_REG_RSI, arg_insn->op1);
 					}
-				} else if (insn->op == IR_CTTZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	tzcnt Rw(def_reg), Rw(op1_reg)
+					if (stack_offset == 0) {
+						|.if X64
+						|	mov	rdi, rsp
+						|.else
+						|	mov	edi, esp
+						|.endif
 					} else {
-						|	bsf Rw(def_reg), Rw(op1_reg)
+						|.if X64
+						|	lea	rdi, [rsp+stack_offset]
+						|.else
+						|	lea	edi, [esp+stack_offset]
+						|.endif
 					}
-				} else {
-					IR_ASSERT(insn->op == IR_CTPOP);
-					|	popcnt Rw(def_reg), Rw(op1_reg)
+					|.if X64
+					|	mov rcx, size
+					|	rep; movsb
+					|.else
+					|	mov ecx, size
+					|	rep; movsb
+					|.endif
 				}
-				break;
-			case 1:
-				|   movzx Rd(op1_reg), Rb(op1_reg)
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	lzcnt Rd(def_reg), Rd(op1_reg)
-						|	sub Rd(def_reg), 24
+				stack_offset += size;
+				stack_offset = IR_ALIGNED_SIZE(stack_offset, sizeof(void*));
+				continue;
+			}
+			if (int_param < cc->int_param_regs_count) {
+				dst_reg = cc->int_param_regs[int_param];
+#if IR_X86_I64
+				if (dst_reg != IR_REG_NONE && (type == IR_I64 || type == IR_U64)) {
+					if (int_param + 1 < cc->int_param_regs_count) {
+						int_param++;
+						if (cc->shadow_param_regs) {
+							fp_param++;
+						}
+					}
+					dst_reg = IR_REG_NONE;
+				}
+#endif
+			} else {
+				dst_reg = IR_REG_NONE; /* pass argument through stack */
+			}
+			int_param++;
+			if (cc->shadow_param_regs) {
+				fp_param++;
+			}
+			if (arg_insn->op == IR_ARGVAL && !cc->pass_struct_by_val) {
+				do_pass3 = 3;
+				continue;
+			}
+#if IR_SIMD && defined(IR_TARGET_X86)
+		} else if (IR_IS_TYPE_VECTOR(type)) {
+			if (vector_param < cc->vector_param_regs_count) {
+				dst_reg = cc->vector_param_regs[vector_param];
+			} else {
+				dst_reg = IR_REG_NONE; /* pass argument through stack */
+			}
+			vector_param++;
+#endif
+		} else {
+			IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
+			if (fp_param < cc->fp_param_regs_count) {
+				dst_reg = cc->fp_param_regs[fp_param];
+			} else {
+				dst_reg = IR_REG_NONE; /* pass argument through stack */
+			}
+			fp_param++;
+			if (cc->shadow_param_regs) {
+				int_param++;
+			}
+		}
+		if (dst_reg != IR_REG_NONE) {
+			if (IR_IS_CONST_REF(arg) ||
+			    src_reg == IR_REG_NONE ||
+			    (IR_REG_SPILLED(src_reg) && !IR_REGSET_IN(cc->preserved_regs, IR_REG_NUM(src_reg)))) {
+				/* delay CONST->REG and MEM->REG moves to third pass */
+				do_pass3 = 1;
+			} else {
+				if (IR_REG_SPILLED(src_reg)) {
+					src_reg = IR_REG_NUM(src_reg);
+					ir_emit_load(ctx, type, src_reg, arg);
+				}
+				if (src_reg != dst_reg) {
+					/* delay REG->REG moves to second pass */
+					copies[count].type = type;
+					copies[count].from = src_reg;
+					copies[count].to = dst_reg;
+					count++;
+				}
+			}
+		} else {
+			/* Pass register arguments to stack (REG->MEM moves) */
+			if (!IR_IS_CONST_REF(arg) && src_reg != IR_REG_NONE && !IR_REG_SPILLED(src_reg)) {
+#if IR_X86_I64
+				if (type == IR_I64 || type == IR_U64) {
+					ir_mem mem = IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset);
+					ir_mem mem_hi = IR_MEM_I64_HI(mem);
+					ir_reg src_reg_hi = IR_REG_I64_HI(src_reg);
+					src_reg = IR_REG_I64_LO(src_reg);
+					ir_emit_store_mem(ctx, IR_U32, mem, src_reg);
+					ir_emit_store_mem(ctx, IR_U32, mem_hi, src_reg_hi);
+				} else
+#endif
+				ir_emit_store_mem(ctx, type, IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset), src_reg);
+			} else {
+				do_pass3 = 1;
+			}
+			stack_offset += IR_MAX(sizeof(void*), ir_get_type_size(type));
+		}
+	}
+
+	/* 2. move all arguments that should be passed from one register to another (REG->REG movs) */
+	if (count) {
+		ir_parallel_copy(ctx, copies, count, tmp_reg, tmp_fp_reg);
+	}
+	ir_mem_free(copies);
+
+	/* 3. move the remaining memory and immediate values */
+	if (do_pass3) {
+		int copy_stack_offset = 0;
+
+		stack_offset = cc->shadow_store_size;
+		int_param = 0;
+		fp_param = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+		vector_param = 0;
+#endif
+		for (j = 3; j <= n; j++) {
+			arg = ir_insn_op(insn, j);
+			src_reg = ir_get_alocated_reg(ctx, def, j);
+			arg_insn = &ctx->ir_base[arg];
+			type = arg_insn->type;
+			if (IR_IS_TYPE_INT(type)) {
+				if (arg_insn->op == IR_ARGVAL) {
+					int size = arg_insn->op2;
+					int align = arg_insn->op3;
+
+					if (cc->pass_struct_by_val) {
+						align = IR_MAX((int)sizeof(void*), align);
+						stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
+						stack_offset += size;
+						stack_offset = IR_ALIGNED_SIZE(stack_offset, sizeof(void*));
+						continue;
 					} else {
-						|	bsr Rd(def_reg), Rd(op1_reg)
-						|	xor Rw(def_reg), 0x7
+						/* pass pointer to the copy on stack */
+						copy_stack_offset += size;
+						align = IR_MAX((int)sizeof(void*), align);
+						copy_stack_offset = IR_ALIGNED_SIZE(copy_stack_offset, align);
+						if (int_param < cc->int_param_regs_count) {
+							dst_reg = cc->int_param_regs[int_param];
+							|	lea Ra(dst_reg), [r4 + (used_stack - copy_stack_offset)]
+						} else {
+							|	lea Ra(tmp_reg), [r4 + (used_stack - copy_stack_offset)]
+							ir_emit_store_mem_int(ctx, IR_ADDR, IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset), tmp_reg);
+							stack_offset += sizeof(void*);
+						}
+						int_param++;
+						if (cc->shadow_param_regs) {
+							fp_param++;
+						}
+						continue;
 					}
-					break;
 				}
-				IR_FALLTHROUGH;
-			case 4:
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	lzcnt Rd(def_reg), Rd(op1_reg)
-					} else {
-						|	bsr Rd(def_reg), Rd(op1_reg)
-						|	xor Rw(def_reg), 0x1f
-					}
-				} else if (insn->op == IR_CTTZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	tzcnt Rd(def_reg), Rd(op1_reg)
-					} else {
-						|	bsf Rd(def_reg), Rd(op1_reg)
+				if (int_param < cc->int_param_regs_count) {
+					dst_reg = cc->int_param_regs[int_param];
+#if IR_X86_I64
+					if (dst_reg != IR_REG_NONE && (type == IR_I64 || type == IR_U64)) {
+						if (int_param + 1 < cc->int_param_regs_count) {
+							int_param++;
+							if (cc->shadow_param_regs) {
+								fp_param++;
+							}
+						}
+						dst_reg = IR_REG_NONE;
 					}
+#endif
 				} else {
-					IR_ASSERT(insn->op == IR_CTPOP);
-					|	popcnt Rd(def_reg), Rd(op1_reg)
+					dst_reg = IR_REG_NONE; /* argument already passed through stack */
 				}
-				break;
-|.if X64
-			case 8:
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	lzcnt Rq(def_reg), Rq(op1_reg)
-					} else {
-						|	bsr Rq(def_reg), Rq(op1_reg)
-						|	xor Rw(def_reg), 0x3f
-					}
-				} else if (insn->op == IR_CTTZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	tzcnt Rq(def_reg), Rq(op1_reg)
-					} else {
-						|	bsf Rq(def_reg), Rq(op1_reg)
-					}
-				} else {
-					IR_ASSERT(insn->op == IR_CTPOP);
-					|	popcnt Rq(def_reg), Rq(op1_reg)
+				int_param++;
+				if (cc->shadow_param_regs) {
+					fp_param++;
 				}
-				break;
-|.endif
-		}
-	} else {
-		ir_mem mem;
-
-		if (ir_rule(ctx, op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, op1);
-		} else {
-			mem = ir_ref_spill_slot(ctx, op1);
-		}
-		switch (ir_type_size[insn->type]) {
-			default:
-				IR_ASSERT(0);
-			case 2:
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	ASM_TXT_TMEM_OP lzcnt, Rw(def_reg), word, mem
-					} else {
-						|	ASM_TXT_TMEM_OP bsr, Rw(def_reg), word, mem
-						|	xor Rw(def_reg), 0xf
-					}
-				} else if (insn->op == IR_CTTZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	ASM_TXT_TMEM_OP tzcnt, Rw(def_reg), word, mem
-					} else {
-						|	ASM_TXT_TMEM_OP bsf, Rw(def_reg), word, mem
-					}
+#if IR_SIMD && defined(IR_TARGET_X86)
+			} else if (IR_IS_TYPE_VECTOR(type)) {
+				if (vector_param < cc->vector_param_regs_count) {
+					dst_reg = cc->vector_param_regs[vector_param];
 				} else {
-					|	ASM_TXT_TMEM_OP popcnt, Rw(def_reg), word, mem
+					dst_reg = IR_REG_NONE; /* pass argument through stack */
 				}
-				break;
-			case 4:
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	ASM_TXT_TMEM_OP lzcnt, Rd(def_reg), dword, mem
-					} else {
-						|	ASM_TXT_TMEM_OP bsr, Rd(def_reg), dword, mem
-						|	xor Rw(def_reg), 0x1f
-					}
-				} else if (insn->op == IR_CTTZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	ASM_TXT_TMEM_OP tzcnt, Rd(def_reg), dword, mem
-					} else {
-						|	ASM_TXT_TMEM_OP bsf, Rd(def_reg), dword, mem
-					}
+				vector_param++;
+#endif
+			} else {
+				IR_ASSERT(IR_IS_TYPE_FP(type) || IR_IS_TYPE_VECTOR(type));
+				if (fp_param < cc->fp_param_regs_count) {
+					dst_reg = cc->fp_param_regs[fp_param];
 				} else {
-					|	ASM_TXT_TMEM_OP popcnt, Rd(def_reg), dword, mem
+					dst_reg = IR_REG_NONE; /* argument already passed through stack */
 				}
-				break;
-|.if X64
-			case 8:
-				if (insn->op == IR_CTLZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	ASM_TXT_TMEM_OP lzcnt, Rq(def_reg), qword, mem
-					} else {
-						|	ASM_TXT_TMEM_OP bsr, Rq(def_reg), qword, mem
-						|	xor Rw(def_reg), 0x3f
-					}
-				} else if (insn->op == IR_CTTZ) {
-					if (ctx->mflags & IR_X86_BMI1) {
-						|	ASM_TXT_TMEM_OP tzcnt, Rq(def_reg), qword, mem
+				fp_param++;
+				if (cc->shadow_param_regs) {
+					int_param++;
+				}
+			}
+			if (dst_reg != IR_REG_NONE) {
+				if (IR_IS_CONST_REF(arg) ||
+				    src_reg == IR_REG_NONE ||
+				    (IR_REG_SPILLED(src_reg) && !IR_REGSET_IN(cc->preserved_regs, IR_REG_NUM(src_reg)))) {
+					if (IR_IS_TYPE_INT(type)) {
+						if (IR_IS_CONST_REF(arg)) {
+							if (type == IR_I8 || type == IR_I16) {
+								type = IR_I32;
+							} else if (type == IR_U8 || type == IR_U16) {
+								type = IR_U32;
+							}
+							ir_emit_load(ctx, type, dst_reg, arg);
+						} else if (ctx->vregs[arg]) {
+							ir_mem mem = ir_ref_spill_slot(ctx, arg);
+							uint32_t size = ir_type_size[type];
+
+							if (size > 2) {
+								ir_emit_load_mem_int(ctx, type, dst_reg, mem);
+							} else if (size == 2) {
+								if (type == IR_I16) {
+									|	ASM_TXT_TMEM_OP movsx, Rd(dst_reg), word, mem
+								} else {
+									|	ASM_TXT_TMEM_OP movzx, Rd(dst_reg), word, mem
+								}
+							} else {
+								IR_ASSERT(size == 1);
+								if (type == IR_I8) {
+									|	ASM_TXT_TMEM_OP movsx, Rd(dst_reg), byte, mem
+								} else {
+									|	ASM_TXT_TMEM_OP movzx, Rd(dst_reg), byte, mem
+								}
+							}
+						} else {
+							ir_load_local_addr(ctx, dst_reg, arg);
+						}
 					} else {
-						|	ASM_TXT_TMEM_OP bsf, Rq(def_reg), qword, mem
+						ir_emit_load(ctx, type, dst_reg, arg);
 					}
-				} else {
-					|	ASM_TXT_TMEM_OP popcnt, Rq(def_reg), qword, mem
 				}
-				break;
-|.endif
-		}
-	}
+			} else {
+				ir_mem mem = IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset);

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
-}
+				if (IR_IS_TYPE_INT(type)) {
+#if IR_X86_I64
+					if (type == IR_I64 || type == IR_U64) {
+						ir_mem mem_hi = IR_MEM_I64_HI(mem);

-static void ir_emit_ctpop(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg tmp_reg = ctx->regs[def][2];
-|.if X64
-||	ir_reg const_reg = ctx->regs[def][3];
-|.endif
+						if (IR_IS_CONST_REF(arg)) {
+							ir_insn *val = &ctx->ir_base[arg];
+							|	ASM_TMEM_TXT_OP mov, dword, mem, val->val.u32;
+							|	ASM_TMEM_TXT_OP mov, dword, mem_hi, val->val.u32_hi;
+						} else if (src_reg == IR_REG_NONE) {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							ir_emit_load_i64_lo(ctx, tmp_reg, arg);
+							ir_emit_store_mem_int(ctx, IR_U32, mem, tmp_reg);
+							ir_emit_load_i64_hi(ctx, tmp_reg, arg);
+							ir_emit_store_mem_int(ctx, IR_U32, mem_hi, tmp_reg);
+						} else if (IR_REG_SPILLED(src_reg)) {
+							ir_reg src_reg_hi;

-	IR_ASSERT(def_reg != IR_REG_NONE && tmp_reg != IR_REG_NONE);
-	if (op1_reg == IR_REG_NONE) {
-		ir_emit_load(ctx, type, def_reg, op1);
-		if (ir_type_size[insn->type] == 1) {
-			|	movzx Rd(def_reg), Rb(def_reg)
-		} else if (ir_type_size[insn->type] == 2) {
-			|	movzx Rd(def_reg), Rw(def_reg)
-		}
-	} else {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, type, op1_reg, op1);
-		}
-		switch (ir_type_size[insn->type]) {
-			default:
-				IR_ASSERT(0);
-			case 1:
-				|	movzx Rd(def_reg), Rb(op1_reg)
-				break;
-			case 2:
-				|	movzx Rd(def_reg), Rw(op1_reg)
-				break;
-			case 4:
-				|	mov Rd(def_reg), Rd(op1_reg)
-				break;
-|.if X64
-||			case 8:
-				|	mov Rq(def_reg), Rq(op1_reg)
-||				break;
-|.endif
+							src_reg = IR_REG_NUM(src_reg);
+							src_reg_hi = IR_REG_I64_HI(src_reg);
+							src_reg = IR_REG_I64_LO(src_reg);
+							ir_emit_load_i64_lo(ctx, src_reg, arg);
+							ir_emit_load_i64_hi(ctx, src_reg_hi, arg);
+							ir_emit_store_mem_int(ctx, IR_U32, mem, src_reg);
+							ir_emit_store_mem_int(ctx, IR_U32, mem_hi, src_reg_hi);
+						}
+					} else
+#endif
+					if (IR_IS_CONST_REF(arg)) {
+						ir_emit_store_mem_int_const(ctx, type, mem, arg, tmp_reg, 1);
+					} else if (src_reg == IR_REG_NONE) {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						ir_emit_load(ctx, type, tmp_reg, arg);
+						ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
+					} else if (IR_REG_SPILLED(src_reg)) {
+						src_reg = IR_REG_NUM(src_reg);
+						ir_emit_load(ctx, type, src_reg, arg);
+						ir_emit_store_mem_int(ctx, type, mem, src_reg);
+					}
+				} else {
+					if (IR_IS_CONST_REF(arg)) {
+						ir_emit_store_mem_fp_const(ctx, type, mem, arg, tmp_reg, tmp_fp_reg);
+					} else if (src_reg == IR_REG_NONE) {
+						IR_ASSERT(tmp_fp_reg != IR_REG_NONE);
+						ir_emit_load(ctx, type, tmp_fp_reg, arg);
+						ir_emit_store_mem_fp(ctx, type, mem, tmp_fp_reg);
+					} else if (IR_REG_SPILLED(src_reg)) {
+						src_reg = IR_REG_NUM(src_reg);
+						ir_emit_load(ctx, type, src_reg, arg);
+						ir_emit_store_mem_fp(ctx, type, mem, src_reg);
+					}
+				}
+				stack_offset += IR_MAX(sizeof(void*), ir_get_type_size(type));
+			}
 		}
 	}
-	switch (ir_type_size[insn->type]) {
-		default:
-			IR_ASSERT(0);
-		case 1:
-			|	mov Rd(tmp_reg), Rd(def_reg)
-			|	shr Rd(def_reg), 1
-			|	and Rd(def_reg), 0x55
-			|	sub Rd(tmp_reg), Rd(def_reg)
-			|	mov Rd(def_reg), Rd(tmp_reg)
-			|	and Rd(def_reg), 0x33
-			|	shr Rd(tmp_reg), 2
-			|	and Rd(tmp_reg), 0x33
-			|	add Rd(tmp_reg), Rd(def_reg)
-			|	mov Rd(def_reg), Rd(tmp_reg)
-			|	shr Rd(def_reg), 4
-			|	add Rd(def_reg), Rd(tmp_reg)
-			|	and Rd(def_reg), 0x0f
-			break;
-		case 2:
-			|	mov Rd(tmp_reg), Rd(def_reg)
-			|	shr Rd(def_reg), 1
-			|	and Rd(def_reg), 0x5555
-			|	sub Rd(tmp_reg), Rd(def_reg)
-			|	mov Rd(def_reg), Rd(tmp_reg)
-			|	and Rd(def_reg), 0x3333
-			|	shr Rd(tmp_reg), 2
-			|	and Rd(tmp_reg), 0x3333
-			|	add Rd(tmp_reg), Rd(def_reg)
-			|	mov Rd(def_reg), Rd(tmp_reg)
-			|	shr Rd(def_reg), 4
-			|	add Rd(def_reg), Rd(tmp_reg)
-			|	and Rd(def_reg), 0x0f0f
-			|	mov	Rd(tmp_reg), Rd(def_reg)
-			|	shr Rd(tmp_reg), 8
-			|	and Rd(def_reg), 0x0f
-			|	add Rd(def_reg), Rd(tmp_reg)
-			break;
-		case 4:
-			|	mov Rd(tmp_reg), Rd(def_reg)
-			|	shr Rd(def_reg), 1
-			|	and Rd(def_reg), 0x55555555
-			|	sub Rd(tmp_reg), Rd(def_reg)
-			|	mov Rd(def_reg), Rd(tmp_reg)
-			|	and Rd(def_reg), 0x33333333
-			|	shr Rd(tmp_reg), 2
-			|	and Rd(tmp_reg), 0x33333333
-			|	add Rd(tmp_reg), Rd(def_reg)
-			|	mov Rd(def_reg), Rd(tmp_reg)
-			|	shr Rd(def_reg), 4
-			|	add Rd(def_reg), Rd(tmp_reg)
-			|	and Rd(def_reg), 0x0f0f0f0f
-			|	imul Rd(def_reg), 0x01010101
-			|	shr Rd(def_reg), 24
-			break;
+
+	/* WIN64 calling convention requires duplcation of parameters passed in FP register into GP ones */
+	if (proto && (proto->flags & IR_VARARG_FUNC) && cc->shadow_param_regs) {
+		n = IR_MIN(n, IR_MIN(cc->int_param_regs_count, cc->fp_param_regs_count) + 2);
+		for (j = 3; j <= n; j++) {
+			arg = ir_insn_op(insn, j);
+			arg_insn = &ctx->ir_base[arg];
+			type = arg_insn->type;
+			if (IR_IS_TYPE_FP(type)) {
+				src_reg = cc->fp_param_regs[j-3];
+				dst_reg = cc->int_param_regs[j-3];
 |.if X64
-||		case 8:
-||			IR_ASSERT(const_reg != IR_REG_NONE);
-			|	mov Rq(tmp_reg), Rq(def_reg)
-			|	shr Rq(def_reg), 1
-			|	mov64 Rq(const_reg), 0x5555555555555555
-			|	and Rq(def_reg), Rq(const_reg)
-			|	sub Rq(tmp_reg), Rq(def_reg)
-			|	mov Rq(def_reg), Rq(tmp_reg)
-			|	mov64 Rq(const_reg), 0x3333333333333333
-			|	and Rq(def_reg), Rq(const_reg)
-			|	shr Rq(tmp_reg), 2
-			|	and Rq(tmp_reg), Rq(const_reg)
-			|	add Rq(tmp_reg), Rq(def_reg)
-			|	mov Rq(def_reg), Rq(tmp_reg)
-			|	shr Rq(def_reg), 4
-			|	add Rq(def_reg), Rq(tmp_reg)
-			|	mov64 Rq(const_reg), 0x0f0f0f0f0f0f0f0f
-			|	and Rq(def_reg), Rq(const_reg)
-			|	mov64 Rq(const_reg), 0x0101010101010101
-			|	imul Rq(def_reg), Rq(const_reg)
-			|	shr Rq(def_reg), 56
-||			break;
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vmovq Rq(dst_reg), xmm(src_reg-IR_REG_FP_FIRST)
+				} else {
+					|	movq Rq(dst_reg), xmm(src_reg-IR_REG_FP_FIRST)
+				}
 |.endif
+			}
+		}
 	}

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+	if (op == IR_CALL && (ctx->flags2 & IR_PREALLOCATED_STACK) && !cc->cleanup_stack_by_callee) {
+		used_stack = 0;
+	}
+
+	if (proto && (proto->flags & IR_VARARG_FUNC) && cc->fp_varargs_reg != IR_REG_NONE) {
+		/* set hidden argument to specify the number of vector registers used */
+		fp_param = IR_MIN(fp_param, cc->fp_param_regs_count);
+		if (fp_param) {
+			|	mov Rd(cc->fp_varargs_reg), fp_param
+		} else {
+			|	xor Rd(cc->fp_varargs_reg), Rd(cc->fp_varargs_reg)
+		}
 	}
+
+	return used_stack;
 }

-static void ir_emit_mem_op_int(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
+static void ir_emit_call_ex(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_proto_t *proto, const ir_call_conv_dsc *cc, int32_t used_stack)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *op_insn = &ctx->ir_base[insn->op3];
-	ir_type type = op_insn->type;
-	ir_mem mem;
+	ir_reg def_reg;
+	ir_ref func = insn->op2;

-	if (insn->op == IR_STORE) {
-		mem = ir_fuse_mem(ctx, def, def, insn, ctx->regs[def][2]);
-	} else {
-		IR_ASSERT(insn->op == IR_VSTORE);
-		mem = ir_var_spill_slot(ctx, insn->op2);
+	if (!IR_IS_CONST_REF(func) && ctx->rules[func] == (IR_FUSED | IR_SIMPLE | IR_PROTO)) {
+		func = ctx->ir_base[func].op1;
 	}
+	if (IR_IS_CONST_REF(func)) {
+		void *addr = ir_call_addr(ctx, insn, &ctx->ir_base[func]);

-	if (rule == IR_MEM_INC) {
-		|	ASM_MEM_OP inc, type, mem
-	} else if (rule == IR_MEM_DEC) {
-		|	ASM_MEM_OP dec, type, mem
-	} else if (op_insn->op == IR_NOT) {
-		|	ASM_MEM_OP not, type, mem
-	} else {
-		IR_ASSERT(op_insn->op == IR_NEG);
-		|	ASM_MEM_OP neg, type, mem
-	}
-}
+		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+			|	call aword &addr
+		} else {
+|.if X64
+||			ir_reg tmp_reg = cc->int_ret_reg;
+||
+||			if (proto && (proto->flags & IR_VARARG_FUNC) && tmp_reg == cc->fp_varargs_reg) {
+||				tmp_reg = IR_REG_R11; // TODO: avoid usage of hardcoded temporary register ???
+||			}
+||			if (IR_IS_SIGNED_32BIT(addr)) {
+				|	mov Rq(tmp_reg), ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+||			} else {
+				|	mov64 Rq(tmp_reg), ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+||			}
+			|	call Rq(tmp_reg)
+|.endif
+		}
+    } else {
+		ir_reg op2_reg = ctx->regs[def][2];

-static void ir_emit_abs_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, IR_ADDR, op2_reg, func);
+			}
+			|	call Ra(op2_reg)
+		} else {
+			ir_mem mem;

-	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+			if (ir_rule(ctx, func) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, func);
+			} else {
+				mem = ir_ref_spill_slot(ctx, func);
+			}

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+			|	ASM_TMEM_OP call, aword, mem
+		}
+    }
+
+	if (used_stack) {
+		int32_t aligned_stack = IR_ALIGNED_SIZE(used_stack, 16);
+
+		ctx->call_stack_size -= aligned_stack;
+		if (cc->cleanup_stack_by_callee) {
+			aligned_stack -= used_stack;
+			if (aligned_stack) {
+				|	add Ra(IR_REG_RSP), aligned_stack
+			}
+		} else {
+			|	add Ra(IR_REG_RSP), aligned_stack
+		}
 	}

-	IR_ASSERT(def_reg != op1_reg);
+	if (insn->type != IR_VOID) {
+		if (IR_IS_TYPE_INT(insn->type)) {
+			def_reg = IR_REG_NUM(ctx->regs[def][0]);
+			if (def_reg != IR_REG_NONE) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					ir_reg def_reg_hi = IR_REG_I64_HI(def_reg);
+
+					def_reg = IR_REG_I64_LO(def_reg);
+					IR_ASSERT(cc->int_ret2_reg != IR_REG_NONE);
+					if (def_reg != cc->int_ret_reg) {
+						ir_emit_mov(ctx, IR_U32, def_reg, cc->int_ret_reg);
+					}
+					if (def_reg_hi != cc->int_ret2_reg) {
+						ir_emit_mov(ctx, IR_U32, def_reg_hi, cc->int_ret2_reg);
+					}
+					if (IR_REG_SPILLED(ctx->regs[def][0])) {
+						ir_emit_store_i64_lo(ctx, def, def_reg);
+						ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+					}
+				} else
+#endif
+				{
+					if (def_reg != cc->int_ret_reg) {
+						ir_emit_mov(ctx, insn->type, def_reg, cc->int_ret_reg);
+					}
+					if (IR_REG_SPILLED(ctx->regs[def][0])) {
+						ir_emit_store(ctx, insn->type, def, def_reg);
+					}
+				}
+			} else if (ctx->use_lists[def].count > 1) {
+#if IR_X86_I64
+				if (insn->type == IR_I64 || insn->type == IR_U64) {
+					IR_ASSERT(cc->int_ret2_reg != IR_REG_NONE);
+					ir_emit_store_i64_lo(ctx, def, cc->int_ret_reg);
+					ir_emit_store_i64_hi(ctx, def, cc->int_ret2_reg);
+				} else
+#endif
+				ir_emit_store(ctx, insn->type, def, cc->int_ret_reg);
+			}
+		} else {
+			ir_reg ret_reg = cc->fp_ret_reg;

-	ir_emit_mov(ctx, insn->type, def_reg, op1_reg);
-	|	ASM_REG_OP neg, insn->type, def_reg
-	|	ASM_REG_REG_OP2, cmovs, type, def_reg, op1_reg
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+			IR_ASSERT(IR_IS_TYPE_FP(insn->type) || IR_IS_TYPE_VECTOR(insn->type));
+#if IR_SIMD && defined(IR_TARGET_X86)
+			if (IR_IS_TYPE_VECTOR(insn->type)) {
+				ret_reg = cc->vector_ret_reg;
+			}
+#endif
+			def_reg = IR_REG_NUM(ctx->regs[def][0]);
+			if (ret_reg != IR_REG_NONE) {
+				if (def_reg != IR_REG_NONE) {
+					if (def_reg != ret_reg) {
+						ir_emit_fp_mov(ctx, insn->type, def_reg, ret_reg);
+					}
+					if (IR_REG_SPILLED(ctx->regs[def][0])) {
+						ir_emit_store(ctx, insn->type, def, def_reg);
+					}
+				} else if (ctx->use_lists[def].count > 1) {
+					ir_emit_store(ctx, insn->type, def, ret_reg);
+				}
+			}
+#ifdef IR_TARGET_X86
+			if (insn->op == IR_TAILCALL) {
+				/* pass */
+			} else if (ctx->use_lists[def].count > 1 && ret_reg == IR_REG_NONE) {
+				int32_t offset;
+				ir_reg fp;
+
+				if (def_reg == IR_REG_NONE) {
+					offset = ir_ref_spill_slot_offset(ctx, def, &fp);
+					if (insn->type == IR_DOUBLE) {
+						|	fstp qword [Ra(fp)+offset]
+					} else {
+						IR_ASSERT(insn->type == IR_FLOAT);
+						|	fstp dword [Ra(fp)+offset]
+					}
+				} else {
+					offset = ctx->ret_slot;
+					IR_ASSERT(offset != -1);
+					offset = IR_SPILL_POS_TO_OFFSET(offset);
+					fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+					if (insn->type == IR_DOUBLE) {
+						|	fstp qword [Ra(fp)+offset]
+					} else {
+						IR_ASSERT(insn->type == IR_FLOAT);
+						|	fstp dword [Ra(fp)+offset]
+					}
+					ir_emit_load_mem_fp(ctx, insn->type, def_reg, IR_MEM_BO(fp, offset));
+					if (IR_REG_SPILLED(ctx->regs[def][0])) {
+						ir_emit_store(ctx, insn->type, def, def_reg);
+					}
+				}
+			} else {
+				/* discard return value on top of FPU stack */
+				|	fstp st0
+			}
+#endif
+		}
 	}
 }

-static void ir_emit_bool_not(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_call(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = ctx->ir_base[insn->op1].type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-
-	IR_ASSERT(def_reg != IR_REG_NONE);
-
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-
-	if (def_reg != op1_reg) {
-		|	mov Rb(def_reg), Rb(op1_reg)
-	}
-
-	|	xor Rb(def_reg), 1
-
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
+	const ir_proto_t *proto = ir_call_proto(ctx, insn);
+	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
+	int32_t used_stack = ir_emit_arguments(ctx, def, insn, proto, cc, IR_CALL, ctx->regs[def][1]);
+	ir_emit_call_ex(ctx, def, insn, proto, cc, used_stack);
 }

-static void ir_emit_bool_not_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_tailcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = ctx->ir_base[insn->op1].type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-
-	IR_ASSERT(def_reg != IR_REG_NONE);
+	const ir_proto_t *proto = ir_call_proto(ctx, insn);
+	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
+	int32_t used_stack = ir_emit_arguments(ctx, def, insn, proto, cc, IR_TAILCALL, ctx->regs[def][1]);
+	ir_ref func = insn->op2;

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+	if (used_stack != 0) {
+		ir_emit_call_ex(ctx, def, insn, proto, cc, used_stack);
+		ir_emit_return_void(ctx);
+		return;
 	}

-	if (op1_reg != IR_REG_NONE) {
-		|	ASM_REG_REG_OP test, type, op1_reg, op1_reg
-	} else {
-		ir_mem mem = ir_ref_spill_slot(ctx, op1);
+	/* Move op2 to a tmp register before epilogue if it's in
+	 * used_preserved_regs, because it will be overridden. */

-		|	ASM_MEM_IMM_OP cmp, type, mem, 0
+	ir_reg op2_reg = IR_REG_NONE;
+	ir_mem mem = IR_MEM_B(IR_REG_NONE);
+	if (!IR_IS_CONST_REF(func) && ctx->rules[func] == (IR_FUSED | IR_SIMPLE | IR_PROTO)) {
+		func = ctx->ir_base[func].op1;
 	}
-	|	sete Rb(def_reg)
+	if (!IR_IS_CONST_REF(func)) {
+		op2_reg = ctx->regs[def][2];

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
-}
+		ir_regset preserved_regs = (ir_regset)ctx->used_preserved_regs | IR_REGSET(IR_REG_STACK_POINTER);
+		if (ctx->flags & IR_USE_FRAME_POINTER) {
+			preserved_regs |= IR_REGSET(IR_REG_FRAME_POINTER);
+		}

-static void ir_emit_mul_div_mod(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];
-	ir_mem mem;
+		bool is_spill_slot = op2_reg != IR_REG_NONE
+			&& IR_REG_SPILLED(op2_reg)
+			&& ctx->vregs[func];

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (op1_reg != IR_REG_RAX) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_mov(ctx, type, IR_REG_RAX, op1_reg);
-		} else {
-			ir_emit_load(ctx, type, IR_REG_RAX, op1);
-		}
-	}
-	if (op2_reg == IR_REG_NONE && op1 == op2) {
-		op2_reg = IR_REG_RAX;
-	} else if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, op2);
-		}
-	} else if (IR_IS_CONST_REF(op2)
-	 && (insn->op == IR_MUL || insn->op == IR_MUL_OV)) {
-		op2_reg = IR_REG_RDX;
-		ir_emit_load(ctx, type, op2_reg, op2);
-	}
-	if (insn->op == IR_MUL || insn->op == IR_MUL_OV) {
-		if (IR_IS_TYPE_SIGNED(insn->type)) {
-			if (op2_reg != IR_REG_NONE) {
-				|	ASM_REG_OP imul, type, op2_reg
-			} else {
-				if (ir_rule(ctx, op2) & IR_FUSED) {
-					mem = ir_fuse_load(ctx, def, op2);
-				} else {
-					mem = ir_ref_spill_slot(ctx, op2);
-				}
-				|	ASM_MEM_OP imul, type, mem
-			}
-		} else {
-			if (op2_reg != IR_REG_NONE) {
-				|	ASM_REG_OP mul, type, op2_reg
-			} else {
-				if (ir_rule(ctx, op2) & IR_FUSED) {
-					mem = ir_fuse_load(ctx, def, op2);
+		if (op2_reg != IR_REG_NONE && !is_spill_slot) {
+			if (IR_REGSET_IN(preserved_regs, IR_REG_NUM(op2_reg))) {
+				ir_ref orig_op2_reg = op2_reg;
+				op2_reg = IR_REG_RAX;
+
+				if (IR_REG_SPILLED(orig_op2_reg)) {
+					ir_emit_load(ctx, IR_ADDR, op2_reg, func);
 				} else {
-					mem = ir_ref_spill_slot(ctx, op2);
+					ir_type type = ctx->ir_base[func].type;
+					| ASM_REG_REG_OP mov, type, op2_reg, IR_REG_NUM(orig_op2_reg)
 				}
-				|	ASM_MEM_OP mul, type, mem
-			}
-		}
-	} else {
-		if (IR_IS_TYPE_SIGNED(type)) {
-			if (ir_type_size[type] == 8) {
-				|	cqo
-			} else if (ir_type_size[type] == 4) {
-				|	cdq
-			} else if (ir_type_size[type] == 2) {
-				|	cwd
-			} else {
-				|	cbw
-			}
-			if (op2_reg != IR_REG_NONE) {
-				|	ASM_REG_OP idiv, type, op2_reg
 			} else {
-				if (ir_rule(ctx, op2) & IR_FUSED) {
-					mem = ir_fuse_load(ctx, def, op2);
-				} else {
-					mem = ir_ref_spill_slot(ctx, op2);
-				}
-				|	ASM_MEM_OP idiv, type, mem
+				op2_reg = IR_REG_NUM(op2_reg);
 			}
 		} else {
-			if (ir_type_size[type] == 1) {
-				|	movzx ax, al
+			if (ir_rule(ctx, func) & IR_FUSED) {
+				IR_ASSERT(op2_reg == IR_REG_NONE);
+				mem = ir_fuse_load(ctx, def, func);
 			} else {
-				|	ASM_REG_REG_OP xor, type, IR_REG_RDX, IR_REG_RDX
+				mem = ir_ref_spill_slot(ctx, func);
 			}
-			if (op2_reg != IR_REG_NONE) {
-				|	ASM_REG_OP div, type, op2_reg
+			ir_reg base = IR_MEM_BASE(mem);
+			ir_reg index = IR_MEM_INDEX(mem);
+			if ((base != IR_REG_NONE && IR_REGSET_IN(preserved_regs, base)) ||
+					(index != IR_REG_NONE && IR_REGSET_IN(preserved_regs, index))) {
+				op2_reg = IR_REG_RAX;
+
+				ir_type type = ctx->ir_base[func].type;
+				ir_emit_load_mem_int(ctx, type, op2_reg, mem);
 			} else {
-				if (ir_rule(ctx, op2) & IR_FUSED) {
-					mem = ir_fuse_load(ctx, def, op2);
-				} else {
-					mem = ir_ref_spill_slot(ctx, op2);
-				}
-				|	ASM_MEM_OP div, type, mem
+				op2_reg = IR_REG_NONE;
 			}
 		}
 	}

-	if (insn->op == IR_MUL || insn->op == IR_MUL_OV || insn->op == IR_DIV) {
-		if (def_reg != IR_REG_NONE) {
-			if (def_reg != IR_REG_RAX) {
-				ir_emit_mov(ctx, type, def_reg, IR_REG_RAX);
-			}
-			if (IR_REG_SPILLED(ctx->regs[def][0])) {
-				ir_emit_store(ctx, type, def, def_reg);
-			}
-		} else {
-			ir_emit_store(ctx, type, def, IR_REG_RAX);
+	if (IR_IS_CONST_REF(func)
+	 && (ctx->flags2 & IR_RECURSIVE_TAILCALL)
+	 && ctx->ir_base[func].op == IR_FUNC
+	 && ctx->func_name == ctx->ir_base[func].val.name) {
+		if (ctx->flags2 & IR_HAS_ALLOCA) {
+			int offset = ctx->stack_frame_size + ctx->call_stack_size;
+
+			IR_ASSERT(ctx->flags & IR_USE_FRAME_POINTER);
+			|	lea Ra(IR_REG_RSP), [Ra(IR_REG_RBP)+offset]
 		}
-	} else {
-		IR_ASSERT(insn->op == IR_MOD);
-		if (ir_type_size[type] == 1) {
-			if (def_reg != IR_REG_NONE) {
-				|	mov al, ah
-				if (def_reg != IR_REG_RAX) {
-					|	mov Rb(def_reg), al
-				}
-				if (IR_REG_SPILLED(ctx->regs[def][0])) {
-					ir_emit_store(ctx, type, def, def_reg);
-				}
-			} else {
-				ir_reg fp;
-				int32_t offset = ir_ref_spill_slot_offset(ctx, def, &fp);

-//?????
-				|	mov byte [Ra(fp)+offset], ah
-			}
+		|	jmp =>0
+
+		return;
+	}
+
+	ir_emit_epilogue(ctx);
+
+	if (IR_IS_CONST_REF(func)) {
+		void *addr = ir_call_addr(ctx, insn, &ctx->ir_base[func]);
+
+		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+			|	jmp aword &addr
 		} else {
-			if (def_reg != IR_REG_NONE) {
-				if (def_reg != IR_REG_RDX) {
-					ir_emit_mov(ctx, type, def_reg, IR_REG_RDX);
-				}
-				if (IR_REG_SPILLED(ctx->regs[def][0])) {
-					ir_emit_store(ctx, type, def, def_reg);
-				}
-			} else {
-				ir_emit_store(ctx, type, def, IR_REG_RDX);
-			}
+|.if X64
+||			ir_reg tmp_reg = cc->int_ret_reg;
+||
+||			if (proto && (proto->flags & IR_VARARG_FUNC) && tmp_reg == cc->fp_varargs_reg) {
+||				tmp_reg = IR_REG_R11; // TODO: avoid usage of hardcoded temporary register ???
+||			}
+||			if (IR_IS_SIGNED_32BIT(addr)) {
+				|	mov Rq(tmp_reg), ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+||			} else {
+				|	mov64 Rq(tmp_reg), ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+||			}
+			|	jmp Rq(tmp_reg)
+|.endif
 		}
-	}
+    } else {
+		if (op2_reg != IR_REG_NONE) {
+			IR_ASSERT(!IR_REGSET_IN((ir_regset)ctx->used_preserved_regs, op2_reg));
+			|	jmp Ra(op2_reg)
+		} else {
+			|	ASM_TMEM_OP jmp, aword, mem
+		}
+    }
 }

-static void ir_rodata(ir_ctx *ctx)
+static void ir_emit_ijmp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_reg op2_reg = ctx->regs[def][2];

-	|.rodata
-	if (!data->rodata_label) {
-		int label = data->rodata_label = ctx->cfg_blocks_count + ctx->consts_count + 2;
-		|=>label:
+	if (IR_IS_CONST_REF(insn->op2)) {
+		if (ctx->ir_base[insn->op2].op == IR_LABEL) {
+			if (!data->resolved_label_syms) {
+				data->resolved_label_syms = 1;
+				ir_resolve_label_syms(ctx);
+			}
+
+			uint32_t target = ctx->ir_base[insn->op2].val.u32_hi;
+			target = ir_skip_empty_target_blocks(ctx, target);
+
+			|	jmp =>target
+			return;
+		}
+
+		void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op2]);
+
+		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+			|	jmp aword &addr
+		} else {
+|.if X64
+			if (IR_IS_SIGNED_32BIT(addr)) {
+				|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+			} else {
+				|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+			}
+			|	jmp rax
+|.endif
+		}
+	} else if (ir_rule(ctx, insn->op2) & IR_FUSED) {
+	    ir_mem mem = ir_fuse_load(ctx, def, insn->op2);
+		|	ASM_TMEM_OP jmp, aword, mem
+	} else if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+		}
+		|	jmp Ra(op2_reg)
+	} else {
+		ir_mem mem = ir_ref_spill_slot(ctx, insn->op2);
+
+		|	ASM_TMEM_OP jmp, aword, mem
 	}
 }

-static void ir_emit_op_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static bool ir_emit_guard_jcc(ir_ctx *ctx, uint32_t b, ir_ref def, uint32_t next_block, uint8_t op, void *addr, bool int_cmp, bool after_op)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	ir_insn *next_insn = &ctx->ir_base[def + 1];

-	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (next_insn->op == IR_END || next_insn->op == IR_LOOP_END) {
+		ir_block *bb = &ctx->cfg_blocks[b];
+		uint32_t target;

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
-		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
-		}
-	}
-	if (insn->op == IR_NEG) {
-		if (insn->type == IR_DOUBLE) {
-			if (!data->double_neg_const) {
-				data->double_neg_const = 1;
-				ir_rodata(ctx);
-				|.align 16
-				|->double_neg_const:
-				|.dword 0, 0x80000000, 0, 0
-				|.code
-			}
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vxorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->double_neg_const]
+		if (!(bb->flags & IR_BB_DESSA_MOVES)) {
+			target = ctx->cfg_edges[bb->successors];
+			if (UNEXPECTED(bb->successors_count == 2)) {
+				if (ctx->cfg_blocks[target].flags & IR_BB_ENTRY) {
+					target = ctx->cfg_edges[bb->successors + 1];
+				} else {
+					IR_ASSERT(ctx->cfg_blocks[ctx->cfg_edges[bb->successors + 1]].flags & IR_BB_ENTRY);
+				}
 			} else {
-				|	xorpd xmm(def_reg-IR_REG_FP_FIRST), [->double_neg_const]
-			}
-		} else {
-			IR_ASSERT(insn->type == IR_FLOAT);
-			if (!data->float_neg_const) {
-				data->float_neg_const = 1;
-				ir_rodata(ctx);
-				|.align 16
-				|->float_neg_const:
-				|.dword 0x80000000, 0, 0, 0
-				|.code
+				IR_ASSERT(bb->successors_count == 1);
 			}
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->float_neg_const]
-			} else {
-				|	xorps xmm(def_reg-IR_REG_FP_FIRST), [->float_neg_const]
+			target = ir_skip_empty_target_blocks(ctx, target);
+			if (target != next_block) {
+				if (int_cmp) {
+					switch (op) {
+						default:
+							IR_ASSERT(0 && "NIY binary op");
+						case IR_EQ:
+							|	jne =>target
+							break;
+						case IR_NE:
+							|	je =>target
+							break;
+						case IR_LT:
+							if (after_op) {
+								|	jns =>target
+							} else {
+								|	jge =>target
+							}
+							break;
+						case IR_GE:
+							if (after_op) {
+								|	js =>target
+							} else {
+								|	jl =>target
+							}
+							break;
+						case IR_LE:
+							|	jg =>target
+							break;
+						case IR_GT:
+							|	jle =>target
+							break;
+						case IR_ULT:
+							|	jae =>target
+							break;
+						case IR_UGE:
+							|	jb =>target
+							break;
+						case IR_ULE:
+							|	ja =>target
+							break;
+						case IR_UGT:
+							|	jbe =>target
+							break;
+					}
+				} else {
+					switch (op) {
+						default:
+							IR_ASSERT(0 && "NIY binary op");
+						case IR_EQ:
+							|	jne =>target
+							|	jp =>target
+							break;
+						case IR_NE:
+							|	jp &addr
+							|	je =>target
+							break;
+						case IR_LT:
+							|	jae =>target
+							break;
+						case IR_GE:
+							|	jp &addr
+							|	jb =>target
+							break;
+						case IR_LE:
+							|	ja =>target
+							break;
+						case IR_GT:
+							|	jp &addr
+							|	jbe =>target
+							break;
+						case IR_ULT:
+							|	jp =>target
+							|	jae =>target
+							break;
+						case IR_UGE:
+							|	jb =>target
+							break;
+						case IR_ULE:
+							|	jp =>target
+							|	ja =>target
+							break;
+						case IR_UGT:
+							|	jbe =>target
+							break;
+						case IR_ORDERED:
+							|	jnp =>target
+							break;
+						case IR_UNORDERED:
+							|	jp =>target
+							break;
+					}
+				}
+				|	jmp &addr
+				return 1;
 			}
 		}
-	} else {
-		IR_ASSERT(insn->op == IR_ABS);
-		if (insn->type == IR_DOUBLE) {
-			if (!data->double_abs_const) {
-				data->double_abs_const = 1;
-				ir_rodata(ctx);
-				|.align 16
-				|->double_abs_const:
-				|.dword 0xffffffff, 0x7fffffff, 0, 0
-				|.code
-			}
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vandpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->double_abs_const]
-			} else {
-				|	andpd xmm(def_reg-IR_REG_FP_FIRST), [->double_abs_const]
-			}
-		} else {
-			IR_ASSERT(insn->type == IR_FLOAT);
-			if (!data->float_abs_const) {
-				data->float_abs_const = 1;
-				ir_rodata(ctx);
-				|.align 16
-				|->float_abs_const:
-				|.dword 0x7fffffff, 0, 0, 0
-				|.code
-			}
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vandps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->float_abs_const]
+	} else if (next_insn->op == IR_IJMP && IR_IS_CONST_REF(next_insn->op2)) {
+		void *target_addr = ir_jmp_addr(ctx, next_insn, &ctx->ir_base[next_insn->op2]);
+
+		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, target_addr)) {
+			if (int_cmp) {
+				switch (op) {
+					default:
+						IR_ASSERT(0 && "NIY binary op");
+					case IR_EQ:
+						|	jne &target_addr
+						break;
+					case IR_NE:
+						|	je &target_addr
+						break;
+					case IR_LT:
+						if (after_op) {
+							|	jns &target_addr
+						} else {
+							|	jge &target_addr
+						}
+						break;
+					case IR_GE:
+						if (after_op) {
+							|	js &target_addr
+						} else {
+							|	jl &target_addr
+						}
+						break;
+					case IR_LE:
+						|	jg &target_addr
+						break;
+					case IR_GT:
+						|	jle &target_addr
+						break;
+					case IR_ULT:
+						|	jae &target_addr
+						break;
+					case IR_UGE:
+						|	jb &target_addr
+						break;
+					case IR_ULE:
+						|	ja &target_addr
+						break;
+					case IR_UGT:
+						|	jbe &target_addr
+						break;
+				}
 			} else {
-				|	andps xmm(def_reg-IR_REG_FP_FIRST), [->float_abs_const]
+				switch (op) {
+					default:
+						IR_ASSERT(0 && "NIY binary op");
+					case IR_EQ:
+						|	jne &target_addr
+						|	jp &target_addr
+						break;
+					case IR_NE:
+						|	jp &addr
+						|	je &target_addr
+						break;
+					case IR_LT:
+						|	jae &target_addr
+						break;
+					case IR_GE:
+						|	jp &addr
+						|	jb &target_addr
+						break;
+					case IR_LE:
+						|	ja &target_addr
+						break;
+					case IR_GT:
+						|	jp &addr
+						|	jbe &target_addr
+						break;
+					case IR_ULT:
+						|	jp &target_addr
+						|	jae &target_addr
+						break;
+					case IR_UGE:
+						|	jb &target_addr
+						break;
+					case IR_ULE:
+						|	jp &target_addr
+						|	ja &target_addr
+						break;
+					case IR_UGT:
+						|	jbe &target_addr
+						break;
+					case IR_ORDERED:
+						|	jnp &target_addr
+						break;
+					case IR_UNORDERED:
+						|	jp &target_addr
+						break;
+				}
 			}
+			|	jmp &addr
+			return 1;
 		}
 	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
-	}
-}
-
-static void ir_emit_binop_sse2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];

-	IR_ASSERT(def_reg != IR_REG_NONE);
-
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (def_reg != op1_reg) {
-		if (op1_reg != IR_REG_NONE) {
-			ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
-		} else {
-			ir_emit_load(ctx, type, def_reg, op1);
-		}
-		if (op1 == op2) {
-			op2_reg = def_reg;
-		}
-	}
-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			if (op1 != op2) {
-				ir_emit_load(ctx, type, op2_reg, op2);
-			}
-		}
-		switch (insn->op) {
+	if (int_cmp) {
+		switch (op) {
 			default:
 				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_SSE2_REG_REG_OP adds, type, def_reg, op2_reg
+			case IR_EQ:
+				|	je &addr
 				break;
-			case IR_SUB:
-				|	ASM_SSE2_REG_REG_OP subs, type, def_reg, op2_reg
+			case IR_NE:
+				|	jne &addr
 				break;
-			case IR_MUL:
-				|	ASM_SSE2_REG_REG_OP muls, type, def_reg, op2_reg
+			case IR_LT:
+				if (after_op) {
+					|	js &addr
+				} else {
+					|	jl &addr
+				}
 				break;
-			case IR_DIV:
-				|	ASM_SSE2_REG_REG_OP divs, type, def_reg, op2_reg
+			case IR_GE:
+				if (after_op) {
+					|	jns &addr
+				} else {
+					|	jge &addr
+				}
 				break;
-			case IR_MIN:
-				|	ASM_SSE2_REG_REG_OP mins, type, def_reg, op2_reg
+			case IR_LE:
+				|	jle &addr
 				break;
-			case IR_MAX:
-				|	ASM_SSE2_REG_REG_OP maxs, type, def_reg, op2_reg
+			case IR_GT:
+				|	jg &addr
+				break;
+			case IR_ULT:
+				|	jb &addr
+				break;
+			case IR_UGE:
+				|	jae &addr
+				break;
+			case IR_ULE:
+				|	jbe &addr
+				break;
+			case IR_UGT:
+				|	ja &addr
 				break;
 		}
-	} else if (IR_IS_CONST_REF(op2)) {
-		int label = ir_get_const_label(ctx, op2);
-
-		switch (insn->op) {
+	} else {
+		switch (op) {
 			default:
 				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_SSE2_REG_TXT_OP adds, type, def_reg, [=>label]
+			case IR_EQ:
+				|	jp >1
+				|	je &addr
+				|1:
 				break;
-			case IR_SUB:
-				|	ASM_SSE2_REG_TXT_OP subs, type, def_reg, [=>label]
+			case IR_NE:
+				|	jne &addr
+				|	jp &addr
 				break;
-			case IR_MUL:
-				|	ASM_SSE2_REG_TXT_OP muls, type, def_reg, [=>label]
+			case IR_LT:
+				|	jp >1
+				|	jb &addr
+				|1:
 				break;
-			case IR_DIV:
-				|	ASM_SSE2_REG_TXT_OP divs, type, def_reg, [=>label]
+			case IR_GE:
+				|	jae &addr
 				break;
-			case IR_MIN:
-				|	ASM_SSE2_REG_TXT_OP mins, type, def_reg, [=>label]
+			case IR_LE:
+				|	jp >1
+				|	jbe &addr
+				|1:
 				break;
-			case IR_MAX:
-				|	ASM_SSE2_REG_TXT_OP maxs, type, def_reg, [=>label]
+			case IR_GT:
+				|	ja &addr
+				break;
+			case IR_ULT:
+				|	jb &addr
+				break;
+			case IR_UGE:
+				|	jp &addr
+				|	jae &addr
+				break;
+			case IR_ULE:
+				|	jbe &addr
+				break;
+			case IR_UGT:
+				|	jp &addr
+				|	ja &addr
+				break;
+			case IR_ORDERED:
+				|	jp &addr
 				break;
+			case IR_UNORDERED:
+				|	jnp &addr
+				break;
+		}
+	}
+	return 0;
+}
+
+static bool ir_emit_guard(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_type type = ctx->ir_base[insn->op2].type;
+	void *addr;
+
+	IR_ASSERT(IR_IS_TYPE_INT(type));
+	if (IR_IS_CONST_REF(insn->op2)) {
+		bool is_true = ir_ref_is_true(ctx, insn->op2);
+
+		if ((insn->op == IR_GUARD && !is_true) || (insn->op == IR_GUARD_NOT && is_true)) {
+			addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+			if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+				|	jmp aword &addr
+			} else {
+|.if X64
+				if (IR_IS_SIGNED_32BIT(addr)) {
+					|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+				} else {
+					|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+				}
+				|	jmp aword [rax]
+|.endif
+			}
+		}
+		return 0;
+	}
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
+		}
+		|	ASM_REG_REG_OP test, type, op2_reg, op2_reg
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, insn->op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op2);
 		}
-	} else {
-		ir_mem mem;
+		|	ASM_MEM_IMM_OP cmp, type, mem, 0
+	}

-		if (ir_rule(ctx, op2) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, op2);
+	addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+	if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+		ir_op op;
+
+		if (insn->op == IR_GUARD) {
+			op = IR_EQ;
 		} else {
-			mem = ir_ref_spill_slot(ctx, op2);
+			op = IR_NE;
 		}
-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_SSE2_REG_MEM_OP adds, type, def_reg, mem
-				break;
-			case IR_SUB:
-				|	ASM_SSE2_REG_MEM_OP subs, type, def_reg, mem
-				break;
-			case IR_MUL:
-				|	ASM_SSE2_REG_MEM_OP muls, type, def_reg, mem
-				break;
-			case IR_DIV:
-				|	ASM_SSE2_REG_MEM_OP divs, type, def_reg, mem
-				break;
-			case IR_MIN:
-				|	ASM_SSE2_REG_MEM_OP mins, type, def_reg, mem
-				break;
-			case IR_MAX:
-				|	ASM_SSE2_REG_MEM_OP maxs, type, def_reg, mem
-				break;
+		return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 0);
+	} else {
+|.if X64
+		if (insn->op == IR_GUARD) {
+			|	je >1
+		} else {
+			|	jne >1
 		}
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
+		|.cold_code
+		|1:
+		if (IR_IS_SIGNED_32BIT(addr)) {
+			|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+		} else {
+			|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+		}
+		|	jmp aword [rax]
+		|.code
+|.endif
+		return 0;
 	}
 }

-static void ir_emit_binop_avx(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static bool ir_emit_guard_cmp_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op2];
+	ir_op op = cmp_insn->op;
+	ir_type type = ctx->ir_base[cmp_insn->op1].type;
+	ir_ref op1 = cmp_insn->op1;
+	ir_ref op2 = cmp_insn->op2;
+	void *addr;
+	ir_reg op1_reg, op2_reg;

-	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+	if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[insn->op2][1];
+		op2_reg = ctx->regs[insn->op2][2];
+	}

-	if (IR_REG_SPILLED(op1_reg)) {
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
 		ir_emit_load(ctx, type, op1_reg, op1);
 	}
-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			if (op1 != op2) {
-				ir_emit_load(ctx, type, op2_reg, op2);
-			}
-		}
-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_AVX_REG_REG_REG_OP vadds, type, def_reg, op1_reg, op2_reg
-				break;
-			case IR_SUB:
-				|	ASM_AVX_REG_REG_REG_OP vsubs, type, def_reg, op1_reg, op2_reg
-				break;
-			case IR_MUL:
-				|	ASM_AVX_REG_REG_REG_OP vmuls, type, def_reg, op1_reg, op2_reg
-				break;
-			case IR_DIV:
-				|	ASM_AVX_REG_REG_REG_OP vdivs, type, def_reg, op1_reg, op2_reg
-				break;
-			case IR_MIN:
-				|	ASM_AVX_REG_REG_REG_OP vmins, type, def_reg, op1_reg, op2_reg
-				break;
-			case IR_MAX:
-				|	ASM_AVX_REG_REG_REG_OP vmaxs, type, def_reg, op1_reg, op2_reg
-				break;
+	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		if (op1 != op2) {
+			ir_emit_load(ctx, type, op2_reg, op2);
 		}
-	} else if (IR_IS_CONST_REF(op2)) {
-		int label = ir_get_const_label(ctx, op2);
+	}

-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_AVX_REG_REG_TXT_OP vadds, type, def_reg, op1_reg, [=>label]
-				break;
-			case IR_SUB:
-				|	ASM_AVX_REG_REG_TXT_OP vsubs, type, def_reg, op1_reg, [=>label]
-				break;
-			case IR_MUL:
-				|	ASM_AVX_REG_REG_TXT_OP vmuls, type, def_reg, op1_reg, [=>label]
-				break;
-			case IR_DIV:
-				|	ASM_AVX_REG_REG_TXT_OP vdivs, type, def_reg, op1_reg, [=>label]
-				break;
-			case IR_MIN:
-				|	ASM_AVX_REG_REG_TXT_OP vmins, type, def_reg, op1_reg, [=>label]
-				break;
-			case IR_MAX:
-				|	ASM_AVX_REG_REG_TXT_OP vmaxs, type, def_reg, op1_reg, [=>label]
-				break;
+	addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
+		if (op == IR_ULT) {
+			/* always false */
+			if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
+				|	jmp aword &addr
+			} else {
+|.if X64
+				if (IR_IS_SIGNED_32BIT(addr)) {
+					|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+				} else {
+					|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+				}
+				|	jmp aword [rax]
+|.endif
+			}
+			return 0;
+		} else if (op == IR_UGE) {
+			/* always true */
+			return 0;
+		} else if (op == IR_ULE) {
+			op = IR_EQ;
+		} else if (op == IR_UGT) {
+			op = IR_NE;
 		}
-	} else {
-		ir_mem mem;
+	}
+	ir_emit_cmp_int_common(ctx, type, def, cmp_insn, op1_reg, op1, op2_reg, op2);

-		if (ir_rule(ctx, op2) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, op2);
+	if (insn->op == IR_GUARD) {
+		op ^= 1; // reverse
+	}
+
+	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 0);
+}
+
+static bool ir_emit_guard_cmp_fp(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_op op = ir_emit_cmp_fp_common(ctx, def, insn->op2, &ctx->ir_base[insn->op2]);
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+
+	if (insn->op == IR_GUARD) {
+		if (op == IR_EQ || op == IR_NE || op == IR_ORDERED || op == IR_UNORDERED) {
+			op ^= 1; // reverse
 		} else {
-			mem = ir_ref_spill_slot(ctx, op2);
+			op ^= 5; // reverse
 		}
-		switch (insn->op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_ADD:
-				|	ASM_AVX_REG_REG_MEM_OP vadds, type, def_reg, op1_reg, mem
-				break;
-			case IR_SUB:
-				|	ASM_AVX_REG_REG_MEM_OP vsubs, type, def_reg, op1_reg, mem
-				break;
-			case IR_MUL:
-				|	ASM_AVX_REG_REG_MEM_OP vmuls, type, def_reg, op1_reg, mem
-				break;
-			case IR_DIV:
-				|	ASM_AVX_REG_REG_MEM_OP vdivs, type, def_reg, op1_reg, mem
-				break;
-			case IR_MIN:
-				|	ASM_AVX_REG_REG_MEM_OP vmins, type, def_reg, op1_reg, mem
-				break;
-			case IR_MAX:
-				|	ASM_AVX_REG_REG_MEM_OP vmaxs, type, def_reg, op1_reg, mem
-				break;
+	}
+	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 0, 0);
+}
+
+static bool ir_emit_guard_test_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+	ir_op op = (insn->op == IR_GUARD) ? IR_EQ : IR_NE;
+
+	ir_emit_test_int_common(ctx, def, insn->op2, op);
+	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 0);
+}
+
+static bool ir_emit_guard_test_bit(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+
+	ir_emit_test_bit_common(ctx, def, insn->op2);
+	if (insn->op == IR_GUARD) {
+		|	jnc &addr
+	} else {
+		|	jc &addr
+	}
+	return 0;
+}
+
+static bool ir_emit_guard_jcc_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+	ir_op op = ctx->ir_base[insn->op2].op;
+
+	if (insn->op == IR_GUARD) {
+		op ^= 1; // reverse
+	}
+	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 1);
+}
+
+static bool ir_emit_guard_overflow(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type;
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+
+	type = ctx->ir_base[ctx->ir_base[insn->op2].op1].type;
+
+	IR_ASSERT(IR_IS_TYPE_INT(type));
+	if (IR_IS_TYPE_SIGNED(type)) {
+		if (insn->op == IR_GUARD) {
+			|	jno &addr
+		} else {
+			|	jo &addr
+		}
+	} else {
+		if (insn->op == IR_GUARD) {
+			|	jnc &addr
+		} else {
+			|	jc &addr
 		}
 	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
-	}
+	return 0;
 }

-static void ir_emit_cmp_int_common(ir_ctx *ctx, ir_type type, ir_ref root, ir_insn *insn, ir_reg op1_reg, ir_ref op1, ir_reg op2_reg, ir_ref op2)
+static void ir_emit_lea(ir_ctx *ctx, ir_ref def, ir_type type)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_mem mem = ir_fuse_addr(ctx, def, def);

-	if (op1_reg != IR_REG_NONE) {
-		if (op2_reg != IR_REG_NONE) {
-			|	ASM_REG_REG_OP cmp, type, op1_reg, op2_reg
-		} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
-			|	ASM_REG_REG_OP test, type, op1_reg, op1_reg
-		} else if (IR_IS_CONST_REF(op2)) {
-			int32_t val = ir_fuse_imm(ctx, op2);
-			|	ASM_REG_IMM_OP cmp, type, op1_reg, val
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (ir_type_size[type] == 4) {
+		if (IR_MEM_BASE(mem) == def_reg
+		 && IR_MEM_OFFSET(mem) == 0
+		 && IR_MEM_SCALE(mem) == 1
+		 && IR_MEM_INDEX(mem) != IR_REG_NONE) {
+			ir_reg reg = IR_MEM_INDEX(mem);
+			|	add Rd(def_reg), Rd(reg)
+		} else if (IR_MEM_INDEX(mem) == def_reg
+		 && IR_MEM_OFFSET(mem) == 0
+		 && IR_MEM_SCALE(mem) == 1
+		 && IR_MEM_BASE(mem) != IR_REG_NONE) {
+			ir_reg reg = IR_MEM_BASE(mem);
+			|	add Rd(def_reg), Rd(reg)
+		} else if (IR_MEM_INDEX(mem) == def_reg
+		 && IR_MEM_OFFSET(mem) == 0
+		 && IR_MEM_SCALE(mem) == 2
+		 && IR_MEM_BASE(mem) == IR_REG_NONE) {
+			|	add Rd(def_reg), Rd(def_reg)
 		} else {
-			ir_mem mem;
-
-			if (ir_rule(ctx, op2) & IR_FUSED) {
-				mem = ir_fuse_load(ctx, root, op2);
-			} else {
-				mem = ir_ref_spill_slot(ctx, op2);
+			if (IR_MEM_SCALE(mem) == 2 && IR_MEM_BASE(mem) == IR_REG_NONE) {
+				mem = IR_MEM(IR_MEM_INDEX(mem), IR_MEM_OFFSET(mem), IR_MEM_INDEX(mem), 1);
 			}
-			|	ASM_REG_MEM_OP cmp, type, op1_reg, mem
+			|	ASM_TXT_TMEM_OP lea, Rd(def_reg), dword, mem
 		}
-	} else if (IR_IS_CONST_REF(op1)) {
-		IR_ASSERT(0);
 	} else {
-		ir_mem mem;
-
-		if (ir_rule(ctx, op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, root, op1);
-		} else {
-			mem = ir_ref_spill_slot(ctx, op1);
-		}
-		if (op2_reg != IR_REG_NONE) {
-			|	ASM_MEM_REG_OP cmp, type, mem, op2_reg
+		if (IR_MEM_BASE(mem) == def_reg
+		 && IR_MEM_OFFSET(mem) == 0
+		 && IR_MEM_SCALE(mem) == 1
+		 && IR_MEM_INDEX(mem) != IR_REG_NONE) {
+			ir_reg reg = IR_MEM_INDEX(mem);
+			|	add Ra(def_reg), Ra(reg)
+		} else if (IR_MEM_INDEX(mem) == def_reg
+		 && IR_MEM_OFFSET(mem) == 0
+		 && IR_MEM_SCALE(mem) == 1
+		 && IR_MEM_BASE(mem) != IR_REG_NONE) {
+			ir_reg reg = IR_MEM_BASE(mem);
+			|	add Ra(def_reg), Ra(reg)
+		} else if (IR_MEM_INDEX(mem) == def_reg
+		 && IR_MEM_OFFSET(mem) == 0
+		 && IR_MEM_SCALE(mem) == 2
+		 && IR_MEM_BASE(mem) == IR_REG_NONE) {
+			|	add Ra(def_reg), Ra(def_reg)
 		} else {
-			int32_t val = ir_fuse_imm(ctx, op2);
-			|	ASM_MEM_IMM_OP cmp, type, mem, val
+			if (IR_MEM_SCALE(mem) == 2 && IR_MEM_BASE(mem) == IR_REG_NONE) {
+				mem = IR_MEM(IR_MEM_INDEX(mem), IR_MEM_OFFSET(mem), IR_MEM_INDEX(mem), 1);
+			}
+			|	ASM_TXT_TMEM_OP lea, Ra(def_reg), aword, mem
 		}
 	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, type, def, def_reg);
+	}
 }

-static void ir_emit_cmp_int_common2(ir_ctx *ctx, ir_ref root, ir_ref ref, ir_insn *cmp_insn)
+static void ir_emit_tls_prefix(ir_ctx *ctx)
 {
-	ir_type type = ctx->ir_base[cmp_insn->op1].type;
-	ir_ref op1 = cmp_insn->op1;
-	ir_ref op2 = cmp_insn->op2;
-	ir_reg op1_reg, op2_reg;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;

-	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
-		op1_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 1);
-		op2_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 2);
-	} else {
-		op1_reg = ctx->regs[ref][1];
-		op2_reg = ctx->regs[ref][2];
-	}
+||#if defined(_WIN64)
+|	gs
+||#elif defined(_WIN32)
+|	fs
+||#elif defined(__APPLE__)
+|	gs
+||#else
+|.if X64
+|	fs
+|.else
+|	gs
+|.endif
+||#endif
+}

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		if (op1 != op2) {
-			ir_emit_load(ctx, type, op2_reg, op2);
-		}
-	}
+static void ir_emit_tls_base_addr(ir_ctx *ctx, ir_reg reg, int mod)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;

-	ir_emit_cmp_int_common(ctx, type, root, cmp_insn, op1_reg, op1, op2_reg, op2);
+||#if defined(_WIN64)
+|	gs
+||	if (mod < 0) {
+|		mov Ra(reg), [0]
+||	} else {
+|		mov Ra(reg), aword [0x58]
+|		mov Ra(reg), aword [Ra(reg)+mod]
+||	}
+||#elif defined(_WIN32)
+|	fs
+||	if (mod < 0) {
+|		mov Ra(reg), [0]
+||	} else {
+|		mov Ra(reg), aword [0x2c]
+|		mov Ra(reg), aword [Ra(reg)+mod]
+||	}
+||#elif defined(__APPLE__)
+|	gs
+||	if (mod < 0) {
+|		mov Ra(reg), [0]
+||	} else {
+|		mov Ra(reg), aword [mod]
+||	}
+||#else
+|.if X64
+|	fs
+||	if (mod < 0) {
+|		mov Ra(reg), [0]
+||	} else {
+|		mov Ra(reg), [0x8]
+|		mov Ra(reg), aword [Ra(reg)+mod]
+||	}
+|.else
+|	gs
+||	if (mod < 0) {
+|		mov Ra(reg), aword [0]
+||	} else {
+|		mov Ra(reg), [0x4]
+|		mov Ra(reg), aword [Ra(reg)+mod]
+||	}
+|.endif
+||#endif
 }

-static void _ir_emit_setcc_int(ir_ctx *ctx, uint8_t op, ir_reg def_reg, bool after_op)
+static void ir_emit_tls_addr(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_reg reg = IR_REG_NUM(ctx->regs[def][0]);

-	switch (op) {
-		default:
-			IR_ASSERT(0 && "NIY binary op");
-		case IR_EQ:
-			|	sete Rb(def_reg)
-			break;
-		case IR_NE:
-			|	setne Rb(def_reg)
-			break;
-		case IR_LT:
-			if (after_op) {
-				|	sets Rb(def_reg)
-			} else {
-				|	setl Rb(def_reg)
-			}
-			break;
-		case IR_GE:
-			if (after_op) {
-				|	setns Rb(def_reg)
-			} else {
-				|	setge Rb(def_reg)
-			}
-			break;
-		case IR_LE:
-			|	setle Rb(def_reg)
-			break;
-		case IR_GT:
-			|	setg Rb(def_reg)
-			break;
-		case IR_ULT:
-			|	setb Rb(def_reg)
-			break;
-		case IR_UGE:
-			|	setae Rb(def_reg)
-			break;
-		case IR_ULE:
-			|	setbe Rb(def_reg)
-			break;
-		case IR_UGT:
-			|	seta Rb(def_reg)
-			break;
+	IR_ASSERT(reg != IR_REG_NONE);
+	ir_emit_tls_base_addr(ctx, reg, insn->op2);
+	if (insn->op3) {
+		|	add Ra(reg), aword insn->op3
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, IR_ADDR, def, reg);
 	}
 }

-static void _ir_emit_setcc_int_mem(ir_ctx *ctx, uint8_t op, ir_mem mem)
+static void ir_emit_tls_load(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg reg = IR_IS_TYPE_INT(insn->type) ? def_reg : ctx->regs[def][3];
+	ir_insn *addr_insn = &ctx->ir_base[insn->op2];
+	ir_mem mem;

+	if (ctx->use_lists[def].count == 1) {
+		/* dead load */
+		return;
+	}

-	switch (op) {
-		default:
-			IR_ASSERT(0 && "NIY binary op");
-		case IR_EQ:
-			|	ASM_TMEM_OP sete, byte, mem
-			break;
-		case IR_NE:
-			|	ASM_TMEM_OP setne, byte, mem
-			break;
-		case IR_LT:
-			|	ASM_TMEM_OP setl, byte, mem
-			break;
-		case IR_GE:
-			|	ASM_TMEM_OP setge, byte, mem
-			break;
-		case IR_LE:
-			|	ASM_TMEM_OP setle, byte, mem
-			break;
-		case IR_GT:
-			|	ASM_TMEM_OP setg, byte, mem
-			break;
-		case IR_ULT:
-			|	ASM_TMEM_OP setb, byte, mem
-			break;
-		case IR_UGE:
-			|	ASM_TMEM_OP setae, byte, mem
-			break;
-		case IR_ULE:
-			|	ASM_TMEM_OP setbe, byte, mem
-			break;
-		case IR_UGT:
-			|	ASM_TMEM_OP seta, byte, mem
-			break;
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	IR_ASSERT(addr_insn->op == IR_TLS_ADDR);
+
+	if (addr_insn->op2 < 0) {
+		ir_emit_tls_prefix(ctx);
+		mem = IR_MEM_O(addr_insn->op3);
+	} else {
+		ir_emit_tls_base_addr(ctx, reg, addr_insn->op2);
+		mem = IR_MEM_BO(reg, addr_insn->op3);
+	}
+
+	ir_emit_load_mem(ctx, insn->type, def_reg, mem);
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_tls_store(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+{
+	ir_type type = ctx->ir_base[insn->op3].type;
+	ir_reg op3_reg = ctx->regs[ref][3];
+	ir_insn *addr_insn = &ctx->ir_base[insn->op2];
+	ir_reg reg = ctx->regs[ref][0];
+	ir_mem mem;
+
+	IR_ASSERT(addr_insn->op == IR_TLS_ADDR);
+
+	if (op3_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, insn->op3);
+		}
+	}
+
+	if (addr_insn->op2 < 0) {
+		ir_emit_tls_prefix(ctx);
+		mem = IR_MEM_O(addr_insn->op3);
+	} else {
+		ir_emit_tls_base_addr(ctx, reg, addr_insn->op2);
+		mem = IR_MEM_BO(reg, addr_insn->op3);
+	}
+
+	if (op3_reg != IR_REG_NONE) {
+		ir_emit_store_mem(ctx, type, mem, op3_reg);
+	} else {
+		IR_ASSERT(IR_IS_CONST_REF(insn->op3));
+		ir_emit_store_mem_int_const(ctx, type, mem, insn->op3, op3_reg, 0);
 	}
 }

-static void ir_emit_cmp_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_sse_sqrt(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = ctx->ir_base[insn->op1].type;
-	ir_op op = insn->op;
-	ir_ref op1 = insn->op1;
-	ir_ref op2 = insn->op2;
+	ir_reg op3_reg = ctx->regs[def][3];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	ir_reg op2_reg = ctx->regs[def][2];

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		if (op1 != op2) {
-			ir_emit_load(ctx, type, op2_reg, op2);
-		}
-	}
-	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
-		if (op == IR_ULT) {
-			/* always false */
-			|	xor Ra(def_reg), Ra(def_reg)
-			if (IR_REG_SPILLED(ctx->regs[def][0])) {
-				ir_emit_store(ctx, insn->type, def, def_reg);
-			}
-			return;
-		} else if (op == IR_UGE) {
-			/* always true */
-			|	ASM_REG_IMM_OP mov, insn->type, def_reg, 1
-			if (IR_REG_SPILLED(ctx->regs[def][0])) {
-				ir_emit_store(ctx, insn->type, def, def_reg);
-			}
-			return;
-		} else if (op == IR_ULE) {
-			op = IR_EQ;
-		} else if (op == IR_UGT) {
-			op = IR_NE;
-		}
+	IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+	IR_ASSERT(def_reg != IR_REG_NONE && op3_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, insn->type, op3_reg, insn->op3);
 	}
-	ir_emit_cmp_int_common(ctx, type, def, insn, op1_reg, op1, op2_reg, op2);
-	_ir_emit_setcc_int(ctx, op, def_reg, 0);
+
+	|	ASM_FP_REG_REG_OP sqrts, insn->type, def_reg, op3_reg
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
 		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_test_int_common(ir_ctx *ctx, ir_ref root, ir_ref ref, ir_op op)
+static void ir_emit_sse_round(ir_ctx *ctx, ir_ref def, ir_insn *insn, int round_op)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *binop_insn = &ctx->ir_base[ref];
-	ir_type type = binop_insn->type;
-	ir_ref op1 = binop_insn->op1;
-	ir_ref op2 = binop_insn->op2;
-	ir_reg op1_reg, op2_reg;
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);

-	if (UNEXPECTED(ctx->rules[ref] & IR_FUSED_REG)) {
-		op1_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 1);
-		op2_reg = ir_get_fused_reg(ctx, root, ref * sizeof(ir_ref) + 2);
+	IR_ASSERT(IR_IS_TYPE_FP(insn->type));
+	IR_ASSERT(def_reg != IR_REG_NONE && op3_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, insn->type, op3_reg, insn->op3);
+	}
+
+	if (ctx->mflags & IR_X86_AVX) {
+		|	ASM_SSE2_REG_REG_REG_TXT_OP vrounds, insn->type, def_reg, def_reg, op3_reg, round_op
 	} else {
-		op1_reg = ctx->regs[ref][1];
-		op2_reg = ctx->regs[ref][2];
+		IR_ASSERT(ctx->mflags & IR_X86_SSE41);
+		|	ASM_SSE2_REG_REG_TXT_OP rounds, insn->type, def_reg, op3_reg, round_op
 	}

-	IR_ASSERT(binop_insn->op == IR_AND);
-	if (op1_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, type, op1_reg, op1);
-		}
-		if (op2_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op2_reg)) {
-				op2_reg = IR_REG_NUM(op2_reg);
-				if (op1 != op2) {
-					ir_emit_load(ctx, type, op2_reg, op2);
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+#if IR_X86_I64
+|.if not X64
+static void ir_emit_cmp_i64_common(ir_ctx *ctx, ir_op op, ir_ref root,
+                                   ir_reg def_reg, ir_reg tmp_reg,
+                                   ir_reg op1_reg, ir_reg op1_reg_hi, ir_ref op1,
+                                   ir_reg op2_reg, ir_reg op2_reg_hi, ir_ref op2)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_mem mem;
+
+	if (op == IR_EQ || op == IR_NE) {
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		if (def_reg == op2_reg) {
+			if (op1_reg != IR_REG_NONE) {
+				|	xor Rd(def_reg), Rd(op1_reg)
+			} else if (IR_IS_CONST_REF(op1)) {
+				ir_insn *val = &ctx->ir_base[op1];
+				|	xor Rd(def_reg), val->val.u32
+			} else {
+				if (ir_rule(ctx, op1) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, root, op1);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op1);
 				}
+				|	ASM_REG_MEM_OP xor, IR_U32, def_reg, mem
 			}
-			|	ASM_REG_REG_OP test, type, op1_reg, op2_reg
-		} else if (IR_IS_CONST_REF(op2)) {
-			int32_t val = ir_fuse_imm(ctx, op2);
-
-			if ((op == IR_EQ || op == IR_NE) && val == 0xff && (sizeof(void*) == 8 || op1_reg <= IR_REG_R3)) {
-				|	test Rb(op1_reg), Rb(op1_reg)
-			} else if ((op == IR_EQ || op == IR_NE) && val == 0xff00 && op1_reg <= IR_REG_R3) {
-				if (op1_reg == IR_REG_RAX) {
-					|	test ah, ah
-				} else if (op1_reg == IR_REG_RBX) {
-					|	test bh, bh
-				} else if (op1_reg == IR_REG_RCX) {
-					|	test ch, ch
-				} else if (op1_reg == IR_REG_RDX) {
-					|	test dh, dh
+		} else {
+			if (def_reg != op1_reg) {
+				if (op1_reg != IR_REG_NONE) {
+					|	mov Rd(def_reg), Rd(op1_reg)
+				} else if (IR_IS_CONST_REF(op1)) {
+					ir_insn *val = &ctx->ir_base[op1];
+					|	mov Rd(def_reg), val->val.u32
 				} else {
-					IR_ASSERT(0);
+					if (ir_rule(ctx, op1) & IR_FUSED) {
+						mem = ir_fuse_load(ctx, root, op1);
+					} else {
+						mem = ir_ref_spill_slot(ctx, op1);
+					}
+					|	ASM_REG_MEM_OP mov, IR_U32, def_reg, mem
 				}
-			} else if ((op == IR_EQ || op == IR_NE) && val == 0xffff) {
-				|	test Rw(op1_reg), Rw(op1_reg)
-			} else if ((op == IR_EQ || op == IR_NE) && val == -1) {
-				|	test Rd(op1_reg), Rd(op1_reg)
+			}
+			if (op2_reg != IR_REG_NONE) {
+				|	xor Rd(def_reg), Rd(op2_reg)
+			} else if (IR_IS_CONST_REF(op2)) {
+				ir_insn *val = &ctx->ir_base[op2];
+				|	xor Rd(def_reg), val->val.u32
 			} else {
-				|	ASM_REG_IMM_OP test, type, op1_reg, val
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, root, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				|	ASM_REG_MEM_OP xor, IR_U32, def_reg, mem
 			}
-		} else {
-			ir_mem mem;
+		}

-			if (ir_rule(ctx, op2) & IR_FUSED) {
-				mem = ir_fuse_load(ctx, root, op2);
+		if (tmp_reg == op2_reg_hi) {
+			if (op1_reg_hi != IR_REG_NONE) {
+				|	xor Rd(tmp_reg), Rd(op1_reg_hi)
+			} else if (IR_IS_CONST_REF(op1)) {
+				ir_insn *val = &ctx->ir_base[op1];
+				|	xor Rd(tmp_reg), val->val.u32_hi
 			} else {
-				mem = ir_ref_spill_slot(ctx, op2);
+				if (ir_rule(ctx, op1) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, root, op1);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op1);
+				}
+				mem = IR_MEM_I64_HI(mem);
+				|	ASM_REG_MEM_OP xor, IR_U32, tmp_reg, mem
+			}
+		} else {
+			if (tmp_reg != op1_reg_hi) {
+				if (op1_reg_hi != IR_REG_NONE) {
+					|	mov Rd(tmp_reg), Rd(op1_reg_hi)
+				} else if (IR_IS_CONST_REF(op1)) {
+					ir_insn *val = &ctx->ir_base[op1];
+					|	mov Rd(tmp_reg), val->val.u32_hi
+				} else {
+					if (ir_rule(ctx, op1) & IR_FUSED) {
+						mem = ir_fuse_load(ctx, root, op1);
+					} else {
+						mem = ir_ref_spill_slot(ctx, op1);
+					}
+					mem = IR_MEM_I64_HI(mem);
+					|	ASM_REG_MEM_OP mov, IR_U32, tmp_reg, mem
+				}
+			}
+			if (op2_reg_hi != IR_REG_NONE) {
+				|	xor Rd(tmp_reg), Rd(op2_reg_hi)
+			} else if (IR_IS_CONST_REF(op2)) {
+				ir_insn *val = &ctx->ir_base[op2];
+				|	xor Rd(tmp_reg), val->val.u32_hi
+			} else {
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, root, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				mem = IR_MEM_I64_HI(mem);
+				|	ASM_REG_MEM_OP xor, IR_U32, tmp_reg, mem
 			}
-			|	ASM_REG_MEM_OP test, type, op1_reg, mem
 		}
-	} else if (IR_IS_CONST_REF(op1)) {
-		IR_ASSERT(0);
-	} else {
-		ir_mem mem;
+		|	or Rd(def_reg), Rd(tmp_reg);

-		if (ir_rule(ctx, op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, root, op1);
-		} else {
-			mem = ir_ref_spill_slot(ctx, op1);
+	} else {
+		if (op == IR_LE || op == IR_GT || op == IR_ULE || op == IR_UGT) {
+			SWAP_REGS(op1_reg, op2_reg);
+			SWAP_REGS(op1_reg_hi, op2_reg_hi);
+			SWAP_REFS(op1, op2);
 		}
-		if (op2_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op2_reg)) {
-				op2_reg = IR_REG_NUM(op2_reg);
-				if (op1 != op2) {
-					ir_emit_load(ctx, type, op2_reg, op2);
+
+		if (op1_reg != IR_REG_NONE) {
+			if (op2_reg != IR_REG_NONE) {
+				|	cmp Rd(op1_reg), Rd(op2_reg)
+				if (op1_reg_hi != def_reg) {
+					|	mov Rd(def_reg), Rd(op1_reg_hi)
+				}
+				|	sbb Rd(def_reg), Rd(op2_reg_hi)
+			} else if (IR_IS_CONST_REF(op2)) {
+				ir_insn *val = &ctx->ir_base[op2];
+
+				if (val->val.u32 == 0) {
+					|	test Rd(op1_reg), Rd(op1_reg)
+				} else {
+					|	cmp Rd(op1_reg), val->val.u32
+				}
+				if (op1_reg_hi != def_reg) {
+					|	mov Rd(def_reg), Rd(op1_reg_hi)
+				}
+				|	sbb Rd(def_reg), val->val.u32_hi
+			} else {
+				ir_mem mem, mem_hi;
+
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, root, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				mem_hi = IR_MEM_I64_HI(mem);
+
+				|	ASM_REG_MEM_OP cmp, IR_U32, op1_reg, mem
+				if (op1_reg_hi != def_reg) {
+					|	mov Rd(def_reg), Rd(op1_reg_hi)
 				}
+				|	ASM_REG_MEM_OP sbb, IR_U32, def_reg, mem_hi
 			}
-			|	ASM_MEM_REG_OP test, type, mem, op2_reg
-		} else {
-			IR_ASSERT(!IR_IS_CONST_REF(op1));
-			int32_t val = ir_fuse_imm(ctx, op2);
-			|	ASM_MEM_IMM_OP test, type, mem, val
-		}
-	}
-}
+		} else if (IR_IS_CONST_REF(op1)) {
+			ir_insn *op1_val = &ctx->ir_base[op1];

-static void ir_emit_testcc_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+			if (op2_reg != IR_REG_NONE) {
+				|	mov Rd(def_reg), op1_val->val.u32
+				|	cmp Rd(def_reg), Rd(op2_reg)
+				|	mov Rd(def_reg), op1_val->val.u32_hi
+				|	sbb Rd(def_reg), Rd(op2_reg_hi)
+			} else if (IR_IS_CONST_REF(op2)) {
+				ir_insn *val = &ctx->ir_base[op2];
+
+				|	mov Rd(def_reg), op1_val->val.u32
+				|	cmp Rd(def_reg), val->val.u32
+				|	mov Rd(def_reg), op1_val->val.u32_hi
+				|	sbb Rd(def_reg), val->val.u32_hi
+			} else {
+				ir_mem mem, mem_hi;

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	ir_emit_test_int_common(ctx, def, insn->op1, insn->op);
-	_ir_emit_setcc_int(ctx, insn->op, def_reg, 0);
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
-	}
-}
+				if (ir_rule(ctx, op2) & IR_FUSED) {
+					mem = ir_fuse_load(ctx, root, op2);
+				} else {
+					mem = ir_ref_spill_slot(ctx, op2);
+				}
+				mem_hi = IR_MEM_I64_HI(mem);

-static void ir_emit_setcc_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+				|	mov Rd(def_reg), op1_val->val.u32
+				|	ASM_REG_MEM_OP cmp, IR_U32, def_reg, mem
+				|	mov Rd(def_reg), op1_val->val.u32_hi
+				|	ASM_REG_MEM_OP sbb, IR_U32, def_reg, mem_hi
+			}
+		} else {
+			ir_mem mem, mem_hi;

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	_ir_emit_setcc_int(ctx, insn->op, def_reg, 1);
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
+			if (ir_rule(ctx, op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, root, op1);
+			} else {
+				mem = ir_ref_spill_slot(ctx, op1);
+			}
+			mem_hi = IR_MEM_I64_HI(mem);
+
+			|	ASM_REG_MEM_OP mov, IR_U32, def_reg, mem_hi
+			if (op2_reg != IR_REG_NONE) {
+				|	ASM_MEM_REG_OP cmp, IR_U32, mem, op2_reg
+				|	sbb Rd(def_reg), Rd(op2_reg_hi)
+			} else {
+				ir_insn *val = &ctx->ir_base[op2];
+
+				|	ASM_MEM_IMM_OP cmp, IR_U32, mem, val->val.u32
+				|	sbb Rd(def_reg), val->val.u32_hi
+			}
+		}
 	}
 }

-static ir_op ir_emit_cmp_fp_common(ir_ctx *ctx, ir_ref root, ir_ref cmp_ref, ir_insn *cmp_insn)
+static void ir_emit_cmp_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = ctx->ir_base[cmp_insn->op1].type;
-	ir_op op = cmp_insn->op;
-	ir_ref op1, op2;
-	ir_reg op1_reg, op2_reg;
-
-	op1 = cmp_insn->op1;
-	op2 = cmp_insn->op2;
-	if (UNEXPECTED(ctx->rules[cmp_ref] & IR_FUSED_REG)) {
-		op1_reg = ir_get_fused_reg(ctx, root, cmp_ref * sizeof(ir_ref) + 1);
-		op2_reg = ir_get_fused_reg(ctx, root, cmp_ref * sizeof(ir_ref) + 2);
-	} else {
-		op1_reg = ctx->regs[cmp_ref][1];
-		op2_reg = ctx->regs[cmp_ref][2];
-	}
-
-	if (op1_reg == IR_REG_NONE && op2_reg != IR_REG_NONE && (op == IR_EQ || op == IR_NE)) {
-		ir_reg tmp_reg;
-
-		SWAP_REFS(op1, op2);
-		tmp_reg = op1_reg;
-		op1_reg = op2_reg;
-		op2_reg = tmp_reg;
-	}
-
+	ir_op op = insn->op;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	ir_reg op1_reg_hi, op2_reg_hi;

-	IR_ASSERT(op1_reg != IR_REG_NONE);
-	if (IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+		}
+	} else {
+		op1_reg_hi = IR_REG_NONE;
 	}
 	if (op2_reg != IR_REG_NONE) {
 		if (IR_REG_SPILLED(op2_reg)) {
 			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 			if (op1 != op2) {
-				ir_emit_load(ctx, type, op2_reg, op2);
+				ir_emit_load_i64_lo(ctx, op2_reg, op2);
+				ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
 			}
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-		|	ASM_FP_REG_REG_OP ucomis, type, op1_reg, op2_reg
-	} else if (IR_IS_CONST_REF(op2)) {
-		int label = ir_get_const_label(ctx, op2);
-
-		|	ASM_FP_REG_TXT_OP ucomis, type, op1_reg, [=>label]
 	} else {
-		ir_mem mem;
-
-		if (ir_rule(ctx, op2) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, root, op2);
-		} else {
-			mem = ir_ref_spill_slot(ctx, op2);
+		op2_reg_hi = IR_REG_NONE;
+	}
+	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
+		if (op == IR_ULT) {
+			/* always false */
+			|	xor Ra(def_reg), Ra(def_reg)
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store(ctx, insn->type, def, def_reg);
+			}
+			return;
+		} else if (op == IR_UGE) {
+			/* always true */
+			|	ASM_REG_IMM_OP mov, insn->type, def_reg, 1
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store(ctx, insn->type, def, def_reg);
+			}
+			return;
 		}
-		|	ASM_FP_REG_MEM_OP ucomis, type, op1_reg, mem
 	}
-	return op;
+	ir_emit_cmp_i64_common(ctx, op, def, def_reg, tmp_reg, op1_reg, op1_reg_hi, op1, op2_reg, op2_reg_hi, op2);
+	if (op == IR_EQ) {
+		|	sete Rb(def_reg)
+	} else if (op == IR_NE) {
+		|	setne Rb(def_reg)
+	} else if (op == IR_LT || op == IR_GT) {
+		|	setl Rb(def_reg)
+	} else if (op == IR_LE || op == IR_GE) {
+		|	setge Rb(def_reg)
+	} else if (op == IR_ULT || op == IR_UGT) {
+		|	setc Rb(def_reg)
+	} else if (op == IR_ULE || op == IR_UGE) {
+		|	setnc Rb(def_reg)
+	} else {
+		IR_ASSERT(0);
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
 }

-static void ir_emit_cmp_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_jcc_i64(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block, uint8_t op)
 {
+	uint32_t true_block, false_block;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_op op = ir_emit_cmp_fp_common(ctx, def, def, insn);
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg tmp_reg = ctx->regs[def][3];

-	IR_ASSERT(def_reg != IR_REG_NONE);
+	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+	if (true_block == next_block) {
+		/* swap to avoid unconditional JMP */
+		op ^= 1; // reverse
+		true_block = false_block;
+		false_block = 0;
+	} else if (false_block == next_block) {
+		false_block = 0;
+	}
+
 	switch (op) {
 		default:
 			IR_ASSERT(0 && "NIY binary op");
 		case IR_EQ:
-			|	setnp Rb(def_reg)
-			|	mov Rd(tmp_reg), 0
-			|	cmovne Rd(def_reg), Rd(tmp_reg)
+			|	je =>true_block
 			break;
 		case IR_NE:
-			|	setp Rb(def_reg)
-			|	mov Rd(tmp_reg), 1
-			|	cmovne Rd(def_reg), Rd(tmp_reg)
+			|	jne =>true_block
 			break;
 		case IR_LT:
-			|	setnp Rb(def_reg)
-			|	mov Rd(tmp_reg), 0
-			|	cmovae Rd(def_reg), Rd(tmp_reg)
+		case IR_GT:
+			|	jl =>true_block
 			break;
 		case IR_GE:
-			|	setae Rb(def_reg)
-			break;
 		case IR_LE:
-			|	setnp Rb(def_reg)
-			|	mov Rd(tmp_reg), 0
-			|	cmova Rd(def_reg), Rd(tmp_reg)
-			break;
-		case IR_GT:
-			|	seta Rb(def_reg)
+			|	jge =>true_block
 			break;
 		case IR_ULT:
-			|	setb Rb(def_reg)
+		case IR_UGT:
+			|	jc =>true_block
 			break;
 		case IR_UGE:
-			|	setp Rb(def_reg)
-			|	mov Rd(tmp_reg), 1
-			|	cmovae Rd(def_reg), Rd(tmp_reg)
-			break;
 		case IR_ULE:
-			|	setbe Rb(def_reg)
-			break;
-		case IR_UGT:
-			|	setp Rb(def_reg)
-			|	mov Rd(tmp_reg), 1
-			|	cmova Rd(def_reg), Rd(tmp_reg)
-			break;
-		case IR_ORDERED:
-			|	setnp Rb(def_reg)
-			break;
-		case IR_UNORDERED:
-			|	setp Rb(def_reg)
+			|	jnc =>true_block
 			break;
 	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
+	if (false_block) {
+		|	jmp =>false_block
 	}
 }

-static void ir_emit_jmp_true(ir_ctx *ctx, uint32_t b, ir_ref def, uint32_t next_block)
+static void ir_emit_cmp_and_branch_i64(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
 {
-	uint32_t true_block, false_block;
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op2];
+	ir_op op = cmp_insn->op;
+	ir_ref op1 = cmp_insn->op1;
+	ir_ref op2 = cmp_insn->op2;
+	ir_reg op1_reg, op2_reg;
+	ir_reg op1_reg_hi, op2_reg_hi;

-	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
-	if (true_block != next_block) {
-		|	jmp =>true_block
+	if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[insn->op2][1];
+		op2_reg = ctx->regs[insn->op2][2];
+	}
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+		}
+	} else {
+		op1_reg_hi = IR_REG_NONE;
+	}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load_i64_lo(ctx, op2_reg, op2);
+				ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+			}
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+		}
+	} else {
+		op2_reg_hi = IR_REG_NONE;
+	}
+	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
+		if (op == IR_ULT) {
+			/* always false */
+			ir_emit_jmp_false(ctx, b, def, next_block);
+			return;
+		} else if (op == IR_UGE) {
+			/* always true */
+			ir_emit_jmp_true(ctx, b, def, next_block);
+			return;
+		}
+	}
+
+	bool same_comparison = 0;
+	ir_insn *prev_insn = &ctx->ir_base[insn->op1];
+	if (prev_insn->op == IR_IF_TRUE || prev_insn->op == IR_IF_FALSE) {
+		if (ir_rule(ctx, prev_insn->op1) == IR_CMP_AND_BRANCH_INT) {
+			prev_insn = &ctx->ir_base[prev_insn->op1];
+			prev_insn = &ctx->ir_base[prev_insn->op2];
+			if (prev_insn->op1 == cmp_insn->op1 && prev_insn->op2 == cmp_insn->op2) {
+				same_comparison = true;
+			}
+		}
+	}
+	if (!same_comparison) {
+		ir_reg tmp_reg = ctx->regs[def][0];
+		ir_reg tmp2_reg = ctx->regs[insn->op2][3];
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+		ir_emit_cmp_i64_common(ctx, op, def, tmp_reg, tmp2_reg, op1_reg, op1_reg_hi, op1, op2_reg, op2_reg_hi, op2);
 	}
+	ir_emit_jcc_i64(ctx, b, def, insn, next_block, op);
 }

-static void ir_emit_jmp_false(ir_ctx *ctx, uint32_t b, ir_ref def, uint32_t next_block)
+static void ir_emit_load_i64(ir_ctx *ctx, ir_ref root, ir_reg def_reg, ir_reg def_reg_hi, ir_reg op1_reg, ir_reg op1_reg_hi, ir_ref op1)
 {
-	uint32_t true_block, false_block;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;

-	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
-	if (false_block != next_block) {
-		|	jmp =>false_block
+	if (def_reg != op1_reg_hi) {
+		if (def_reg != op1_reg) {
+			if (op1_reg != IR_REG_NONE) {
+				ir_emit_mov(ctx, IR_U32, def_reg, op1_reg);
+			} else if (!IR_IS_CONST_REF(op1) && (ir_rule(ctx, op1) & IR_FUSED)) {
+				ir_mem mem = ir_fuse_load(ctx, root, op1);
+				|	ASM_REG_MEM_OP mov, IR_U32, def_reg, mem
+			} else {
+				ir_emit_load_i64_lo(ctx, def_reg, op1);
+			}
+		}
+		if (def_reg_hi != op1_reg_hi) {
+			if (op1_reg_hi != IR_REG_NONE) {
+				ir_emit_mov(ctx, IR_U32, def_reg_hi, op1_reg_hi);
+			} else if (!IR_IS_CONST_REF(op1) && (ir_rule(ctx, op1) & IR_FUSED)) {
+				ir_mem mem = IR_MEM_I64_HI(ir_fuse_load(ctx, root, op1));
+				|	ASM_REG_MEM_OP mov, IR_U32, def_reg_hi, mem
+			} else {
+				ir_emit_load_i64_hi(ctx, def_reg_hi, op1);
+			}
+		}
+	} else {
+		if (def_reg_hi != op1_reg_hi) {
+			if (op1_reg_hi != IR_REG_NONE) {
+				ir_emit_mov(ctx, IR_U32, def_reg_hi, op1_reg_hi);
+			} else if (!IR_IS_CONST_REF(op1) && (ir_rule(ctx, op1) & IR_FUSED)) {
+				ir_mem mem = IR_MEM_I64_HI(ir_fuse_load(ctx, root, op1));
+				|	ASM_REG_MEM_OP mov, IR_U32, def_reg_hi, mem
+			} else {
+				ir_emit_load_i64_hi(ctx, def_reg_hi, op1);
+			}
+		}
+		if (def_reg != op1_reg) {
+			if (op1_reg != IR_REG_NONE) {
+				IR_ASSERT(def_reg_hi != op1_reg);
+				ir_emit_mov(ctx, IR_U32, def_reg, op1_reg);
+			} else if (!IR_IS_CONST_REF(op1) && (ir_rule(ctx, op1) & IR_FUSED)) {
+				ir_mem mem = ir_fuse_load(ctx, root, op1);
+				|	ASM_REG_MEM_OP mov, IR_U32, def_reg, mem
+			} else {
+				ir_emit_load_i64_lo(ctx, def_reg, op1);
+			}
+		}
 	}
 }

-static void ir_emit_jcc(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block, uint8_t op, bool int_cmp, bool after_op)
+static void ir_emit_binop_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	uint32_t true_block, false_block;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg def_reg_hi, op1_reg_hi, op2_reg_hi;

-	ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
-	if (true_block == next_block) {
-		/* swap to avoid unconditional JMP */
-		if (int_cmp || op == IR_EQ || op == IR_NE || op == IR_ORDERED || op == IR_UNORDERED) {
-			op ^= 1; // reverse
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
 		} else {
-			op ^= 5; // reverse
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-		true_block = false_block;
-		false_block = 0;
-	} else if (false_block == next_block) {
-		false_block = 0;
+	} else {
+		op1_reg_hi = IR_REG_NONE;
 	}

-	if (int_cmp) {
-		switch (op) {
+	ir_emit_load_i64(ctx, def, def_reg, def_reg_hi, op1_reg, op1_reg_hi, op1);
+
+	if (op1 == op2) {
+		op2_reg = def_reg;
+		op2_reg_hi = def_reg_hi;
+	}
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load_i64_lo(ctx, op2_reg, op2);
+				ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+			}
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+		}
+		switch (insn->op) {
 			default:
 				IR_ASSERT(0 && "NIY binary op");
-			case IR_EQ:
-				|	je =>true_block
-				break;
-			case IR_NE:
-				|	jne =>true_block
-				break;
-			case IR_LT:
-				if (after_op) {
-					|	js =>true_block
-				} else {
-					|	jl =>true_block
-				}
-				break;
-			case IR_GE:
-				if (after_op) {
-					|	jns =>true_block
-				} else {
-					|	jge =>true_block
-				}
-				break;
-			case IR_LE:
-				|	jle =>true_block
-				break;
-			case IR_GT:
-				|	jg =>true_block
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	add Rd(def_reg), Rd(op2_reg)
+				|	adc Rd(def_reg_hi), Rd(op2_reg_hi)
 				break;
-			case IR_ULT:
-				|	jb =>true_block
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	sub Rd(def_reg), Rd(op2_reg)
+				|	sbb Rd(def_reg_hi), Rd(op2_reg_hi)
 				break;
-			case IR_UGE:
-				|	jae =>true_block
+			case IR_OR:
+				|	or Rd(def_reg), Rd(op2_reg)
+				|	or Rd(def_reg_hi), Rd(op2_reg_hi)
 				break;
-			case IR_ULE:
-				|	jbe =>true_block
+			case IR_AND:
+				|	and Rd(def_reg), Rd(op2_reg)
+				|	and Rd(def_reg_hi), Rd(op2_reg_hi)
 				break;
-			case IR_UGT:
-				|	ja =>true_block
+			case IR_XOR:
+				|	xor Rd(def_reg), Rd(op2_reg)
+				|	xor Rd(def_reg_hi), Rd(op2_reg_hi)
 				break;
 		}
-	} else {
-		switch (op) {
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];
+
+		switch (insn->op) {
 			default:
 				IR_ASSERT(0 && "NIY binary op");
-			case IR_EQ:
-				if (!false_block) {
-					|	jp >1
-					|	je =>true_block
-					|1:
+			case IR_ADD:
+			case IR_ADD_OV:
+				if (val->val.u32) {
+					|	add Rd(def_reg), val->val.u32
+					|	adc Rd(def_reg_hi), val->val.u32_hi
 				} else {
-					|	jp =>false_block
-					|	je =>true_block
+					|	add Rd(def_reg_hi), val->val.u32_hi
 				}
 				break;
-			case IR_NE:
-				|	jne =>true_block
-				|	jp =>true_block
-				break;
-			case IR_LT:
-				if (!false_block) {
-					|	jp >1
-					|	jb =>true_block
-					|1:
+			case IR_SUB:
+			case IR_SUB_OV:
+				if (val->val.u32) {
+					|	sub Rd(def_reg), val->val.u32
+					|	sbb Rd(def_reg_hi), val->val.u32_hi
 				} else {
-					|	jp =>false_block
-					|	jb =>true_block
+					|	sub Rd(def_reg_hi), val->val.u32_hi
 				}
 				break;
-			case IR_GE:
-				|	jae =>true_block
-				break;
-			case IR_LE:
-				if (!false_block) {
-					|	jp >1
-					|	jbe =>true_block
-					|1:
-				} else {
-					|	jp =>false_block
-					|	jbe =>true_block
+			case IR_OR:
+				if (val->val.u32) {
+					|	or Rd(def_reg), val->val.u32
+				}
+				if (val->val.u32_hi) {
+					|	or Rd(def_reg_hi), val->val.u32_hi
 				}
 				break;
-			case IR_GT:
-				|	ja =>true_block
+			case IR_AND:
+				if (val->val.u32 != 0xffffffff) {
+					|	and Rd(def_reg), val->val.u32
+				}
+				if (val->val.u32_hi != 0xffffffff) {
+					|	and Rd(def_reg_hi), val->val.u32_hi
+				}
 				break;
-			case IR_ULT:
-				|	jb =>true_block
+			case IR_XOR:
+				if (val->val.u32) {
+					|	xor Rd(def_reg), val->val.u32
+				}
+				if (val->val.u32_hi) {
+					|	xor Rd(def_reg_hi), val->val.u32_hi
+				}
 				break;
-			case IR_UGE:
-				|	jp =>true_block
-				|	jae =>true_block
+		}
+	} else {
+		ir_mem mem_lo, mem_hi;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem_lo = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem_lo = ir_ref_spill_slot(ctx, op2);
+		}
+		mem_hi = IR_MEM_I64_HI(mem_lo);
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+			case IR_ADD_OV:
+				|	ASM_REG_MEM_OP add, IR_U32, def_reg, mem_lo
+				|	ASM_REG_MEM_OP adc, IR_U32, def_reg_hi, mem_hi
 				break;
-			case IR_ULE:
-				|	jbe =>true_block
+			case IR_SUB:
+			case IR_SUB_OV:
+				|	ASM_REG_MEM_OP sub, IR_U32, def_reg, mem_lo
+				|	ASM_REG_MEM_OP sbb, IR_U32, def_reg_hi, mem_hi
 				break;
-			case IR_UGT:
-				|	jp =>true_block
-				|	ja =>true_block
+			case IR_OR:
+				|	ASM_REG_MEM_OP or, IR_U32, def_reg, mem_lo
+				|	ASM_REG_MEM_OP or, IR_U32, def_reg_hi, mem_hi
 				break;
-			case IR_ORDERED:
-				|	jnp =>true_block
+			case IR_AND:
+				|	ASM_REG_MEM_OP and, IR_U32, def_reg, mem_lo
+				|	ASM_REG_MEM_OP and, IR_U32, def_reg_hi, mem_hi
 				break;
-			case IR_UNORDERED:
-				|	jp =>true_block
+			case IR_XOR:
+				|	ASM_REG_MEM_OP xor, IR_U32, def_reg, mem_lo
+				|	ASM_REG_MEM_OP xor, IR_U32, def_reg_hi, mem_hi
 				break;
 		}
 	}
-	if (false_block) {
-		|	jmp =>false_block
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_cmp_and_branch_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+static void ir_emit_mul_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_insn *cmp_insn = &ctx->ir_base[insn->op2];
-	ir_op op = cmp_insn->op;
-	ir_type type = ctx->ir_base[cmp_insn->op1].type;
-	ir_ref op1 = cmp_insn->op1;
-	ir_ref op2 = cmp_insn->op2;
-	ir_reg op1_reg, op2_reg;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	ir_reg def_reg_hi, op1_reg_hi, op2_reg_hi;

-	if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
-		op1_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 1);
-		op2_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 2);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+		}
 	} else {
-		op1_reg = ctx->regs[insn->op2][1];
-		op2_reg = ctx->regs[insn->op2][2];
+		op1_reg_hi = IR_REG_NONE;
 	}

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
-	}
-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		if (op1 != op2) {
-			ir_emit_load(ctx, type, op2_reg, op2);
+	ir_emit_load_i64(ctx, def, IR_REG_RAX, IR_REG_RDX, op1_reg, op1_reg_hi, op1);
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-	}
-	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
-		if (op == IR_ULT) {
-			/* always false */
-			ir_emit_jmp_false(ctx, b, def, next_block);
-			return;
-		} else if (op == IR_UGE) {
-			/* always true */
-			ir_emit_jmp_true(ctx, b, def, next_block);
-			return;
-		} else if (op == IR_ULE) {
-			op = IR_EQ;
-		} else if (op == IR_UGT) {
-			op = IR_NE;
+
+		IR_ASSERT(op2_reg != IR_REG_RDX && tmp_reg != op2_reg_hi && tmp_reg != op2_reg);
+
+		|	mov Rd(tmp_reg), Rd(op2_reg_hi)
+		|	imul edx, Rd(op2_reg)
+		|	imul Rd(tmp_reg), eax
+		|	add Rd(tmp_reg), edx
+		|	mul Rd(op2_reg)
+		|	add edx, Rd(tmp_reg)
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];
+
+		if (val->val.u32_hi && val->val.u32) {
+			|	imul edx, val->val.u32
+			|	imul Rd(tmp_reg), eax, val->val.u32_hi
+			|	add Rd(tmp_reg), edx
+			|	mov edx, val->val.u32
+			|	mul edx
+			|	add edx, Rd(tmp_reg)
+		} else if (val->val.u32_hi && !val->val.u32) {
+			|	imul edx, eax, val->val.u32_hi
+			|	xor eax, eax
+		} else {
+			|	imul Rd(tmp_reg), edx, val->val.u32
+			|	mov edx, val->val.u32
+			|	mul edx
+			|	add edx, Rd(tmp_reg)
+		}
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);
+
+		|	ASM_TXT_TMEM_OP mov, Rd(tmp_reg), dword, mem_hi
+		|	ASM_TXT_TMEM_OP imul, edx, dword, mem
+		|	imul Rd(tmp_reg), eax
+		|	add Rd(tmp_reg), edx
+		|	ASM_TMEM_OP mul, dword, mem
+		|	add edx, Rd(tmp_reg)
 	}

-	bool same_comparison = 0;
-	ir_insn *prev_insn = &ctx->ir_base[insn->op1];
-	if (prev_insn->op == IR_IF_TRUE || prev_insn->op == IR_IF_FALSE) {
-		if (ir_rule(ctx, prev_insn->op1) == IR_CMP_AND_BRANCH_INT) {
-			prev_insn = &ctx->ir_base[prev_insn->op1];
-			prev_insn = &ctx->ir_base[prev_insn->op2];
-			if (prev_insn->op1 == cmp_insn->op1 && prev_insn->op2 == cmp_insn->op2) {
-				same_comparison = true;
+	if (def_reg != IR_REG_NONE) {
+		def_reg_hi = IR_REG_I64_HI(def_reg);
+		def_reg = IR_REG_I64_LO(def_reg);
+		if (def_reg != IR_REG_RDX) {
+			if (def_reg != IR_REG_RAX) {
+				|	mov Rd(def_reg), eax
+			}
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+		} else {
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+			if (def_reg != IR_REG_RAX) {
+				IR_ASSERT(def_reg_hi != IR_REG_RAX);
+				|	mov Rd(def_reg), eax
 			}
 		}
+
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, def_reg);
+			ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+		}
+	} else {
+		ir_emit_store_i64_lo(ctx, def, IR_REG_RAX);
+		ir_emit_store_i64_hi(ctx, def, IR_REG_RDX);
 	}
-	if (!same_comparison) {
-		ir_emit_cmp_int_common(ctx, type, def, cmp_insn, op1_reg, op1, op2_reg, op2);
-	}
-	ir_emit_jcc(ctx, b, def, insn, next_block, op, 1, 0);
 }

-static void ir_emit_test_and_branch_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+static void ir_emit_mul_ov_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_ref op1 = insn->op1;
 	ir_ref op2 = insn->op2;
-	ir_op op = ctx->ir_base[op2].op;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	ir_reg tmp2_reg = ctx->tmp_regs[def];
+	ir_reg def_reg_hi, op1_reg_hi, op2_reg_hi;

-	if (op >= IR_EQ && op <= IR_UGT) {
-		op2 = ctx->ir_base[op2].op1;
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+		}
 	} else {
-		IR_ASSERT(op == IR_AND);
-		op = IR_NE;
+		op1_reg_hi = IR_REG_NONE;
 	}

-	ir_emit_test_int_common(ctx, def, op2, op);
-	ir_emit_jcc(ctx, b, def, insn, next_block, op, 1, 0);
-}
-
-static void ir_emit_cmp_and_branch_fp(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
-{
-	ir_op op = ir_emit_cmp_fp_common(ctx, def, insn->op2, &ctx->ir_base[insn->op2]);
-	ir_emit_jcc(ctx, b, def, insn, next_block, op, 0, 0);
-}
-
-static void ir_emit_if_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
-{
-	ir_type type = ctx->ir_base[insn->op2].type;
-	ir_reg op2_reg = ctx->regs[def][2];
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
+	ir_emit_load_i64(ctx, def, IR_REG_RAX, IR_REG_RDX, op1_reg, op1_reg_hi, op1);

 	if (op2_reg != IR_REG_NONE) {
 		if (IR_REG_SPILLED(op2_reg)) {
 			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, insn->op2);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-		|	ASM_REG_REG_OP test, type, op2_reg, op2_reg
-	} else if (IR_IS_CONST_REF(insn->op2)) {
-		uint32_t true_block, false_block;

-		ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
-		if (ir_const_is_true(&ctx->ir_base[insn->op2])) {
-			if (true_block != next_block) {
-				|	jmp =>true_block
-			}
+		IR_ASSERT(op2_reg != IR_REG_RDX && tmp_reg != op2_reg_hi && tmp_reg != op2_reg);
+
+		if (type == IR_I64) {
+			|	mov Rd(tmp2_reg), Rd(op2_reg_hi)                   // expected sign = sign A ^ sign B
+			|	xor Rd(tmp2_reg), edx
+			|	and Rd(tmp2_reg), 0x80000000
 		} else {
-			if (false_block != next_block) {
-				|	jmp =>false_block
-			}
+			|	xor Rd(tmp2_reg), Rd(tmp2_reg)
+		}
+
+		|	mov Rd(tmp_reg), edx
+		|	imul Rd(tmp_reg), Rd(op2_reg_hi)                       // hi(Ah * Bh) == 0
+		|	adc Rd(tmp2_reg), 0
+		|	add Rd(tmp_reg), -1                                    // lo(Ah * Bh) == 0 ; set C flag if tmp_reg != 0
+		|	adc Rd(tmp2_reg), 0
+		|	mov Rd(tmp_reg), Rd(op2_reg_hi)
+		|	imul edx, Rd(op2_reg)                                  // hi(Ah * Bl) == 0
+		|	adc Rd(tmp2_reg), 0
+		|	imul Rd(tmp_reg), eax                                  // hi(Al * Bh) == 0
+		|	adc Rd(tmp2_reg), 0
+		|	add Rd(tmp_reg), edx                                   // c = lo(Ah * Bl) + lo(Al * Bh)
+		|	adc Rd(tmp2_reg), 0
+		|	mul Rd(op2_reg)                                        // Al * Bl + c
+		|	add edx, Rd(tmp_reg)
+
+		if (type == IR_I64) {
+			|	adc Rd(tmp2_reg), 0
+			|	mov Rd(tmp_reg), edx                               // check sign of result
+			|	xor Rd(tmp_reg), Rd(tmp2_reg)
+			|	and Rd(tmp2_reg), 0x7f
+			|	and Rd(tmp_reg), 0x80000000
+			|	add Rd(tmp_reg), -1
+			|	adc Rd(tmp2_reg), 0x7fffffff                       // set OV flag if tmp2_reg != 0
+	    } else {
+			|	adc Rd(tmp2_reg), -1                               // set C flag if UMUL overflow
 		}
-		return;
-	} else if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
-		uint32_t true_block, false_block;
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];

-		ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
-		if (true_block != next_block) {
-			|	jmp =>true_block
+		if (type == IR_I64) {
+			|	mov Rd(tmp2_reg), edx                              // expected sign = sign A ^ sign B
+			if (val->val.i64 >= 0) {
+				|	and Rd(tmp2_reg), 0x80000000
+			} else {
+				|	not Rd(tmp2_reg)
+				|	and Rd(tmp2_reg), 0x80000000
+			}
+	    } else {
+			|	xor Rd(tmp2_reg), Rd(tmp2_reg)
+		}
+
+		if (val->val.u32_hi) {
+			|	mov Rd(tmp_reg), edx
+			|	imul Rd(tmp_reg), edx, val->val.u32_hi             // hi(Ah * Bh) == 0
+			|	adc Rd(tmp2_reg), 0
+			|	add Rd(tmp_reg), -1                                // lo(Ah * Bh) == 0; set C flag if tmp_reg != 0
+			|	adc Rd(tmp2_reg), 0
+			|	mov Rd(tmp_reg), val->val.u32_hi
+			|	imul edx, val->val.u32                             // hi(Ah * Bl) == 0
+			|	adc Rd(tmp2_reg), 0
+			|	imul Rd(tmp_reg), eax                              // hi(Al * Bh) == 0
+			|	adc Rd(tmp2_reg), 0
+			|	add Rd(tmp_reg), edx                               // c = lo(Ah * Bl) + lo(Al * Bh)
+			|	adc Rd(tmp2_reg), 0
+		} else {
+			|	mov Rd(tmp_reg), edx
+			|	imul Rd(tmp_reg), val->val.u32                     // hi(Ah * Bl) == 0, c = lo(Ah * Bl)
+			|	adc Rd(tmp2_reg), 0
+		}
+		|	mov edx, val->val.u32                                  // Al * Bl + c
+		|	mul edx
+		|	add edx, Rd(tmp_reg)
+		|	adc Rd(tmp2_reg), -1                                   // set C flag if UMUL overflow
+
+		if (type == IR_I64) {
+			|	adc Rd(tmp2_reg), 0
+			|	mov Rd(tmp_reg), edx                               // check sign of result
+			|	xor Rd(tmp_reg), Rd(tmp2_reg)
+			|	and Rd(tmp2_reg), 0x7f
+			|	and Rd(tmp_reg), 0x80000000
+			|	add Rd(tmp_reg), -1
+			|	adc Rd(tmp2_reg), 0x7fffffff                       // set OV flag if tmp2_reg != 0
+	    } else {
+			|   adc Rd(tmp2_reg), -1                               // set C flag if UMUL overflow
+		}
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+		}
+		mem_hi = IR_MEM_I64_HI(mem);
+
+		if (type == IR_I64) {
+			|	ASM_TXT_TMEM_OP mov, Rd(tmp2_reg), dword, mem_hi   // expected sign = sign A ^ sign B
+			|	xor Rd(tmp2_reg), edx
+			|	and Rd(tmp2_reg), 0x80000000
+		} else {
+			|	xor Rd(tmp2_reg), Rd(tmp2_reg)
+		}
+
+		|	mov Rd(tmp_reg), edx
+		|	ASM_TXT_TMEM_OP imul, Rd(tmp_reg), dword, mem_hi       // hi(Ah * Bh) == 0
+		|	adc Rd(tmp2_reg), 0
+		|	add Rd(tmp_reg), -1                                    // lo(Ah * Bh) == 0 ; set C flag if tmp_reg != 0
+		|	adc Rd(tmp2_reg), 0
+		|	ASM_TXT_TMEM_OP mov, Rd(tmp_reg), dword, mem_hi
+		|	ASM_TXT_TMEM_OP imul, edx, dword, mem                  // hi(Ah * Bl) == 0
+		|	adc Rd(tmp2_reg), 0
+		|	imul Rd(tmp_reg), eax                                  // hi(Al * Bh) == 0
+		|	adc Rd(tmp2_reg), 0
+		|	add Rd(tmp_reg), edx                                   // c = lo(Ah * Bl) + lo(Al * Bh)
+		|	adc Rd(tmp2_reg), 0
+		|	ASM_TMEM_OP mul, dword, mem                            // Al * Bl + c
+		|	add edx, Rd(tmp_reg)
+
+		if (type == IR_I64) {
+			|	adc Rd(tmp2_reg), 0
+			|	mov Rd(tmp_reg), edx                               // check sign of result
+			|	xor Rd(tmp_reg), Rd(tmp2_reg)
+			|	and Rd(tmp2_reg), 0x7f
+			|	and Rd(tmp_reg), 0x80000000
+			|	add Rd(tmp_reg), -1
+			|	adc Rd(tmp2_reg), 0x7fffffff                       // set OV flag if tmp2_reg != 0
+	    } else {
+			|   adc Rd(tmp2_reg), -1                               // set C flag if UMUL overflow
+		}
+	}
+
+	if (def_reg != IR_REG_NONE) {
+		def_reg_hi = IR_REG_I64_HI(def_reg);
+		def_reg = IR_REG_I64_LO(def_reg);
+		if (def_reg != IR_REG_RDX) {
+			if (def_reg != IR_REG_RAX) {
+				|	mov Rd(def_reg), eax
+			}
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+		} else {
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+			if (def_reg != IR_REG_RAX) {
+				IR_ASSERT(def_reg_hi != IR_REG_RAX);
+				|	mov Rd(def_reg), eax
+			}
 		}
-		return;
-	} else {
-		ir_mem mem;

-		if (ir_rule(ctx, insn->op2) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, insn->op2);
-		} else {
-			mem = ir_ref_spill_slot(ctx, insn->op2);
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, def_reg);
+			ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 		}
-		|	ASM_MEM_IMM_OP cmp, type, mem, 0
+	} else {
+		ir_emit_store_i64_lo(ctx, def, IR_REG_RAX);
+		ir_emit_store_i64_hi(ctx, def, IR_REG_RDX);
 	}
-	ir_emit_jcc(ctx, b, def, insn, next_block, IR_NE, 1, 0);
 }

-static void ir_emit_cond(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+#ifdef _WIN32
+extern int64_t _alldiv(int64_t, int64_t);
+extern int64_t _allrem(int64_t, int64_t);
+extern uint64_t _aulldiv(uint64_t, uint64_t);
+extern uint64_t _aullrem(uint64_t, uint64_t);
+# define __divdi3   _alldiv
+# define __moddi3   _allrem
+# define __udivdi3  _aulldiv
+# define __umoddi3  _aullrem
+
+static int __popcountdi2(uint64_t arg)
+{
+	unsigned y = (unsigned)(arg >> 32ULL);
+	unsigned x = (unsigned)arg;
+
+    x = x - ((x >> 1) & 0x55555555);
+    x = (x & 0x33333333) + ((x >> 2) & 0x33333333);
+    x = (x + (x >> 4)) & 0x0f0f0f0f;
+    x = x + (x >> 8);
+    x = x + (x >> 16);
+    y = y - ((y >> 1) & 0x55555555);
+    y = (y & 0x33333333) + ((y >> 2) & 0x33333333);
+    y = (y + (y >> 4)) & 0x0f0f0f0f;
+    y = y + (y >> 8);
+    y = y + (y >> 16);
+    return (x & 0x3F) + (y & 0x3f);
+}
+#else
+extern int64_t __divdi3(int64_t, int64_t);
+extern int64_t __moddi3(int64_t, int64_t);
+extern uint64_t __udivdi3(uint64_t, uint64_t);
+extern uint64_t __umoddi3(uint64_t, uint64_t);
+extern int32_t __popcountdi2(uint64_t);
+#endif
+
+static uint64_t __roldi3(uint64_t a, uint64_t b)
+{
+	b &= 63;
+	return (a << b) | (a >> (64 - b));
+}
+
+static uint64_t __rordi3(uint64_t a, uint64_t b)
+{
+	b &= 63;
+	return (a >> b) | (a << (64 - b));
+}
+
+static void ir_emit_binop_helper_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
 	ir_ref op1 = insn->op1;
 	ir_ref op2 = insn->op2;
-	ir_ref op3 = insn->op3;
-	ir_type op1_type = ctx->ir_base[op1].type;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg op2_reg = ctx->regs[def][2];
-	ir_reg op3_reg = ctx->regs[def][3];
-
-	IR_ASSERT(def_reg != IR_REG_NONE);
+	ir_reg op1_reg_hi, op2_reg_hi, def_reg_hi;
+	void *addr;

-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, op2);
-		if (op1 == op2) {
-			op1_reg = op2_reg;
+	|	sub esp, 12
+	ctx->call_stack_size += 12;
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load_i64_lo(ctx, op2_reg, op2);
+				ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+			}
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-		if (op3 == op2) {
-			op3_reg = op2_reg;
+		|	push Rd(op2_reg_hi)
+		|	push Rd(op2_reg)
+		ctx->call_stack_size += 8;
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];
+
+		|	push dword val->val.u32_hi
+		|	push dword val->val.u32
+		ctx->call_stack_size += 8;
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+			mem_hi = IR_MEM_I64_HI(mem);
+			ctx->call_stack_size += 4;
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+			mem_hi = IR_MEM_I64_HI(mem);
+			ctx->call_stack_size += 4;
 		}
-	}
-	if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-		op3_reg = IR_REG_NUM(op3_reg);
-		ir_emit_load(ctx, type, op3_reg, op3);
-		if (op1 == op3) {
-			op1_reg = op2_reg;
+		ctx->call_stack_size += 4;
+		|	ASM_MEM_PUSH_OP push, IR_U32, mem_hi
+		if (IR_MEM_BASE(mem) == IR_REG_RSP) {
+			|	ASM_MEM_PUSH_OP push, IR_U32, mem_hi
+		} else {
+			|	ASM_MEM_PUSH_OP push, IR_U32, mem
 		}
 	}
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, op1_type, op1_reg, op1);
-	}

-	if (IR_IS_TYPE_INT(op1_type)) {
-		if (op1_reg != IR_REG_NONE) {
-			|	ASM_REG_REG_OP test, op1_type, op1_reg, op1_reg
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
 		} else {
-			ir_mem mem;
-
-			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-				mem = ir_fuse_load(ctx, def, insn->op1);
-			} else {
-				mem = ir_ref_spill_slot(ctx, insn->op1);
-			}
-
-			|	ASM_MEM_IMM_OP cmp, op1_type, mem, 0
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-		if (IR_IS_TYPE_INT(type)) {
-			IR_ASSERT(op2_reg != IR_REG_NONE || op3_reg != IR_REG_NONE);
-			if (op3_reg != IR_REG_NONE) {
-				if (op3_reg == def_reg) {
-					IR_ASSERT(op2_reg != IR_REG_NONE);
-					|	ASM_REG_REG_OP2 cmovne, type, def_reg, op2_reg
-				} else {
-					if (op2_reg != IR_REG_NONE) {
-						if (def_reg != op2_reg) {
-							if (IR_IS_TYPE_INT(type)) {
-								ir_emit_mov(ctx, type, def_reg, op2_reg);
-							} else {
-								ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
-							}
-						}
-					} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
-						/* prevent "xor" and flags clobbering */
-						ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
-					} else {
-						ir_emit_load_ex(ctx, type, def_reg, op2, def);
-					}
-					|	ASM_REG_REG_OP2 cmove, type, def_reg, op3_reg
-				}
-			} else {
-				IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
-				if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
-					/* prevent "xor" and flags clobbering */
-					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
-				} else {
-					ir_emit_load_ex(ctx, type, def_reg, op3, def);
-				}
-				|	ASM_REG_REG_OP2 cmovne, type, def_reg, op2_reg
-			}
+		|	push Rd(op1_reg_hi)
+		|	push Rd(op1_reg)
+		ctx->call_stack_size += 8;
+	} else if (IR_IS_CONST_REF(op1)) {
+		ir_insn *val = &ctx->ir_base[op1];

-			if (IR_REG_SPILLED(ctx->regs[def][0])) {
-				ir_emit_store(ctx, type, def, def_reg);
-			}
-			return;
-		}
-		|	je >2
+		|	push dword val->val.u32_hi
+		|	push dword val->val.u32
+		ctx->call_stack_size += 8;
 	} else {
-		if (!data->double_zero_const) {
-			data->double_zero_const = 1;
-			ir_rodata(ctx);
-			|.align 16
-			|->double_zero_const:
-			|.dword 0, 0
-			|.code
-		}
-		|	ASM_FP_REG_TXT_OP ucomis, op1_type, op1_reg, [->double_zero_const]
-		|	jp >1
-		|	je >2
-		|1:
-	}
+		ir_mem mem, mem_hi;

-	if (op2_reg != IR_REG_NONE) {
-		if (def_reg != op2_reg) {
-			if (IR_IS_TYPE_INT(type)) {
-				ir_emit_mov(ctx, type, def_reg, op2_reg);
-			} else {
-				ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
-			}
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op1);
+			mem_hi = IR_MEM_I64_HI(mem);
+			ctx->call_stack_size += 4;
+		} else {
+			mem = ir_ref_spill_slot(ctx, op1);
+			mem_hi = IR_MEM_I64_HI(mem);
+			ctx->call_stack_size += 4;
 		}
+		ctx->call_stack_size += 4;
+		|	ASM_MEM_PUSH_OP push, IR_U32, mem_hi
+		if (IR_MEM_BASE(mem) == IR_REG_RSP) {
+			|	ASM_MEM_PUSH_OP push, IR_U32, mem_hi
+		} else {
+			|	ASM_MEM_PUSH_OP push, IR_U32, mem
+		}
+	}
+
+	if (insn->opt == IR_OPT(IR_DIV, IR_I64)) {
+		addr = __divdi3;
+	} else if (insn->opt == IR_OPT(IR_DIV, IR_U64)) {
+		addr = __udivdi3;
+	} else if (insn->opt == IR_OPT(IR_MOD, IR_I64)) {
+		addr = __moddi3;
+	} else if (insn->opt == IR_OPT(IR_MOD, IR_U64)) {
+		addr = __umoddi3;
+	} else if (insn->op == IR_ROL) {
+		addr = __roldi3;
+	} else if (insn->op == IR_ROR) {
+		addr = __rordi3;
 	} else {
-		ir_emit_load_ex(ctx, type, def_reg, op2, def);
+		IR_ASSERT(0);
+		addr = NULL;
 	}
-	|	jmp >3
-	|2:
-	if (op3_reg != IR_REG_NONE) {
-		if (def_reg != op3_reg) {
-			if (IR_IS_TYPE_INT(type)) {
-				ir_emit_mov(ctx, type, def_reg, op3_reg);
-			} else {
-				ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
-			}
-		}
+	|	call aword &addr
+#ifdef _WIN32
+	if (insn->op == IR_ROL || insn->op == IR_ROR) {
+		|	add esp, 28
 	} else {
-		ir_emit_load_ex(ctx, type, def_reg, op3, def);
+		/* Windows helpers use fastcall calling convention */
+		|	add esp, 12
 	}
-	|3:
+#else
+	|	add esp, 28
+#endif
+	ctx->call_stack_size -= 28;

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+	if (def_reg != IR_REG_NONE) {
+		def_reg_hi = IR_REG_I64_HI(def_reg);
+		def_reg = IR_REG_I64_LO(def_reg);
+
+		if (def_reg != IR_REG_RDX) {
+			if (def_reg != IR_REG_RAX) {
+				|	mov Rd(def_reg), eax
+			}
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+		} else {
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+			if (def_reg != IR_REG_RAX) {
+				IR_ASSERT(def_reg_hi != IR_REG_RAX);
+				|	mov Rd(def_reg), eax
+			}
+		}
+
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, IR_REG_RAX);
+			ir_emit_store_i64_hi(ctx, def, IR_REG_RDX);
+		}
+	} else {
+		ir_emit_store_i64_lo(ctx, def, IR_REG_RAX);
+		ir_emit_store_i64_hi(ctx, def, IR_REG_RDX);
 	}
 }

-static void ir_emit_cond_test_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_bit_count_helper_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op2 = insn->op2;
-	ir_ref op3 = insn->op3;
+	ir_ref op1 = insn->op1;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op2_reg = ctx->regs[def][2];
-	ir_reg op3_reg = ctx->regs[def][3];
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op1_reg_hi;
+	void *addr;

-	if (op2 != op3) {
-		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, op2);
+	|	sub esp, 4
+	ctx->call_stack_size += 4;
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, op3);
+		|	push Rd(op1_reg_hi)
+		|	push Rd(op1_reg)
+		ctx->call_stack_size += 8;
+	} else if (IR_IS_CONST_REF(op1)) {
+		ir_insn *val = &ctx->ir_base[op1];
+
+		|	push dword val->val.u32_hi
+		|	push dword val->val.u32
+		ctx->call_stack_size += 8;
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem_hi = IR_MEM_I64_HI(ir_fuse_load(ctx, def, op1));
+			ctx->call_stack_size += 4;
+			mem = ir_fuse_load(ctx, def, op1);
+		} else {
+			mem_hi = IR_MEM_I64_HI(ir_ref_spill_slot(ctx, op1));
+			ctx->call_stack_size += 4;
+			mem = ir_ref_spill_slot(ctx, op1);
 		}
-	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, op2);
-		op3_reg = op2_reg;
-	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-		op3_reg = IR_REG_NUM(op3_reg);
-		ir_emit_load(ctx, type, op3_reg, op3);
-		op2_reg = op3_reg;
+		ctx->call_stack_size += 4;
+		|	ASM_MEM_PUSH_OP push, IR_U32, mem_hi
+		|	ASM_MEM_PUSH_OP push, IR_U32, mem
 	}

-	ir_emit_test_int_common(ctx, def, insn->op1, IR_NE);
+	if (insn->op == IR_CTPOP) {
+		addr = __popcountdi2;
+	} else {
+		IR_ASSERT(0);
+		addr = NULL;
+	}
+	|	call aword &addr
+	|	add esp, 12
+	ctx->call_stack_size -= 12;

-	if (IR_IS_TYPE_INT(type)) {
-		bool eq = 0;
+	if (def_reg != IR_REG_NONE) {
+		if (def_reg != IR_REG_RAX) {
+			|	mov Rd(def_reg), eax
+		}

-		if (op3_reg != IR_REG_NONE) {
-			if (op3_reg == def_reg) {
-				IR_ASSERT(op2_reg != IR_REG_NONE);
-				op3_reg = op2_reg;
-				eq = 1; // reverse
-			} else {
-				if (op2_reg != IR_REG_NONE) {
-					if (def_reg != op2_reg) {
-//						if (IR_IS_TYPE_INT(type)) {
-							ir_emit_mov(ctx, type, def_reg, op2_reg);
-//						} else {
-//							ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
-//						}
-					}
-				} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
-					/* prevent "xor" and flags clobbering */
-					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
-				} else {
-					ir_emit_load_ex(ctx, type, def_reg, op2, def);
-				}
-			}
-		} else {
-			IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
-			if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
-				/* prevent "xor" and flags clobbering */
-				ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
-			} else {
-				ir_emit_load_ex(ctx, type, def_reg, op3, def);
-			}
-			op3_reg = op2_reg;
-			eq = 1; // reverse
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store(ctx, insn->type, def, IR_REG_RAX);
 		}
+	} else {
+		ir_emit_store(ctx, insn->type, def, IR_REG_RAX);
+	}
+}

-		if (eq) {
-			|	ASM_REG_REG_OP2 cmovne, type, def_reg, op3_reg
+static void ir_emit_op_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_ref op1 = insn->op1;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_ref def_reg_hi, op1_reg_hi;
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
 		} else {
-			|	ASM_REG_REG_OP2 cmove, type, def_reg, op3_reg
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
 	} else {
-		|	jne >2
+		op1_reg_hi = IR_REG_NONE;
+	}
+
+	ir_emit_load_i64(ctx, def, def_reg, def_reg_hi, op1_reg, op1_reg_hi, op1);
+
+	if (insn->op == IR_COPY || insn->op == IR_BITCAST ||
+		insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR) {
+	} else if (insn->op == IR_NOT) {
+		|	not Rd(def_reg)
+		|	not Rd(def_reg_hi)
+	} else if (insn->op == IR_NEG) {
+		|	neg Rd(def_reg_hi)
+		|	neg Rd(def_reg)
+		|	sbb Rd(def_reg_hi), 0
+	} else if (insn->op == IR_ABS) {
+		|	test Rd(def_reg_hi), Rd(def_reg_hi)
+		|	jge >1
+		|	neg Rd(def_reg_hi)
+		|	neg Rd(def_reg)
+		|	sbb Rd(def_reg_hi), 0
 		|1:
+	} else if (insn->op == IR_BSWAP) {
+		|	bswap Rd(def_reg)
+		|	bswap Rd(def_reg_hi)
+		|	xchg Rd(def_reg), Rd(def_reg_hi)
+	} else {
+		IR_ASSERT(0);
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+	}
+}

-		if (op2_reg != IR_REG_NONE) {
-			if (def_reg != op2_reg) {
-				if (IR_IS_TYPE_INT(type)) {
-					ir_emit_mov(ctx, type, def_reg, op2_reg);
-				} else {
-					ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
-				}
-			}
+static void ir_emit_sext_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg def_reg_hi;
+
+	if (ir_type_size[ctx->ir_base[insn->op1].type] < 4) {
+		ir_emit_sext_common(ctx, def, insn, IR_I32, IR_REG_RAX);
+	} else if (op1_reg != IR_REG_RAX) {
+		if (op1_reg != IR_REG_NONE && !IR_REG_SPILLED(op1_reg)) {
+			|	mov eax, Rd(op1_reg)
 		} else {
-			ir_emit_load_ex(ctx, type, def_reg, op2, def);
+			ir_emit_load(ctx, IR_I32, IR_REG_RAX, insn->op1);
 		}
-		|	jmp >3
-		|2:
-		if (op3_reg != IR_REG_NONE) {
-			if (def_reg != op3_reg) {
-				if (IR_IS_TYPE_INT(type)) {
-					ir_emit_mov(ctx, type, def_reg, op3_reg);
-				} else {
-					ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
-				}
+	}
+	|	cdq
+
+	if (def_reg != IR_REG_NONE) {
+		def_reg_hi = IR_REG_I64_HI(def_reg);
+		def_reg = IR_REG_I64_LO(def_reg);
+		if (def_reg != IR_REG_RDX) {
+			if (def_reg != IR_REG_RAX) {
+				|	mov Rd(def_reg), eax
+			}
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
 			}
 		} else {
-			ir_emit_load_ex(ctx, type, def_reg, op3, def);
+			if (def_reg_hi != IR_REG_RDX) {
+				|	mov Rd(def_reg_hi), edx
+			}
+			if (def_reg != IR_REG_RAX) {
+				|	mov Rd(def_reg), eax
+			}
 		}
-		|3:
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, def_reg);
+			ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+		}
+	} else {
+		ir_emit_store_i64_lo(ctx, def, IR_REG_RAX);
+		ir_emit_store_i64_hi(ctx, def, IR_REG_RDX);
 	}
+}
+
+static void ir_emit_zext_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg def_reg_hi;

+	IR_ASSERT(def_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+	if (ir_type_size[ctx->ir_base[insn->op1].type] < 4) {
+		ir_type src_type = ctx->ir_base[insn->op1].type;
+		ir_emit_zext_common(ctx, def, insn, IR_I32, src_type, ir_type_size[src_type], def_reg, 1);
+	} else if (op1_reg != def_reg) {
+		if (op1_reg != IR_REG_NONE && !IR_REG_SPILLED(op1_reg)) {
+			|	mov Rd(def_reg), Rd(op1_reg)
+		} else {
+			ir_emit_load(ctx, IR_I32, def_reg, insn->op1);
+		}
+	}
+	|	xor Rd(def_reg_hi), Rd(def_reg_hi)
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_cond_cmp_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_shift_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op2 = insn->op2;
-	ir_ref op3 = insn->op3;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg op2_reg = ctx->regs[def][2];
-	ir_reg op3_reg = ctx->regs[def][3];
-	ir_op op;
+	ir_ref op1_reg_hi, def_reg_hi;

-	if (op2 != op3) {
-		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, op2);
-		}
-		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, op3);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+	IR_ASSERT(def_reg != IR_REG_RCX && def_reg_hi != IR_REG_RCX);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, insn->op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, op2);
-		op3_reg = op2_reg;
-	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-		op3_reg = IR_REG_NUM(op3_reg);
-		ir_emit_load(ctx, type, op3_reg, op3);
-		op2_reg = op3_reg;
+	} else {
+		op1_reg_hi = IR_REG_NONE;
 	}
-
-	ir_emit_cmp_int_common2(ctx, def, insn->op1, &ctx->ir_base[insn->op1]);
-	op = ctx->ir_base[insn->op1].op;
-
-	if (IR_IS_TYPE_INT(type)) {
-		if (op3_reg != IR_REG_NONE) {
-			if (op3_reg == def_reg) {
-				IR_ASSERT(op2_reg != IR_REG_NONE);
-				op3_reg = op2_reg;
-				op ^= 1; // reverse
-			} else {
-				if (op2_reg != IR_REG_NONE) {
-					if (def_reg != op2_reg) {
-//						if (IR_IS_TYPE_INT(type)) {
-							ir_emit_mov(ctx, type, def_reg, op2_reg);
-//						} else {
-//							ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
-//						}
-					}
-				} else if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op)) {
-					/* prevent "xor" and flags clobbering */
-					ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op2].val.i64);
-				} else {
-					ir_emit_load_ex(ctx, type, def_reg, op2, def);
-				}
-			}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, insn->op2);
 		} else {
-			IR_ASSERT(op2_reg != IR_REG_NONE && op2_reg != def_reg);
-			if (IR_IS_CONST_REF(op3) && !IR_IS_SYM_CONST(ctx->ir_base[op3].op)) {
-				/* prevent "xor" and flags clobbering */
-				ir_emit_mov_imm_int(ctx, type, def_reg, ctx->ir_base[op3].val.i64);
-			} else {
-				ir_emit_load_ex(ctx, type, def_reg, op3, def);
-			}
-			op3_reg = op2_reg;
-			op ^= 1; // reverse
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-
-		switch (op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_EQ:
-				|	ASM_REG_REG_OP2 cmovne, type, def_reg, op3_reg
-				break;
-			case IR_NE:
-				|	ASM_REG_REG_OP2 cmove, type, def_reg, op3_reg
-				break;
-			case IR_LT:
-				|	ASM_REG_REG_OP2 cmovge, type, def_reg, op3_reg
-				break;
-			case IR_GE:
-				|	ASM_REG_REG_OP2 cmovl, type, def_reg, op3_reg
-				break;
-			case IR_LE:
-				|	ASM_REG_REG_OP2 cmovg, type, def_reg, op3_reg
-				break;
-			case IR_GT:
-				|	ASM_REG_REG_OP2 cmovle, type, def_reg, op3_reg
-				break;
-			case IR_ULT:
-				|	ASM_REG_REG_OP2 cmovae, type, def_reg, op3_reg
-				break;
-			case IR_UGE:
-				|	ASM_REG_REG_OP2 cmovb, type, def_reg, op3_reg
-				break;
-			case IR_ULE:
-				|	ASM_REG_REG_OP2 cmova, type, def_reg, op3_reg
-				break;
-			case IR_UGT:
-				|	ASM_REG_REG_OP2 cmovbe, type, def_reg, op3_reg
-				break;
+	}
+	if (op2_reg != IR_REG_RCX) {
+		if (op1_reg == IR_REG_RCX) {
+			|	mov Rd(def_reg), Rd(op1_reg)
+			op1_reg = def_reg;
 		}
-	} else {
-		switch (op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_EQ:
-				|	jne >2
-				break;
-			case IR_NE:
-				|	je >2
-				break;
-			case IR_LT:
-				|	jge >2
-				break;
-			case IR_GE:
-				|	jl >2
-				break;
-			case IR_LE:
-				|	jg >2
-				break;
-			case IR_GT:
-				|	jle >2
-				break;
-			case IR_ULT:
-				|	jae >2
-				break;
-			case IR_UGE:
-				|	jb >2
-				break;
-			case IR_ULE:
-				|	ja >2
-				break;
-			case IR_UGT:
-				|	jbe >2
-				break;
+		if (op1_reg_hi == IR_REG_RCX) {
+			|	mov Rd(def_reg_hi), Rd(op1_reg_hi)
+			op1_reg_hi = def_reg_hi;
 		}
-		|1:
-
 		if (op2_reg != IR_REG_NONE) {
-			if (def_reg != op2_reg) {
-				if (IR_IS_TYPE_INT(type)) {
-					ir_emit_mov(ctx, type, def_reg, op2_reg);
-				} else {
-					ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
-				}
-			}
-		} else {
-			ir_emit_load_ex(ctx, type, def_reg, op2, def);
-		}
-		|	jmp >3
-		|2:
-		if (op3_reg != IR_REG_NONE) {
-			if (def_reg != op3_reg) {
-				if (IR_IS_TYPE_INT(type)) {
-					ir_emit_mov(ctx, type, def_reg, op3_reg);
-				} else {
-					ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
-				}
-			}
+			|	mov ecx, Rd(op2_reg)
 		} else {
-			ir_emit_load_ex(ctx, type, def_reg, op3, def);
+			ir_emit_load_i64_lo(ctx, IR_REG_RCX, insn->op2);
 		}
-		|3:
 	}
-
+
+	ir_emit_load_i64(ctx, def, def_reg, def_reg_hi, op1_reg, op1_reg_hi, insn->op1);
+
+	switch (insn->op) {
+		default:
+			IR_ASSERT(0);
+		case IR_SHL:
+			|	shld Rd(def_reg_hi), Rd(def_reg), cl
+			|	shl	Rd(def_reg), cl
+			|	test cl, 32
+			|	je	>1
+			|	mov Rd(def_reg_hi), Rd(def_reg)
+			|	xor	Rd(def_reg), Rd(def_reg)
+			|1:
+			break;
+		case IR_SHR:
+			|	shrd Rd(def_reg), Rd(def_reg_hi), cl
+			|	shr	Rd(def_reg_hi), cl
+			|	test cl, 32
+			|	je	>1
+			|	mov Rd(def_reg), Rd(def_reg_hi)
+			|	xor	Rd(def_reg_hi), Rd(def_reg_hi)
+			|1:
+			break;
+		case IR_SAR:
+			|	shrd Rd(def_reg), Rd(def_reg_hi), cl
+			|	sar	Rd(def_reg_hi), cl
+			|	test cl, 32
+			|	je	>1
+			|	mov Rd(def_reg), Rd(def_reg_hi)
+			|	sar	Rd(def_reg_hi), 31
+			|1:
+			break;
+	}
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_cond_cmp_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_shift_const_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type = insn->type;
-	ir_ref op2 = insn->op2;
-	ir_ref op3 = insn->op3;
+	int32_t shift;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op2_reg = ctx->regs[def][2];
-	ir_reg op3_reg = ctx->regs[def][3];
-	ir_op op;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_ref op1_reg_hi, def_reg_hi;

-	if (op2 != op3) {
-		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, op2);
-		}
-		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, op3);
+	IR_ASSERT(!IR_IS_SYM_CONST(ctx->ir_base[insn->op2].op));
+	IR_ASSERT(IR_IS_SIGNED_32BIT(ctx->ir_base[insn->op2].val.i64));
+	shift = ctx->ir_base[insn->op2].val.i32;
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, insn->op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, op2);
-		op3_reg = op2_reg;
-	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
-		op3_reg = IR_REG_NUM(op3_reg);
-		ir_emit_load(ctx, type, op3_reg, op3);
-		op2_reg = op3_reg;
+	} else {
+		op1_reg_hi = IR_REG_NONE;
 	}

-	op = ir_emit_cmp_fp_common(ctx, def, insn->op1, &ctx->ir_base[insn->op1]);
+	ir_emit_load_i64(ctx, def, def_reg, def_reg_hi, op1_reg, op1_reg_hi, insn->op1);

-	switch (op) {
+	switch (insn->op) {
 		default:
-			IR_ASSERT(0 && "NIY binary op");
-		case IR_EQ:
-			|	jne >2
-			|	jp >2
-			break;
-		case IR_NE:
-			|	jp >1
-			|	je >2
-			break;
-		case IR_LT:
-			|	jp >2
-			|	jae >2
-			break;
-		case IR_GE:
-			|	jb >2
-			break;
-		case IR_LE:
-			|	jp >2
-			|	ja >2
-			break;
-		case IR_GT:
-			|	jbe >2
-			break;
-		case IR_ULT:
-			|	jae >2
-			break;
-		case IR_UGE:
-			|	jp >1
-			|	jb >2
-			break;
-		case IR_ULE:
-			|	ja >2
-			break;
-		case IR_UGT:
-			|	jp >1
-			|	jbe >2
+			IR_ASSERT(0);
+		case IR_SHL:
+			if ((shift & 63) != 32) {
+				|	shld Rd(def_reg_hi), Rd(def_reg), shift
+				|	shl	Rd(def_reg), shift
+			}
+			if (shift & 32) {
+				|	mov Rd(def_reg_hi), Rd(def_reg)
+				|	xor	Rd(def_reg), Rd(def_reg)
+			}
 			break;
-		case IR_ORDERED:
-			|	jp >2
+		case IR_SHR:
+			if ((shift & 63) != 32) {
+				|	shrd Rd(def_reg), Rd(def_reg_hi), shift
+				|	shr	Rd(def_reg_hi), shift
+			}
+			if (shift & 32) {
+				|	mov Rd(def_reg), Rd(def_reg_hi)
+				|	xor	Rd(def_reg_hi), Rd(def_reg_hi)
+			}
 			break;
-		case IR_UNORDERED:
-			|	jnp >2
+		case IR_SAR:
+			|	shrd Rd(def_reg), Rd(def_reg_hi), shift
+			|	sar	Rd(def_reg_hi), shift
+			if (shift & 32) {
+				|	mov Rd(def_reg), Rd(def_reg_hi)
+				|	sar	Rd(def_reg_hi), 31
+			}
 			break;
 	}
-	|1:
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+	}
+}

-	if (op2_reg != IR_REG_NONE) {
-		if (def_reg != op2_reg) {
-			if (IR_IS_TYPE_INT(type)) {
-				ir_emit_mov(ctx, type, def_reg, op2_reg);
+static void ir_emit_bitcast_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg def_reg_hi, op1_reg_hi;
+
+	if (insn->type == IR_DOUBLE || IR_IS_TYPE_VECTOR(insn->type)) {
+		IR_ASSERT(ctx->ir_base[insn->op1].type == IR_I64 || ctx->ir_base[insn->op1].type == IR_U64);
+
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
+				ir_emit_load_i64_lo(ctx, op1_reg, insn->op1);
+				ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
 			} else {
-				ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
+			}
+
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				|	vpinsrd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg_hi), 1
+			} else if (ctx->mflags & IR_X86_SSE41) {
+				|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+				|	pinsrd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg_hi), 1
+			} else {
+				int32_t offset = ctx->ret_slot;
+				ir_reg fp;
+
+				IR_ASSERT(offset != -1);
+				offset = IR_SPILL_POS_TO_OFFSET(offset);
+				fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				ir_emit_store_mem_int(ctx, IR_U32, IR_MEM_BO(fp, offset), op1_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, IR_MEM_BO(fp, offset + 4), op1_reg_hi);
+				ir_emit_load_mem_fp(ctx, IR_DOUBLE, def_reg, IR_MEM_BO(fp, offset));
+			}
+		} else if (IR_IS_CONST_REF(insn->op1)) {
+			int label = ir_get_const_label(ctx, insn->op1);
+
+			|	ASM_FP_REG_TXT_OP movs, IR_DOUBLE, def_reg, [=>label]
+		} else {
+			ir_mem mem;
+
+			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, insn->op1);
+			} else {
+				mem = ir_ref_spill_slot(ctx, insn->op1);
 			}
+			ir_emit_load_mem_fp(ctx, IR_DOUBLE, def_reg, mem);
+		}
+
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store(ctx, insn->type, def, def_reg);
 		}
 	} else {
-		ir_emit_load_ex(ctx, type, def_reg, op2, def);
-	}
-	|	jmp >3
-	|2:
-	if (op3_reg != IR_REG_NONE) {
-		if (def_reg != op3_reg) {
-			if (IR_IS_TYPE_INT(type)) {
-				ir_emit_mov(ctx, type, def_reg, op3_reg);
+		IR_ASSERT(insn->type == IR_I64 || insn->type == IR_U64);
+		IR_ASSERT(ctx->ir_base[insn->op1].type == IR_DOUBLE || IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op1].type));
+
+		IR_ASSERT(def_reg != IR_REG_NONE);
+		def_reg_hi = IR_REG_I64_HI(def_reg);
+		def_reg = IR_REG_I64_LO(def_reg);
+
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, IR_DOUBLE, op1_reg, insn->op1);
+			}
+
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vmovd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				|	vpextrd Rd(def_reg_hi), xmm(op1_reg-IR_REG_FP_FIRST), 1
+			} else if (ctx->mflags & IR_X86_SSE41) {
+				|	movd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				|	pextrd Rd(def_reg_hi), xmm(op1_reg-IR_REG_FP_FIRST), 1
 			} else {
-				ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
+				int32_t offset = ctx->ret_slot;
+				ir_reg fp;
+
+				IR_ASSERT(offset != -1);
+				offset = IR_SPILL_POS_TO_OFFSET(offset);
+				fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+				ir_emit_store_mem_fp(ctx, IR_DOUBLE, IR_MEM_BO(fp, offset), op1_reg);
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg, IR_MEM_BO(fp, offset));
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, IR_MEM_BO(fp, offset + 4));
+			}
+		} else if (IR_IS_CONST_REF(insn->op1)) {
+			ir_insn *val = &ctx->ir_base[insn->op1];
+
+			|	mov Rd(def_reg), val->val.u32
+			|	mov Rd(def_reg_hi), val->val.u32_hi
+		} else {
+			ir_mem mem, mem_hi;
+
+			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, insn->op1);
+			} else {
+				mem = ir_ref_spill_slot(ctx, insn->op1);
 			}
+			mem_hi = IR_MEM_I64_HI(mem);
+			|	ASM_REG_MEM_OP mov, IR_U32, def_reg, mem
+			|	ASM_REG_MEM_OP mov, IR_U32, def_reg_hi, mem_hi
 		}
-	} else {
-		ir_emit_load_ex(ctx, type, def_reg, op3, def);
-	}
-	|3:

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store_i64_lo(ctx, def, def_reg);
+			ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+		}
 	}
 }

-static void ir_emit_return_void(ir_ctx *ctx)
+static void ir_emit_int2fp_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_ref op1 = insn->op1;
+	ir_type type = ctx->ir_base[op1].type;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
+	int32_t offset = ctx->ret_slot;
+	ir_mem tmp_mem;

-	ir_emit_epilogue(ctx);
-
-	if (data->ra_data.cc->cleanup_stack_by_callee && ctx->param_stack_size) {
-		|	ret ctx->param_stack_size
-	} else {
-		|	ret
-	}
-}
+	IR_ASSERT(offset != -1);
+	offset = IR_SPILL_POS_TO_OFFSET(offset);
+	tmp_mem = IR_MEM_BO(fp, offset);

-static void ir_emit_return_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	ir_reg ret_reg = data->ra_data.cc->int_ret_reg;
-	ir_reg op2_reg = ctx->regs[ref][2];
+	if (op1_reg != IR_REG_NONE) {
+		ir_reg op1_reg_hi;

-	if (op2_reg != ret_reg) {
-		ir_type type = ctx->ir_base[insn->op2].type;
+		if (IR_REG_SPILLED(op1_reg)) {
+			ir_mem mem = ir_ref_spill_slot(ctx, op1);

-		if (op2_reg != IR_REG_NONE && !IR_REG_SPILLED(op2_reg)) {
-			ir_emit_mov(ctx, type, ret_reg, op2_reg);
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, insn->op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
+			|	ASM_TMEM_OP fild, qword, mem
 		} else {
-			ir_emit_load(ctx, type, ret_reg, insn->op2);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			|	ASM_TMEM_TXT_OP mov, dword, tmp_mem, Rd(op1_reg)
+			|	ASM_TMEM_TXT_OP mov, dword, IR_MEM_I64_HI(tmp_mem), Rd(op1_reg_hi)
+			|	ASM_TMEM_OP fild, qword, tmp_mem
 		}
-	}
-	ir_emit_return_void(ctx);
-}
+		if (type == IR_U64) {
+			|	test Rd(op1_reg_hi), Rd(op1_reg_hi)
+		}
+	} else if (IR_IS_CONST_REF(op1)) {
+		ir_insn *val = &ctx->ir_base[op1];

-static void ir_emit_return_fp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	ir_reg op2_reg = ctx->regs[ref][2];
-	ir_type type = ctx->ir_base[insn->op2].type;
-	ir_reg ret_reg = data->ra_data.cc->fp_ret_reg;
+		|	ASM_TMEM_TXT_OP mov, dword, tmp_mem, val->val.u32
+		|	ASM_TMEM_TXT_OP mov, dword, IR_MEM_I64_HI(tmp_mem), val->val.u32_hi
+		|	ASM_TMEM_OP fild, qword, tmp_mem
+		if (type == IR_U64) {
+			|	ASM_TMEM_TXT_OP cmp, dword, IR_MEM_I64_HI(tmp_mem), 0
+		}
+	} else {
+		ir_mem mem;

-	if (op2_reg != ret_reg && ret_reg != IR_REG_NONE) {
-		if (op2_reg != IR_REG_NONE && !IR_REG_SPILLED(op2_reg)) {
-			ir_emit_fp_mov(ctx, type, ret_reg, op2_reg);
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op1);
 		} else {
-			ir_emit_load(ctx, type, ret_reg, insn->op2);
+			mem = ir_ref_spill_slot(ctx, op1);
+		}
+		|	ASM_TMEM_OP fild, qword, mem
+		if (type == IR_U64) {
+			|	ASM_TMEM_TXT_OP cmp, dword, IR_MEM_I64_HI(mem), 0
 		}
 	}

-#ifdef IR_TARGET_X86
-	if (ret_reg == IR_REG_NONE) {
-		dasm_State **Dst = &data->dasm_state;
-
-		if (IR_IS_CONST_REF(insn->op2)) {
-			ir_insn *value = &ctx->ir_base[insn->op2];
-
-			if ((type == IR_FLOAT && value->val.f == 0.0) || (type == IR_DOUBLE && value->val.d == 0.0)) {
-				|	fldz
-			} else if ((type == IR_FLOAT && value->val.f == 1.0) || (type == IR_DOUBLE && value->val.d == 1.0)) {
-				|	fld1
-			} else {
-				int label = ir_get_const_label(ctx, insn->op2);
-
-				if (type == IR_DOUBLE) {
-					|	fld qword [=>label]
-				} else {
-					IR_ASSERT(type == IR_FLOAT);
-					|	fld dword [=>label]
-				}
-			}
-		} else if (op2_reg == IR_REG_NONE || IR_REG_SPILLED(op2_reg)) {
-			ir_reg fp;
-			int32_t offset = ir_ref_spill_slot_offset(ctx, insn->op2, &fp);
+	if (type == IR_U64) {
+		|	ASM_TMEM_TXT_OP mov, dword, IR_MEM_I64_HI(tmp_mem), 0
+		if (!data->ull2fp_const) {
+			data->ull2fp_const = 1;
+			ir_rodata(ctx);
+			|.align 4
+			|->ull2fp_const:
+			|.dword 0x5f800000
+			|.code
+		}
+		|	jns >1
+		|	fadd dword [->ull2fp_const]
+		|1:
+	}

-			if (type == IR_DOUBLE) {
-				|	fld qword [Ra(fp)+offset]
-			} else {
-				IR_ASSERT(type == IR_FLOAT);
-				|	fld dword [Ra(fp)+offset]
-			}
+	if (def_reg != IR_REG_NONE) {
+		if (insn->type == IR_DOUBLE) {
+			|	ASM_TMEM_OP fstp, qword, tmp_mem
 		} else {
-			int32_t offset = ctx->ret_slot;
-			ir_reg fp;
+			IR_ASSERT(insn->type == IR_FLOAT);
+			|	ASM_TMEM_OP fstp, dword, tmp_mem
+		}

-			IR_ASSERT(offset != -1);
-			offset = IR_SPILL_POS_TO_OFFSET(offset);
-			fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), op2_reg);
-			if (type == IR_DOUBLE) {
-				|	fld qword [Ra(fp)+offset]
-			} else {
-				IR_ASSERT(type == IR_FLOAT);
-				|	fld dword [Ra(fp)+offset]
-			}
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, tmp_mem);
+
+		if (IR_REG_SPILLED(ctx->regs[def][0])) {
+			ir_emit_store(ctx, insn->type, def, def_reg);
 		}
-	}
-#endif
+	} else {
+		ir_mem mem = ir_ref_spill_slot(ctx, def);

-	ir_emit_return_void(ctx);
+		if (insn->type == IR_DOUBLE) {
+			|	ASM_TMEM_OP fstp, qword, mem
+		} else {
+			IR_ASSERT(insn->type == IR_FLOAT);
+			|	ASM_TMEM_OP fstp, dword, mem
+		}
+	}
 }

-static void ir_emit_sext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_fp2int_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_ref op1 = insn->op1;
+	ir_type type = ctx->ir_base[op1].type;
 	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][2];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg fp, def_reg_hi;
+	int32_t offset;
+
+	offset = ctx->ret_slot;
+	IR_ASSERT(offset != -1);
+	offset = IR_SPILL_POS_TO_OFFSET(offset);
+	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;

-	IR_ASSERT(IR_IS_TYPE_INT(src_type));
-	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
-	IR_ASSERT(ir_type_size[dst_type] > ir_type_size[src_type]);
 	IR_ASSERT(def_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);

-	if (op1_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
-		}
-		if (ir_type_size[src_type] == 1) {
-			if (ir_type_size[dst_type] == 2) {
-				if (def_reg == IR_REG_RAX && op1_reg == IR_REG_RAX) {
-					|	cbw
-				} else {
-					|	movsx Rw(def_reg), Rb(op1_reg)
+	if (type == IR_DOUBLE) {
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, type, op1_reg, op1);
+			}
+			if (insn->type == IR_U64) {
+				if (!data->ull2d_const) {
+					data->ull2d_const = 1;
+					ir_rodata(ctx);
+					|.align 8
+					|->ull2d_const:
+					|.dword 0, 0x43e00000
+					|.code
 				}
-			} else if (ir_type_size[dst_type] == 4) {
-				|	movsx Rd(def_reg), Rb(op1_reg)
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	movsx Rq(def_reg), Rb(op1_reg)
-|.endif
+				|	ASM_FP_REG_TXT_OP ucomis, type, op1_reg, [->ull2d_const]
+				|	jnb >1
 			}
-		} else if (ir_type_size[src_type] == 2) {
-			if (ir_type_size[dst_type] == 4) {
-				if (def_reg == IR_REG_RAX && op1_reg == IR_REG_RAX) {
-					|	cwde
+
+			ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), op1_reg);
+			|	fld qword [Ra(fp)+offset]
+			|	fisttp qword [Ra(fp)+offset]
+			|	mov Rd(def_reg), dword [Ra(fp)+offset]
+			|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
+
+			if (insn->type == IR_U64) {
+				IR_ASSERT(tmp_reg != IR_REG_NONE);
+				|	jmp >2
+				|1:
+				if (ctx->mflags & IR_X86_AVX) {
+					|	ASM_AVX_REG_REG_TXT_OP vsubs, type, tmp_reg, op1_reg, [->ull2d_const]
+					ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), tmp_reg);
+					|	fld qword [Ra(fp)+offset]
+					|	fisttp qword [Ra(fp)+offset]
 				} else {
-					|	movsx Rd(def_reg), Rw(op1_reg)
+					if (tmp_reg != op1_reg) {
+						|	movsd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	ASM_SSE2_REG_TXT_OP subs, type, tmp_reg, [->ull2d_const]
+					ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), tmp_reg);
+					|	fld qword [Ra(fp)+offset]
+					|	fisttp qword [Ra(fp)+offset]
 				}
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	movsx Rq(def_reg), Rw(op1_reg)
-|.endif
-			}
+				|	mov Rd(def_reg), dword [Ra(fp)+offset]
+				|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
+				|	add Rd(def_reg_hi), 0x80000000
+				|2:
+			}
+		} else if (IR_IS_CONST_REF(op1)) {
+			int label = ir_get_const_label(ctx, op1);
+			|	fld qword [=>label]
+			|	fisttp qword [Ra(fp)+offset]
+			|	mov Rd(def_reg), dword [Ra(fp)+offset]
+			|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
 		} else {
-			IR_ASSERT(ir_type_size[src_type] == 4);
-			IR_ASSERT(ir_type_size[dst_type] == 8);
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			if (def_reg == IR_REG_RAX && op1_reg == IR_REG_RAX) {
-				|	cdqe
+			ir_mem mem;
+
+			if (ir_rule(ctx, op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, op1);
 			} else {
-				|	movsxd Rq(def_reg), Rd(op1_reg)
+				mem = ir_ref_spill_slot(ctx, op1);
 			}
-|.endif
-		}
-	} else if (IR_IS_CONST_REF(insn->op1)) {
-		int64_t val;
-
-		if (ir_type_size[src_type] == 1) {
-			val = ctx->ir_base[insn->op1].val.i8;
-		} else if (ir_type_size[src_type] == 2) {
-			val = ctx->ir_base[insn->op1].val.i16;
-		} else if (ir_type_size[src_type] == 4) {
-			val = ctx->ir_base[insn->op1].val.i32;
-		} else {
-			IR_ASSERT(ir_type_size[src_type] == 8);
-			val = ctx->ir_base[insn->op1].val.i64;
+			|	ASM_TMEM_OP fld, qword, mem
+			|	fisttp qword [Ra(fp)+offset]
+			|	mov Rd(def_reg), dword [Ra(fp)+offset]
+			|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
 		}
-		ir_emit_mov_imm_int(ctx, dst_type, def_reg, val);
 	} else {
-		ir_mem mem;
+		IR_ASSERT(type == IR_FLOAT);
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				ir_emit_load(ctx, type, op1_reg, op1);
+			}
+			if (insn->type == IR_U64) {
+				if (!data->ull2f_const) {
+					data->ull2f_const = 1;
+					ir_rodata(ctx);
+					|.align 4
+					|->ull2f_const:
+					|.dword 0x5f000000
+					|.code
+				}
+				IR_ASSERT(type == IR_FLOAT);
+				|	ASM_FP_REG_TXT_OP ucomis, type, op1_reg, [->ull2f_const]
+				|	jnb >1
+			}

-		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, insn->op1);
+			ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), op1_reg);
+			|	fld dword [Ra(fp)+offset]
+			|	fisttp qword [Ra(fp)+offset]
+			|	mov Rd(def_reg), dword [Ra(fp)+offset]
+			|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
+
+			if (insn->type == IR_U64) {
+				IR_ASSERT(tmp_reg != IR_REG_NONE);
+				|	jmp >2
+				|1:
+				if (ctx->mflags & IR_X86_AVX) {
+					|	ASM_AVX_REG_REG_TXT_OP vsubs, type, tmp_reg, op1_reg, [->ull2f_const]
+					ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), tmp_reg);
+					|	fld dword [Ra(fp)+offset]
+					|	fisttp qword [Ra(fp)+offset]
+				} else {
+					if (tmp_reg != op1_reg) {
+						|	movss xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	ASM_SSE2_REG_TXT_OP subs, type, tmp_reg, [->ull2f_const]
+					ir_emit_store_mem_fp(ctx, type, IR_MEM_BO(fp, offset), tmp_reg);
+					|	fld dword [Ra(fp)+offset]
+					|	fisttp qword [Ra(fp)+offset]
+				}
+				|	mov Rd(def_reg), dword [Ra(fp)+offset]
+				|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
+				|	add Rd(def_reg_hi), 0x80000000
+				|2:
+			}
+		} else if (IR_IS_CONST_REF(op1)) {
+			int label = ir_get_const_label(ctx, op1);
+			|	fld dword [=>label]
+			|	fisttp qword [Ra(fp)+offset]
+			|	mov Rd(def_reg), dword [Ra(fp)+offset]
+			|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
 		} else {
-			mem = ir_ref_spill_slot(ctx, insn->op1);
-		}
+			ir_mem mem;

-		if (ir_type_size[src_type] == 1) {
-			if (ir_type_size[dst_type] == 2) {
-				|	ASM_TXT_TMEM_OP movsx, Rw(def_reg), byte, mem
-			} else if (ir_type_size[dst_type] == 4) {
-				|	ASM_TXT_TMEM_OP movsx, Rd(def_reg), byte, mem
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	ASM_TXT_TMEM_OP movsx, Rq(def_reg), byte, mem
-|.endif
-			}
-		} else if (ir_type_size[src_type] == 2) {
-			if (ir_type_size[dst_type] == 4) {
-				|	ASM_TXT_TMEM_OP movsx, Rd(def_reg), word, mem
+			if (ir_rule(ctx, op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, op1);
 			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	ASM_TXT_TMEM_OP movsx, Rq(def_reg), word, mem
-|.endif
+				mem = ir_ref_spill_slot(ctx, op1);
 			}
-		} else {
-			IR_ASSERT(ir_type_size[src_type] == 4);
-			IR_ASSERT(ir_type_size[dst_type] == 8);
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			|	ASM_TXT_TMEM_OP movsxd, Rq(def_reg), dword, mem
-|.endif
+			|	ASM_TMEM_OP fld, dword, mem
+			|	fisttp qword [Ra(fp)+offset]
+			|	mov Rd(def_reg), dword [Ra(fp)+offset]
+			|	mov Rd(def_reg_hi), dword [Ra(fp)+offset+4]
 		}
 	}
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_zext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_bit_count_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg tmp_reg = ctx->regs[def][2];
+	ir_reg op1_reg_hi;

-	IR_ASSERT(IR_IS_TYPE_INT(src_type));
-	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
-	IR_ASSERT(ir_type_size[dst_type] > ir_type_size[src_type]);
-	IR_ASSERT(def_reg != IR_REG_NONE);
-
+	IR_ASSERT(def_reg != IR_REG_NONE && tmp_reg != IR_REG_NONE);
 	if (op1_reg != IR_REG_NONE) {
 		if (IR_REG_SPILLED(op1_reg)) {
 			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
-		}
-		if (ir_type_size[src_type] == 1) {
-			if (ir_type_size[dst_type] == 2) {
-				|	movzx Rw(def_reg), Rb(op1_reg)
-			} else if (ir_type_size[dst_type] == 4) {
-				|	movzx Rd(def_reg), Rb(op1_reg)
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	movzx Rq(def_reg), Rb(op1_reg)
-|.endif
-			}
-		} else if (ir_type_size[src_type] == 2) {
-			if (ir_type_size[dst_type] == 4) {
-				|	movzx Rd(def_reg), Rw(op1_reg)
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	movzx Rq(def_reg), Rw(op1_reg)
-|.endif
-			}
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, insn->op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
 		} else {
-			IR_ASSERT(ir_type_size[src_type] == 4);
-			IR_ASSERT(ir_type_size[dst_type] == 8);
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			/* Avoid zero extension to the same register. This may be not always safe ??? */
-			if (op1_reg != def_reg) {
-				|	mov Rd(def_reg), Rd(op1_reg)
-			}
-|.endif
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-	} else if (IR_IS_CONST_REF(insn->op1)) {
-		uint64_t val;

-		if (ir_type_size[src_type] == 1)  {
-			val = ctx->ir_base[insn->op1].val.u8;
-		} else if (ir_type_size[src_type] == 2) {
-			val = ctx->ir_base[insn->op1].val.u16;
-		} else if (ir_type_size[src_type] == 4) {
-			val = ctx->ir_base[insn->op1].val.u32;
-		} else {
-			IR_ASSERT(ir_type_size[src_type] == 8);
-			val = ctx->ir_base[insn->op1].val.u64;
+		switch (insn->op) {
+			case IR_CTPOP:
+				|	popcnt Rd(def_reg), Rd(op1_reg)
+				|	popcnt Rd(tmp_reg), Rd(op1_reg_hi)
+				|	add Rd(def_reg), Rd(tmp_reg)
+				break;
+			case IR_CTLZ:
+				|   bsr Rd(tmp_reg), Rd(op1_reg)
+				|	or Rd(tmp_reg), 32
+				|	bsr Rd(def_reg), Rd(op1_reg_hi)
+				|	cmovz Rd(def_reg), Rd(tmp_reg)
+				|	xor Rd(def_reg), 31
+				break;
+			case IR_CTTZ:
+				|	bsf Rd(tmp_reg), Rd(op1_reg_hi)
+				|	add Rd(tmp_reg), 32
+				|	bsf Rd(def_reg), Rd(op1_reg)
+				|	cmovz Rd(def_reg), Rd(tmp_reg)
+				break;
+			default:
+				IR_ASSERT(0);
 		}
-		ir_emit_mov_imm_int(ctx, dst_type, def_reg, val);
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		IR_ASSERT(0);
 	} else {
-		ir_mem mem;
+		ir_mem mem, mem_hi;

 		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
 			mem = ir_fuse_load(ctx, def, insn->op1);
 		} else {
 			mem = ir_ref_spill_slot(ctx, insn->op1);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);

-		if (ir_type_size[src_type] == 1) {
-			if (ir_type_size[dst_type] == 2) {
-				|	ASM_TXT_TMEM_OP movzx, Rw(def_reg), byte, mem
-			} else if (ir_type_size[dst_type] == 4) {
-				|	ASM_TXT_TMEM_OP movzx, Rd(def_reg), byte, mem
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	ASM_TXT_TMEM_OP movzx, Rq(def_reg), byte, mem
-|.endif
-			}
-		} else if (ir_type_size[src_type] == 2) {
-			if (ir_type_size[dst_type] == 4) {
-				|	ASM_TXT_TMEM_OP movzx, Rd(def_reg), word, mem
-			} else {
-				IR_ASSERT(ir_type_size[dst_type] == 8);
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	ASM_TXT_TMEM_OP movzx, Rq(def_reg), word, mem
-|.endif
-			}
-		} else {
-			IR_ASSERT(ir_type_size[src_type] == 4);
-			IR_ASSERT(ir_type_size[dst_type] == 8);
-|.if X64
-			|	ASM_TXT_TMEM_OP mov, Rd(def_reg), dword, mem
-|.endif
+		switch (insn->op) {
+			case IR_CTPOP:
+				|	ASM_TXT_TMEM_OP popcnt, Rd(def_reg), dword, mem
+				|	ASM_TXT_TMEM_OP popcnt, Rd(tmp_reg), dword, mem_hi
+				|	add Rd(def_reg), Rd(tmp_reg)
+				break;
+			case IR_CTLZ:
+				|   ASM_TXT_TMEM_OP bsr, Rd(tmp_reg), dword, mem
+				|	or Rd(tmp_reg), 32
+				|	ASM_TXT_TMEM_OP bsr, Rd(def_reg), dword, mem_hi
+				|	cmovz Rd(def_reg), Rd(tmp_reg)
+				|	xor Rd(def_reg), 31
+				break;
+			case IR_CTTZ:
+				|	ASM_TXT_TMEM_OP bsf, Rd(tmp_reg), dword, mem_hi
+				|	add Rd(tmp_reg), 32
+				|	ASM_TXT_TMEM_OP bsf, Rd(def_reg), dword, mem
+				|	cmovz Rd(def_reg), Rd(tmp_reg)
+				break;
+			default:
+				IR_ASSERT(0);
 		}
 	}
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_trunc(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_min_max_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[insn->op1].type;
+	ir_op op;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	ir_reg op1_reg_hi, op2_reg_hi, def_reg_hi;
+
+	IR_ASSERT(def_reg != IR_REG_NONE && tmp_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);

-	IR_ASSERT(IR_IS_TYPE_INT(src_type));
-	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
-	IR_ASSERT(ir_type_size[dst_type] < ir_type_size[src_type]);
-	IR_ASSERT(def_reg != IR_REG_NONE);
 	if (op1_reg != IR_REG_NONE) {
 		if (IR_REG_SPILLED(op1_reg)) {
 			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
 		}
-		if (op1_reg != def_reg) {
-#ifdef IR_TARGET_X86
-			if (ir_type_size[dst_type] == 1
-			 && (op1_reg == IR_REG_RBP || op1_reg == IR_REG_RSI || op1_reg == IR_REG_RDI)) {
-				ir_backend_data *data = ctx->data;
-				dasm_State **Dst = &data->dasm_state;
-
-				ir_emit_mov(ctx, src_type, def_reg, op1_reg);
-				|	and	Rb(def_reg), 0xff
-			} else {
-				ir_emit_mov(ctx, dst_type, def_reg, op1_reg);
+	} else {
+		op1_reg_hi = IR_REG_NONE;
+	}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load_i64_lo(ctx, op2_reg, op2);
+				ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
 			}
-#else
-			ir_emit_mov(ctx, dst_type, def_reg, op1_reg);
-#endif
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
 	} else {
-		ir_emit_load_ex(ctx, dst_type, def_reg, insn->op1, def);
+		op2_reg_hi = IR_REG_NONE;
+	}
+
+	if (insn->op == IR_MIN) {
+		op = (type == IR_I64) ? IR_GT : IR_UGT;
+	} else {
+		op = (type == IR_I64) ? IR_LT : IR_ULT;
+	}
+
+	ir_emit_cmp_i64_common(ctx, op, def, tmp_reg, IR_REG_NONE, op1_reg, op1_reg_hi, op1, op2_reg, op2_reg_hi, op2);
+
+	if (type == IR_I64) {
+		|	jl >1
+	} else {
+		|	jc >1
 	}
+
+	ir_emit_load_i64(ctx, def, def_reg, def_reg_hi, op1_reg, op1_reg_hi, op1);
+
+	|	jmp >2
+	|1:
+
+	ir_emit_load_i64(ctx, def, def_reg, def_reg_hi, op2_reg, op2_reg_hi, op2);
+
+	|2:
+
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_bitcast(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_cond_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_type op1_type = ctx->ir_base[op1].type;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
 	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_ref op2_reg_hi, op3_reg_hi, def_reg_hi;

-	IR_ASSERT(ir_type_size[dst_type] == ir_type_size[src_type]);
 	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (IR_IS_TYPE_INT(src_type) && IR_IS_TYPE_INT(dst_type)) {
-		if (op1_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op1_reg)) {
-				op1_reg = IR_REG_NUM(op1_reg);
-				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
-			}
-			if (op1_reg != def_reg) {
-				ir_emit_mov(ctx, dst_type, def_reg, op1_reg);
-			}
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
 		} else {
-			ir_emit_load_ex(ctx, dst_type, def_reg, insn->op1, def);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-	} else if (IR_IS_TYPE_FP(src_type) && IR_IS_TYPE_FP(dst_type)) {
-		if (op1_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op1_reg)) {
-				op1_reg = IR_REG_NUM(op1_reg);
-				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
-			}
-			if (op1_reg != def_reg) {
-				ir_emit_fp_mov(ctx, dst_type, def_reg, op1_reg);
+	} else {
+		op2_reg_hi = IR_REG_NONE;
+	}
+
+	if (op3_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			op3_reg = IR_REG_I64_LO(op3_reg);
+			if (op2 != op3) {
+				ir_emit_load_i64_lo(ctx, op3_reg, op3);
+				ir_emit_load_i64_hi(ctx, op3_reg_hi, op3);
 			}
 		} else {
-			ir_emit_load_ex(ctx, dst_type, def_reg, insn->op1, def);
+			op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			op3_reg = IR_REG_I64_LO(op3_reg);
 		}
-	} else if (IR_IS_TYPE_FP(src_type)) {
-		IR_ASSERT(IR_IS_TYPE_INT(dst_type));
+	} else {
+		op3_reg_hi = IR_REG_NONE;
+	}
+
+	if (op1_type == IR_I64 || op1_type == IR_U64) {
+		ir_reg op1_reg_hi, tmp_reg;
+
+		IR_ASSERT(ctx->tmp_regs && ctx->tmp_regs[def] != IR_REG_NONE);
+		tmp_reg = ctx->tmp_regs[def];
+
 		if (op1_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op1_reg)) {
 				op1_reg = IR_REG_NUM(op1_reg);
-				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
-			}
-			if (src_type == IR_DOUBLE) {
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vmovd Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				} else {
-					|	movd Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
+				if (op1 != op2 && op1 != op3) {
+					ir_emit_load_i64_lo(ctx, op1_reg, op1);
+					ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
 				}
-|.endif
 			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vmovd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				} else {
-					|	movd Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				}
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
 			}
-		} else if (IR_IS_CONST_REF(insn->op1)) {
-			ir_insn *_insn = &ctx->ir_base[insn->op1];
-			IR_ASSERT(!IR_IS_SYM_CONST(_insn->op));
-			if (src_type == IR_DOUBLE) {
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	mov64 Rq(def_reg), _insn->val.i64
-|.endif
+		}
+		if (op1_reg != IR_REG_NONE) {
+			|	mov Rd(tmp_reg), Rd(op1_reg)
+			|	or Rd(tmp_reg), Rd(op1_reg_hi)
+		} else {
+			ir_mem mem, mem_hi;
+
+			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, insn->op1);
 			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				|	mov Rd(def_reg), _insn->val.i32
+				mem = ir_ref_spill_slot(ctx, insn->op1);
 			}
+			mem_hi = IR_MEM_I64_HI(mem);
+
+			|	ASM_REG_MEM_OP mov, IR_U32, tmp_reg, mem
+			|	ASM_REG_MEM_OP or, IR_U32, tmp_reg, mem_hi
+		}
+		|	je >2
+	} else if (IR_IS_TYPE_INT(op1_type)) {
+		if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, op1_type, op1_reg, op1);
+		}
+		if (op1_reg != IR_REG_NONE) {
+			|	ASM_REG_REG_OP test, op1_type, op1_reg, op1_reg
 		} else {
 			ir_mem mem;

@@ -7811,3145 +16494,9005 @@ static void ir_emit_bitcast(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 				mem = ir_ref_spill_slot(ctx, insn->op1);
 			}

-			if (src_type == IR_DOUBLE) {
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	ASM_TXT_TMEM_OP mov, Rq(def_reg), qword, mem
-|.endif
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				|	ASM_TXT_TMEM_OP mov, Rd(def_reg), dword, mem
-			}
+			|	ASM_MEM_IMM_OP cmp, op1_type, mem, 0
 		}
-	} else if (IR_IS_TYPE_FP(dst_type)) {
-		IR_ASSERT(IR_IS_TYPE_INT(src_type));
+		|	je >2
+	} else {
+		if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, op1_type, op1_reg, op1);
+		}
+		if (!data->double_zero_const) {
+			data->double_zero_const = 1;
+			ir_rodata(ctx);
+			|.align 16
+			|->double_zero_const:
+			|.dword 0, 0
+			|.code
+		}
+		|	ASM_FP_REG_TXT_OP ucomis, op1_type, op1_reg, [->double_zero_const]
+		|	jp >1
+		|	je >2
+		|1:
+	}
+
+	if (op2_reg != IR_REG_NONE) {
+		if (def_reg != op2_reg) {
+			ir_emit_mov(ctx, IR_U32, def_reg, op2_reg);
+		}
+		if (def_reg_hi != op2_reg_hi) {
+			ir_emit_mov(ctx, IR_U32, def_reg_hi, op2_reg_hi);
+		}
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];
+
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg, val->val.u32);
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg_hi, val->val.u32_hi);
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op2);
+		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg_hi, mem_hi);
+	}
+
+	|	jmp >3
+	|2:
+
+	if (op3_reg != IR_REG_NONE) {
+		if (def_reg != op3_reg) {
+			ir_emit_mov(ctx, IR_U32, def_reg, op3_reg);
+		}
+		if (def_reg_hi != op3_reg_hi) {
+			ir_emit_mov(ctx, IR_U32, def_reg_hi, op3_reg_hi);
+		}
+	} else if (IR_IS_CONST_REF(op3)) {
+		ir_insn *val = &ctx->ir_base[op3];
+
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg, val->val.u32);
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg_hi, val->val.u32_hi);
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op3) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op3);
+		} else {
+			mem = ir_ref_spill_slot(ctx, op3);
+		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg_hi, mem_hi);
+	}
+
+	|3:
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+	}
+}
+
+static void ir_emit_cond_cmp_i64(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op1];
+	ir_type type = insn->type;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+
+	if (op2 != op3) {
+		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, op2);
+		}
+		if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			ir_emit_load(ctx, type, op3_reg, op3);
+		}
+	} else if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
+		op2_reg = IR_REG_NUM(op2_reg);
+		ir_emit_load(ctx, type, op2_reg, op2);
+		op3_reg = op2_reg;
+	} else if (op3_reg != IR_REG_NONE && IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+		ir_emit_load(ctx, type, op3_reg, op3);
+		op2_reg = op3_reg;
+	}
+
+	do {
+		ir_reg tmp_reg = ctx->regs[def][1];
+		ir_reg tmp2_reg = ctx->regs[insn->op1][3];
+		ir_ref op1 = cmp_insn->op1;
+		ir_ref op2 = cmp_insn->op2;
+		ir_reg op1_reg, op2_reg;
+		ir_reg op1_reg_hi, op2_reg_hi;
+
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+
+		if (!IR_IS_CONST_REF(insn->op2) && UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
+			op1_reg = ir_get_fused_reg(ctx, def, insn->op1 * sizeof(ir_ref) + 1);
+			op2_reg = ir_get_fused_reg(ctx, def, insn->op1 * sizeof(ir_ref) + 2);
+		} else {
+			op1_reg = ctx->regs[insn->op1][1];
+			op2_reg = ctx->regs[insn->op1][2];
+		}
+
 		if (op1_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op1_reg)) {
 				op1_reg = IR_REG_NUM(op1_reg);
-				ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
+				ir_emit_load_i64_lo(ctx, op1_reg, op1);
+				ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+			} else {
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
 			}
-			if (dst_type == IR_DOUBLE) {
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
-				} else {
-					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+		} else {
+			op1_reg_hi = IR_REG_NONE;
+		}
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				op2_reg_hi = IR_REG_I64_HI(op2_reg);
+				op2_reg = IR_REG_I64_LO(op2_reg);
+				if (op1 != op2) {
+					ir_emit_load_i64_lo(ctx, op2_reg, op2);
+					ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
 				}
-|.endif
 			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
-				} else {
-					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
-				}
+				op2_reg_hi = IR_REG_I64_HI(op2_reg);
+				op2_reg = IR_REG_I64_LO(op2_reg);
 			}
-		} else if (IR_IS_CONST_REF(insn->op1)) {
-			int label = ir_get_const_label(ctx, insn->op1);
-
-			|	ASM_FP_REG_TXT_OP movs, dst_type, def_reg, [=>label]
 		} else {
-			ir_mem mem;
+			op2_reg_hi = IR_REG_NONE;
+		}

-			if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-				mem = ir_fuse_load(ctx, def, insn->op1);
+		ir_emit_cmp_i64_common(ctx, cmp_insn->op, def, tmp_reg, tmp2_reg, op1_reg, op1_reg_hi, op1, op2_reg, op2_reg_hi, op2);
+	} while (0);
+
+	switch (cmp_insn->op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	je >2
+			break;
+		case IR_NE:
+			|	jne >2
+			break;
+		case IR_LT:
+		case IR_GT:
+			|	jl >2
+			break;
+		case IR_GE:
+		case IR_LE:
+			|	jge >2
+			break;
+		case IR_ULT:
+		case IR_UGT:
+			|	jc >2
+			break;
+		case IR_UGE:
+		case IR_ULE:
+			|	jnc >2
+			break;
+	}
+
+	if (op3_reg != IR_REG_NONE) {
+		if (def_reg != op3_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, op3_reg);
 			} else {
-				mem = ir_ref_spill_slot(ctx, insn->op1);
+				ir_emit_fp_mov(ctx, type, def_reg, op3_reg);
 			}
+		}
+	} else {
+		ir_emit_load_ex(ctx, type, def_reg, op3, def);
+	}
+
+	|	jmp >3
+	|2:

-			|	ASM_FP_REG_MEM_OP movs, dst_type, def_reg, mem
+	if (op2_reg != IR_REG_NONE) {
+		if (def_reg != op2_reg) {
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, op2_reg);
+			} else {
+				ir_emit_fp_mov(ctx, type, def_reg, op2_reg);
+			}
 		}
+	} else {
+		ir_emit_load_ex(ctx, type, def_reg, op2, def);
 	}
+
+	|3:
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+		ir_emit_store(ctx, type, def, def_reg);
 	}
 }

-static void ir_emit_int2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_cond_i64_cmp_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_ref op2_reg_hi, op3_reg_hi, def_reg_hi;

-	IR_ASSERT(IR_IS_TYPE_INT(src_type));
-	IR_ASSERT(IR_IS_TYPE_FP(dst_type));
 	IR_ASSERT(def_reg != IR_REG_NONE);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+		}
+	} else {
+		op2_reg_hi = IR_REG_NONE;
 	}

-	if (IR_IS_TYPE_UNSIGNED(src_type) && ir_type_size[src_type] >= sizeof(void*)) {
-		ir_reg tmp_reg = ctx->regs[def][2];
-
-		IR_ASSERT(tmp_reg != IR_REG_NONE);
-		if (op1_reg == IR_REG_NONE) {
-			if (IR_IS_CONST_REF(insn->op1)) {
-				IR_ASSERT(0);
-			} else {
-				ir_mem mem;
-
-				if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-					mem = ir_fuse_load(ctx, def, insn->op1);
-				} else {
-					mem = ir_ref_spill_slot(ctx, insn->op1);
-				}
-				ir_emit_load_mem_int(ctx, src_type, tmp_reg, mem);
-				op1_reg = tmp_reg;
-			}
-		}
-		if (sizeof(void*) == 4) {
-			if (tmp_reg == op1_reg) {
-				| add Rd(op1_reg), 0x80000000
-			} else {
-				| lea Rd(tmp_reg), dword [Rd(op1_reg)+0x80000000]
-				op1_reg = tmp_reg;
+	if (op3_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			op3_reg = IR_REG_I64_LO(op3_reg);
+			if (op2 != op3) {
+				ir_emit_load_i64_lo(ctx, op3_reg, op3);
+				ir_emit_load_i64_hi(ctx, op3_reg_hi, op3);
 			}
 		} else {
-|.if X64
-			|	test Rq(op1_reg), Rq(op1_reg)
-			|	js >1
-			|.cold_code
-			|1:
-			if (tmp_reg != op1_reg) {
-				| mov Rq(tmp_reg), Rq(op1_reg)
-			}
-			|	shr	Rq(tmp_reg), 1
-			|	adc Rq(tmp_reg), 0
-			if (dst_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
-					|	vaddsd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
-					|	addsd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-				}
-			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
-					|	vaddss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp_reg)
-					|	addss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-				}
-			}
-			|	jmp >2
-			|.code
-|.endif
+			op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			op3_reg = IR_REG_I64_LO(op3_reg);
 		}
+	} else {
+		op3_reg_hi = IR_REG_NONE;
 	}

-	if (op1_reg != IR_REG_NONE) {
-		bool src64 = 0;
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op1];

-		if (IR_IS_TYPE_SIGNED(src_type)) {
-			if (ir_type_size[src_type] < 4) {
-|.if X64
-||				if (ir_type_size[src_type] == 1) {
-					| movsx Rq(op1_reg), Rb(op1_reg)
-||				} else {
-					| movsx Rq(op1_reg), Rw(op1_reg)
-||				}
-||				src64 = 1;
-|.else
-||				if (ir_type_size[src_type] == 1) {
-					| movsx Rd(op1_reg), Rb(op1_reg)
-||				} else if (op1_reg == IR_REG_RAX) {
-					| cwde
-||				} else {
-					| movsx Rd(op1_reg), Rw(op1_reg)
-||				}
-|.endif
-			} else if (ir_type_size[src_type] > 4) {
-				src64 = 1;
-			}
+	if (ctx->ir_base[cmp_insn->op1].type == IR_I64 || ctx->ir_base[cmp_insn->op1].type == IR_U64) {
+		ir_reg tmp_reg = ctx->regs[def][1];
+		ir_reg tmp2_reg = ctx->regs[insn->op1][3];
+		ir_ref op1 = cmp_insn->op1;
+		ir_ref op2 = cmp_insn->op2;
+		ir_reg op1_reg, op2_reg;
+		ir_reg op1_reg_hi, op2_reg_hi;
+
+		IR_ASSERT(tmp_reg != IR_REG_NONE);
+
+		if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
+			op1_reg = ir_get_fused_reg(ctx, def, insn->op1 * sizeof(ir_ref) + 1);
+			op2_reg = ir_get_fused_reg(ctx, def, insn->op1 * sizeof(ir_ref) + 2);
 		} else {
-			if (ir_type_size[src_type] < 8) {
-|.if X64
-||				if (ir_type_size[src_type] == 1) {
-					| movzx Rq(op1_reg), Rb(op1_reg)
-||				} else if (ir_type_size[src_type] == 2) {
-					| movzx Rq(op1_reg), Rw(op1_reg)
-||				}
-||				src64 = 1;
-|.else
-||				if (ir_type_size[src_type] == 1) {
-					| movzx Rd(op1_reg), Rb(op1_reg)
-||				} else if (ir_type_size[src_type] == 2) {
-					| movzx Rd(op1_reg), Rw(op1_reg)
-||				}
-|.endif
-			} else {
-				src64 = 1;
-			}
+			op1_reg = ctx->regs[insn->op1][1];
+			op2_reg = ctx->regs[insn->op1][2];
 		}
-		if (!src64) {
-			if (dst_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
-				}
+
+		if (op1_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op1_reg)) {
+				op1_reg = IR_REG_NUM(op1_reg);
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
+				ir_emit_load_i64_lo(ctx, op1_reg, op1);
+				ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
 			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
-				}
+				op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				op1_reg = IR_REG_I64_LO(op1_reg);
 			}
 		} else {
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			if (dst_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
-				}
-			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
-				}
-			}
-|.endif
+			op1_reg_hi = IR_REG_NONE;
 		}
-		|2:
-		if (sizeof(void*) == 4 && IR_IS_TYPE_UNSIGNED(src_type) && ir_type_size[src_type] >= sizeof(void*)) {
-			if (dst_type == IR_DOUBLE) {
-				uint32_t c = (sizeof(void*) == 4) ? 0x41e00000 : 0x43e00000;
-				if (!data->u2d_const) {
-					data->u2d_const = 1;
-					ir_rodata(ctx);
-					|.align 8
-					|->u2d_const:
-					|.dword 0, c
-					|.code
-				}
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vaddsd	xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword [->u2d_const]
-				} else {
-					|	addsd	xmm(def_reg-IR_REG_FP_FIRST), qword [->u2d_const]
-				}
-			} else {
-				uint32_t c = (sizeof(void*) == 4) ? 0x4f000000 : 0x5f000000;
-				if (!data->u2f_const) {
-					data->u2f_const = 1;
-					ir_rodata(ctx);
-					|.align 4
-					|->u2f_const:
-					|.dword c
-					|.code
-				}
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vaddss	xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword [->u2f_const]
-				} else {
-					|	addss	xmm(def_reg-IR_REG_FP_FIRST), dword [->u2f_const]
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				op2_reg_hi = IR_REG_I64_HI(op2_reg);
+				op2_reg = IR_REG_I64_LO(op2_reg);
+				if (op1 != op2) {
+					ir_emit_load_i64_lo(ctx, op2_reg, op2);
+					ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
 				}
+			} else {
+				op2_reg_hi = IR_REG_I64_HI(op2_reg);
+				op2_reg = IR_REG_I64_LO(op2_reg);
 			}
+		} else {
+			op2_reg_hi = IR_REG_NONE;
+		}
+
+		ir_emit_cmp_i64_common(ctx, cmp_insn->op, def, tmp_reg, tmp2_reg, op1_reg, op1_reg_hi, op1, op2_reg, op2_reg_hi, op2);
+
+		switch (cmp_insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_EQ:
+				|	jne >2
+				break;
+			case IR_NE:
+				|	je >2
+				break;
+			case IR_LT:
+			case IR_GT:
+				|	jge >2
+				break;
+			case IR_GE:
+			case IR_LE:
+				|	jl >2
+				break;
+			case IR_ULT:
+			case IR_UGT:
+				|	jnc >2
+				break;
+			case IR_UGE:
+			case IR_ULE:
+				|	jc >2
+				break;
 		}
-	} else if (IR_IS_CONST_REF(insn->op1)) {
-		IR_ASSERT(0);
 	} else {
-		ir_mem mem;
-		bool src64 = ir_type_size[src_type] == 8;
+		ir_emit_cmp_int_common2(ctx, def, insn->op1, cmp_insn);

-		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, insn->op1);
+		switch (cmp_insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_EQ:
+				|	jne >2
+				break;
+			case IR_NE:
+				|	je >2
+				break;
+			case IR_LT:
+				|	jge >2
+				break;
+			case IR_GE:
+				|	jl >2
+				break;
+			case IR_LE:
+				|	jg >2
+				break;
+			case IR_GT:
+				|	jle >2
+				break;
+			case IR_ULT:
+				|	jae >2
+				break;
+			case IR_UGE:
+				|	jb >2
+				break;
+			case IR_ULE:
+				|	ja >2
+				break;
+			case IR_UGT:
+				|	jbe >2
+				break;
+		}
+	}
+
+	|1:
+
+	if (op2_reg != IR_REG_NONE) {
+		if (def_reg != op2_reg) {
+			ir_emit_mov(ctx, IR_U32, def_reg, op2_reg);
+		}
+		if (def_reg_hi != op2_reg_hi) {
+			ir_emit_mov(ctx, IR_U32, def_reg_hi, op2_reg_hi);
+		}
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];
+
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg, val->val.u32);
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg_hi, val->val.u32_hi);
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
 		} else {
-			mem = ir_ref_spill_slot(ctx, insn->op1);
+			mem = ir_ref_spill_slot(ctx, op2);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg_hi, mem_hi);
+	}

-		if (!src64) {
-			if (dst_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TXT_TMEM_OP vcvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword, mem
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TMEM_OP cvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
-				}
-			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TXT_TMEM_OP vcvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword, mem
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TMEM_OP cvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
-				}
-			}
+	|	jmp >3
+	|2:
+
+	if (op3_reg != IR_REG_NONE) {
+		if (def_reg != op3_reg) {
+			ir_emit_mov(ctx, IR_U32, def_reg, op3_reg);
+		}
+		if (def_reg_hi != op3_reg_hi) {
+			ir_emit_mov(ctx, IR_U32, def_reg_hi, op3_reg_hi);
+		}
+	} else if (IR_IS_CONST_REF(op3)) {
+		ir_insn *val = &ctx->ir_base[op3];
+
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg, val->val.u32);
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg_hi, val->val.u32_hi);
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op3) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op3);
 		} else {
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			if (dst_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TXT_TMEM_OP vcvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword, mem
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TMEM_OP cvtsi2sd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
-				}
-			} else {
-				IR_ASSERT(dst_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TXT_TMEM_OP vcvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword, mem
-				} else {
-					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
-					|	ASM_TXT_TMEM_OP cvtsi2ss, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
-				}
-			}
-|.endif
+			mem = ir_ref_spill_slot(ctx, op3);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg_hi, mem_hi);
 	}
+
+	|3:
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_fp2int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_cond_i64_cmp_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_ref op2 = insn->op2;
+	ir_ref op3 = insn->op3;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
-	bool dst64 = 0;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_op op;
+	ir_ref op2_reg_hi, op3_reg_hi, def_reg_hi;

-	IR_ASSERT(IR_IS_TYPE_FP(src_type));
-	IR_ASSERT(IR_IS_TYPE_INT(dst_type));
 	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (IR_IS_TYPE_SIGNED(dst_type) ? ir_type_size[dst_type] == 8 : ir_type_size[dst_type] >= 4) {
-		// TODO: we might need to perform truncation from 32/64 bit integer
-		dst64 = 1;
-	}
-	if (op1_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	def_reg_hi = IR_REG_I64_HI(def_reg);
+	def_reg = IR_REG_I64_LO(def_reg);
+
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-		if (!dst64) {
-			if (src_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttsd2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				} else {
-					|	cvttsd2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				}
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttss2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				} else {
-					|	cvttss2si Rd(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				}
+	} else {
+		op2_reg_hi = IR_REG_NONE;
+	}
+
+	if (op3_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op3_reg)) {
+			op3_reg = IR_REG_NUM(op3_reg);
+			op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			op3_reg = IR_REG_I64_LO(op3_reg);
+			if (op2 != op3) {
+				ir_emit_load_i64_lo(ctx, op3_reg, op3);
+				ir_emit_load_i64_hi(ctx, op3_reg_hi, op3);
 			}
 		} else {
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			if (src_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttsd2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				} else {
-					|	cvttsd2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				}
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttss2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				} else {
-					|	cvttss2si Rq(def_reg), xmm(op1_reg-IR_REG_FP_FIRST)
-				}
-			}
-|.endif
+			op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			op3_reg = IR_REG_I64_LO(op3_reg);
+		}
+	} else {
+		op3_reg_hi = IR_REG_NONE;
+	}
+
+	op = ir_emit_cmp_fp_common(ctx, def, insn->op1, &ctx->ir_base[insn->op1]);
+
+	switch (op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	jne >2
+			|	jp >2
+			break;
+		case IR_NE:
+			|	jp >1
+			|	je >2
+			break;
+		case IR_LT:
+			|	jp >2
+			|	jae >2
+			break;
+		case IR_GE:
+			|	jb >2
+			break;
+		case IR_LE:
+			|	jp >2
+			|	ja >2
+			break;
+		case IR_GT:
+			|	jbe >2
+			break;
+		case IR_ULT:
+			|	jae >2
+			break;
+		case IR_UGE:
+			|	jp >1
+			|	jb >2
+			break;
+		case IR_ULE:
+			|	ja >2
+			break;
+		case IR_UGT:
+			|	jp >1
+			|	jbe >2
+			break;
+		case IR_ORDERED:
+			|	jp >2
+			break;
+		case IR_UNORDERED:
+			|	jnp >2
+			break;
+	}
+	|1:
+
+	if (op2_reg != IR_REG_NONE) {
+		if (def_reg != op2_reg) {
+			ir_emit_mov(ctx, IR_U32, def_reg, op2_reg);
+		}
+		if (def_reg_hi != op2_reg_hi) {
+			ir_emit_mov(ctx, IR_U32, def_reg_hi, op2_reg_hi);
 		}
-	} else if (IR_IS_CONST_REF(insn->op1)) {
-		int label = ir_get_const_label(ctx, insn->op1);
+	} else if (IR_IS_CONST_REF(op2)) {
+		ir_insn *val = &ctx->ir_base[op2];

-		if (!dst64) {
-			if (src_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttsd2si Rd(def_reg), qword [=>label]
-				} else {
-					|	cvttsd2si Rd(def_reg), qword [=>label]
-				}
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttss2si Rd(def_reg), dword [=>label]
-				} else {
-					|	cvttss2si Rd(def_reg), dword [=>label]
-				}
-			}
-		} else {
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			if (src_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttsd2si Rq(def_reg), qword [=>label]
-				} else {
-					|	cvttsd2si Rq(def_reg), qword [=>label]
-				}
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vcvttss2si Rq(def_reg), dword [=>label]
-				} else {
-					|	cvttss2si Rq(def_reg), dword [=>label]
-				}
-			}
-|.endif
-		}
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg, val->val.u32);
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg_hi, val->val.u32_hi);
 	} else {
-		ir_mem mem;
+		ir_mem mem, mem_hi;

-		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, insn->op1);
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
 		} else {
-			mem = ir_ref_spill_slot(ctx, insn->op1);
+			mem = ir_ref_spill_slot(ctx, op2);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg_hi, mem_hi);
+	}

-		if (!dst64) {
-			if (src_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	ASM_TXT_TMEM_OP vcvttsd2si, Rd(def_reg), qword, mem
-				} else {
-					|	ASM_TXT_TMEM_OP cvttsd2si, Rd(def_reg), qword, mem
-				}
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	ASM_TXT_TMEM_OP vcvttss2si, Rd(def_reg), dword, mem
-				} else {
-					|	ASM_TXT_TMEM_OP cvttss2si, Rd(def_reg), dword, mem
-				}
-			}
+	|	jmp >3
+	|2:
+
+	if (op3_reg != IR_REG_NONE) {
+		if (def_reg != op3_reg) {
+			ir_emit_mov(ctx, IR_U32, def_reg, op3_reg);
+		}
+		if (def_reg_hi != op3_reg_hi) {
+			ir_emit_mov(ctx, IR_U32, def_reg_hi, op3_reg_hi);
+		}
+	} else if (IR_IS_CONST_REF(op3)) {
+		ir_insn *val = &ctx->ir_base[op3];
+
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg, val->val.u32);
+		ir_emit_load_imm_int(ctx, IR_U32, def_reg_hi, val->val.u32_hi);
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, op3) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op3);
 		} else {
-			IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-			if (src_type == IR_DOUBLE) {
-				if (ctx->mflags & IR_X86_AVX) {
-					|	ASM_TXT_TMEM_OP vcvttsd2si, Rq(def_reg), qword, mem
-				} else {
-					|	ASM_TXT_TMEM_OP cvttsd2si, Rq(def_reg), qword, mem
-				}
-			} else {
-				IR_ASSERT(src_type == IR_FLOAT);
-				if (ctx->mflags & IR_X86_AVX) {
-					|	ASM_TXT_TMEM_OP vcvttss2si, Rq(def_reg), dword, mem
-				} else {
-					|	ASM_TXT_TMEM_OP cvttss2si, Rq(def_reg), dword, mem
-				}
-			}
-|.endif
+			mem = ir_ref_spill_slot(ctx, op3);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg, mem);
+		ir_emit_load_mem(ctx, IR_U32, def_reg_hi, mem_hi);
 	}
+
+	|3:
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+		ir_emit_store_i64_lo(ctx, def, def_reg);
+		ir_emit_store_i64_hi(ctx, def, def_reg_hi);
 	}
 }

-static void ir_emit_fp2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_if_i64(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
 {
-	ir_type dst_type = insn->type;
-	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_reg op2_reg = ctx->regs[def][2];
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][0];
+	ir_reg op2_reg_hi;

-	IR_ASSERT(IR_IS_TYPE_FP(src_type));
-	IR_ASSERT(IR_IS_TYPE_FP(dst_type));
-	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (op1_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op1_reg)) {
-			op1_reg = IR_REG_NUM(op1_reg);
-			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+	IR_ASSERT(tmp_reg != IR_REG_NONE);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, insn->op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, insn->op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
 		}
-		if (src_type == dst_type) {
-			if (op1_reg != def_reg) {
-				ir_emit_fp_mov(ctx, dst_type, def_reg, op1_reg);
-			}
-		} else if (src_type == IR_DOUBLE) {
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vcvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
-			} else {
-				|	cvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
-			}
+		if (tmp_reg == op2_reg) {
+			|	or Rd(tmp_reg), Rd(op2_reg_hi);
+		} else if (tmp_reg == op2_reg_hi) {
+			|	or Rd(tmp_reg), Rd(op2_reg);
 		} else {
-			IR_ASSERT(src_type == IR_FLOAT);
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vcvtss2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
-			} else {
-				|	cvtss2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
-			}
+			|	mov Rd(tmp_reg), Rd(op2_reg);
+			|	or Rd(tmp_reg), Rd(op2_reg_hi);
 		}
-	} else if (IR_IS_CONST_REF(insn->op1)) {
-		int label = ir_get_const_label(ctx, insn->op1);
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		uint32_t true_block, false_block;

-		if (src_type == IR_DOUBLE) {
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vcvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword [=>label]
-			} else {
-				|	cvtsd2ss xmm(def_reg-IR_REG_FP_FIRST), qword [=>label]
+		ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+		if (ir_const_is_true(&ctx->ir_base[insn->op2])) {
+			if (true_block != next_block) {
+				|	jmp =>true_block
 			}
 		} else {
-			IR_ASSERT(src_type == IR_FLOAT);
-			if (ctx->mflags & IR_X86_AVX) {
-				|	vcvtss2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword [=>label]
-			} else {
-				|	cvtss2sd xmm(def_reg-IR_REG_FP_FIRST), dword [=>label]
+			if (false_block != next_block) {
+				|	jmp =>false_block
 			}
 		}
+		return;
+	} else if (ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA) {
+		uint32_t true_block, false_block;
+
+		ir_get_true_false_blocks(ctx, b, &true_block, &false_block);
+		if (true_block != next_block) {
+			|	jmp =>true_block
+		}
+		return;
 	} else {
-		ir_mem mem;
+		ir_mem mem, mem_hi;

-		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, insn->op1);
+		if (ir_rule(ctx, insn->op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op2);
 		} else {
-			mem = ir_ref_spill_slot(ctx, insn->op1);
+			mem = ir_ref_spill_slot(ctx, insn->op2);
 		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		|	ASM_REG_MEM_OP mov, IR_U32, tmp_reg, mem
+		|	ASM_REG_MEM_OP or, IR_U32, tmp_reg, mem_hi
+	}
+	ir_emit_jcc(ctx, b, def, insn, next_block, IR_NE, 1, 0);
+}

-		if (src_type == IR_DOUBLE) {
-			if (ctx->mflags & IR_X86_AVX) {
-				|	ASM_TXT_TXT_TMEM_OP vcvtsd2ss, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword, mem
-			} else {
-				|	ASM_TXT_TMEM_OP cvtsd2ss, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+static bool ir_emit_guard_i64(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][0];
+	ir_reg op2_reg_hi;
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+
+	IR_ASSERT(tmp_reg != IR_REG_NONE);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			ir_emit_load_i64_lo(ctx, op2_reg, insn->op2);
+			ir_emit_load_i64_hi(ctx, op2_reg_hi, insn->op2);
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+		}
+		if (tmp_reg == op2_reg) {
+			|	or Rd(tmp_reg), Rd(op2_reg_hi);
+		} else if (tmp_reg == op2_reg_hi) {
+			|	or Rd(tmp_reg), Rd(op2_reg);
+		} else {
+			|	mov Rd(tmp_reg), Rd(op2_reg);
+			|	or Rd(tmp_reg), Rd(op2_reg_hi);
+		}
+	} else if (IR_IS_CONST_REF(insn->op2)) {
+		if (ir_const_is_true(&ctx->ir_base[insn->op2])) {
+			if (insn->op == IR_GUARD_NOT) {
+				|	jmp &addr
+				return 1;
 			}
 		} else {
-			IR_ASSERT(src_type == IR_FLOAT);
-			if (ctx->mflags & IR_X86_AVX) {
-				|	ASM_TXT_TXT_TMEM_OP vcvtss2sd, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword, mem
-			} else {
-				|	ASM_TXT_TMEM_OP cvtss2sd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+			if (insn->op == IR_GUARD) {
+				|	jmp &addr
+				return 1;
 			}
 		}
+		return 0;
+	} else {
+		ir_mem mem, mem_hi;
+
+		if (ir_rule(ctx, insn->op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op2);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op2);
+		}
+		mem_hi = IR_MEM_I64_HI(mem);
+		|	ASM_REG_MEM_OP mov, IR_U32, tmp_reg, mem
+		|	ASM_REG_MEM_OP or, IR_U32, tmp_reg, mem_hi
 	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, dst_type, def, def_reg);
+
+	if (insn->op == IR_GUARD) {
+		|	je &addr
+	} else {
+		|	jne &addr
 	}
+
+	return 0;
 }

-static void ir_emit_copy_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static bool ir_emit_guard_cmp_i64(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
 {
-	ir_ref type = insn->type;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_insn *cmp_insn = &ctx->ir_base[insn->op2];
+	ir_op op = cmp_insn->op;
+	ir_ref op1 = cmp_insn->op1;
+	ir_ref op2 = cmp_insn->op2;
+	ir_reg op1_reg, op2_reg;
+	ir_reg op1_reg_hi, op2_reg_hi;
+	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);

-	IR_ASSERT(def_reg != IR_REG_NONE || op1_reg != IR_REG_NONE);
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
+		op1_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 1);
+		op2_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 2);
+	} else {
+		op1_reg = ctx->regs[insn->op2][1];
+		op2_reg = ctx->regs[insn->op2][2];
 	}
-	if (def_reg == op1_reg) {
-		/* same reg */
-	} else if (def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE) {
-		ir_emit_mov(ctx, type, def_reg, op1_reg);
-	} else if (def_reg != IR_REG_NONE) {
-		ir_emit_load(ctx, type, def_reg, insn->op1);
-	} else if (op1_reg != IR_REG_NONE) {
-		ir_emit_store(ctx, type, def, op1_reg);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+			ir_emit_load_i64_lo(ctx, op1_reg, op1);
+			ir_emit_load_i64_hi(ctx, op1_reg_hi, op1);
+		} else {
+			op1_reg_hi = IR_REG_I64_HI(op1_reg);
+			op1_reg = IR_REG_I64_LO(op1_reg);
+		}
 	} else {
-		IR_ASSERT(0);
+		op1_reg_hi = IR_REG_NONE;
 	}
-	if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load_i64_lo(ctx, op2_reg, op2);
+				ir_emit_load_i64_hi(ctx, op2_reg_hi, op2);
+			}
+		} else {
+			op2_reg_hi = IR_REG_I64_HI(op2_reg);
+			op2_reg = IR_REG_I64_LO(op2_reg);
+		}
+	} else {
+		op2_reg_hi = IR_REG_NONE;
 	}
-}

-static void ir_emit_copy_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_type type = insn->type;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][0];
+	ir_reg tmp2_reg = ctx->regs[insn->op2][3];
+	IR_ASSERT(tmp_reg != IR_REG_NONE);
+	ir_emit_cmp_i64_common(ctx, op, def, tmp_reg, tmp2_reg, op1_reg, op1_reg_hi, op1, op2_reg, op2_reg_hi, op2);

-	IR_ASSERT(def_reg != IR_REG_NONE || op1_reg != IR_REG_NONE);
-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, insn->op1);
-	}
-	if (def_reg == op1_reg) {
-		/* same reg */
-	} else if (def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE) {
-		ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
-	} else if (def_reg != IR_REG_NONE) {
-		ir_emit_load(ctx, type, def_reg, insn->op1);
-	} else if (op1_reg != IR_REG_NONE) {
-		ir_emit_store(ctx, type, def, op1_reg);
-	} else {
-		IR_ASSERT(0);
+	if (insn->op == IR_GUARD) {
+		op ^= 1; // reverse
 	}
-	if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+
+	switch (op) {
+		default:
+			IR_ASSERT(0 && "NIY binary op");
+		case IR_EQ:
+			|	je &addr
+			break;
+		case IR_NE:
+			|	jne &addr
+			break;
+		case IR_LT:
+		case IR_GT:
+			|	jl &addr
+			break;
+		case IR_GE:
+		case IR_LE:
+			|	jge &addr
+			break;
+		case IR_ULT:
+		case IR_UGT:
+			|	jc &addr
+			break;
+		case IR_UGE:
+		case IR_ULE:
+			|	jnc &addr
+			break;
 	}
+
+	return 0;
 }

-static void ir_emit_vaddr(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_return_i64(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_ref type = insn->type;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_mem mem;
-	int32_t offset;
-	ir_reg fp;
+	ir_reg ret_reg = data->ra_data.cc->int_ret_reg;
+	ir_reg ret2_reg = data->ra_data.cc->int_ret2_reg;
+	ir_reg op2_reg = ctx->regs[ref][2];
+	ir_reg op2_reg_hi;

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	mem = ir_var_spill_slot(ctx, insn->op1);
-	fp = IR_MEM_BASE(mem);
-	offset = IR_MEM_OFFSET(mem);
-	if (offset == 0) {
-		|	mov Ra(def_reg), Ra(fp)
+	if (op2_reg != IR_REG_NONE && !IR_REG_SPILLED(op2_reg)) {
+		op2_reg_hi = IR_REG_I64_HI(op2_reg);
+		op2_reg = IR_REG_I64_LO(op2_reg);
+		if (op2_reg != ret_reg) {
+			|	mov Rd(ret_reg), Rd(op2_reg)
+		}
+		if (op2_reg_hi != ret2_reg) {
+			|	mov Rd(ret2_reg), Rd(op2_reg_hi)
+		}
 	} else {
-		|	lea Ra(def_reg), aword [Ra(fp)+offset]
-	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_load_i64_lo(ctx, ret_reg, insn->op2);
+		ir_emit_load_i64_hi(ctx, ret2_reg, insn->op2);
 	}
+	ir_emit_return_void(ctx);
 }

-static void ir_emit_vload(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+|.endif
+#endif
+
+#if IR_SIMD
+static void ir_emit_vector_extract(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_insn *var_insn = &ctx->ir_base[insn->op2];
-	ir_ref type = insn->type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = ctx->ir_base[insn->op1].type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_reg fp;
-	ir_mem mem;

-	if (ctx->use_lists[def].count == 1) {
-		/* dead load */
-		return;
-	}
-	IR_ASSERT(var_insn->op == IR_VAR);
-	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	mem = IR_MEM_BO(fp, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
-	if (def_reg == IR_REG_NONE && ir_is_same_mem_var(ctx, def, var_insn->op3)) {
-		return; // fake load
-	}
-	IR_ASSERT(def_reg != IR_REG_NONE);
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);

-	ir_emit_load_mem(ctx, type, def_reg, mem);
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+	IR_ASSERT(insn->type == element_type ||
+		(IR_IS_TYPE_INT(insn->type) &&
+		 IR_IS_TYPE_INT(element_type) &&
+		 ir_type_size[insn->type] == ir_type_size[element_type]));
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
 	}
-}

-static void ir_emit_vstore_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
-{
-	ir_insn *var_insn = &ctx->ir_base[insn->op2];
-	ir_insn *val_insn = &ctx->ir_base[insn->op3];
-	ir_ref type = val_insn->type;
-	ir_reg op3_reg = ctx->regs[ref][3];
-	ir_reg fp;
-	ir_mem mem;
+	if (IR_IS_CONST_REF(insn->op2)) {
+		uint32_t lane = ctx->ir_base[insn->op2].val.u32;
+		ir_reg tmp_reg = op1_reg;
+
+		if (IR_IS_TYPE_INT(element_type)) {
+			if (element_type == IR_I8 || element_type == IR_U8) {
+				if (width <= 16) {
+					tmp_reg = op1_reg;
+				} else if (width == 32) {
+					if (lane < 16) {
+						tmp_reg = op1_reg;
+					} else {
+						tmp_reg = ctx->regs[def][3];
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						lane -= 16;
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							IR_ASSERT(ctx->mflags & IR_X86_AVX);
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(0 && "unsupprted vector wdith");
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vpextrb Ra(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					|	pextrb Ra(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+				} else {
+					|	pextrw Rd(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), (lane/2)
+					if (lane % 2 == 1) {
+						|	shr Rd(def_reg), 8
+					}
+				}
+			} else if (element_type == IR_I16 || element_type == IR_U16) {
+				if (width <= 16) {
+					tmp_reg = op1_reg;
+				} else if (width == 32) {
+					if (lane < 8) {
+						tmp_reg = op1_reg;
+					} else {
+						tmp_reg = ctx->regs[def][3];
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						lane -= 8;
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							IR_ASSERT(ctx->mflags & IR_X86_AVX);
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(0 && "unsupprted vector wdith");
+				}
+				|	pextrw Rd(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+			} else if (element_type == IR_I32 || element_type == IR_U32) {
+				if (width <= 16) {
+					tmp_reg = op1_reg;
+				} else if (width == 32) {
+					if (lane < 4) {
+						tmp_reg = op1_reg;
+					} else {
+						tmp_reg = ctx->regs[def][3];
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						lane -= 4;
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							IR_ASSERT(ctx->mflags & IR_X86_AVX);
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(0 && "unsupprted vector wdith");
+				}
+				if (lane == 0) {
+					|	movd Rd(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+				} else if (ctx->mflags & IR_X86_AVX) {
+					|	vpextrd Rd(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					|	pextrd Rd(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+				} else {
+					op1_reg = tmp_reg;
+					tmp_reg = ctx->regs[def][3];
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+
+					if (lane == 1) {
+						|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 85
+					} else if (lane == 2) {
+						if (op1_reg != tmp_reg) {
+							|	movapd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	punpckhdq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else if (lane == 3) {
+						|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 255
+					} else {
+						IR_ASSERT(0);
+					}
+					|	movd Rd(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+				}
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				if (width <= 16) {
+					tmp_reg = op1_reg;
+				} else if (width == 32) {
+					if (lane < 2) {
+						tmp_reg = op1_reg;
+					} else {
+						tmp_reg = ctx->regs[def][3];
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						lane -= 2;
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							IR_ASSERT(ctx->mflags & IR_X86_AVX);
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(0 && "unsupprted vector wdith");
+				}
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (lane == 0) {
+					|	movq Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+				} else if (ctx->mflags & IR_X86_AVX) {
+					|	vpextrq Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					|	pextrq Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST), lane
+				} else {
+					op1_reg = tmp_reg;
+					tmp_reg = ctx->regs[def][3];
+					IR_ASSERT(tmp_reg != IR_REG_NONE);

-	IR_ASSERT(var_insn->op == IR_VAR);
-	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	mem = IR_MEM_BO(fp, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
-	if ((op3_reg == IR_REG_NONE || IR_REG_SPILLED(op3_reg))
-	 && !IR_IS_CONST_REF(insn->op3)
-	 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
-	 && ir_is_same_mem_var(ctx, insn->op3, var_insn->op3)) {
-		return; // fake store
-	}
-	if (IR_IS_CONST_REF(insn->op3)) {
-		ir_emit_store_mem_int_const(ctx, type, mem, insn->op3, op3_reg, 0);
+					IR_ASSERT(lane == 1);
+					|	movhlps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	movq Rq(def_reg), xmm(tmp_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif IR_X86_I64
+|.if not X64
+				ir_mem mem = IR_MEM(IR_REG_RSP, -8, IR_REG_NONE, 1);
+				ir_mem mem_hi = IR_MEM_I64_HI(mem);
+				ir_reg def_reg_hi = IR_REG_I64_HI(def_reg);
+
+				def_reg = IR_REG_I64_LO(def_reg);
+
+				if (lane == 0) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovq qword [Ra(IR_REG_RSP)-8], xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movq qword [Ra(IR_REG_RSP)-8], xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					op1_reg = tmp_reg;
+					tmp_reg = ctx->regs[def][3];
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+
+					IR_ASSERT(lane == 1);
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovhlps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vmovq qword [Ra(IR_REG_RSP)-8], xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movhlps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	movq qword [Ra(IR_REG_RSP)-8], xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				}
+
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg, mem);
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, mem_hi);
+
+				if (IR_REG_SPILLED(ctx->regs[def][0])) {
+					ir_emit_store_i64_lo(ctx, def, def_reg);
+					ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+				}
+				return;
+|.endif
+#else
+				IR_ASSERT(0 && "unsupprted vector type");
+#endif
+			} else {
+				IR_ASSERT(0 && "unsupprted vector type");
+			}
+		} else {
+			if (element_type == IR_DOUBLE) {
+				IR_ASSERT(width <= 32);
+				if (width == 32 && lane >= 2) {
+					lane -= 2;
+					IR_ASSERT(ctx->mflags & IR_X86_AVX);
+					|	vextractf128 xmm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+					op1_reg = def_reg;
+				}
+				if (lane == 0) {
+					if (def_reg != op1_reg) {
+						|	movapd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else if (lane == 1) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vunpckhpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						if (def_reg != op1_reg) {
+							|	movapd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	unpckhpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(0);
+				}
+			} else {
+				IR_ASSERT(element_type == IR_FLOAT);
+				IR_ASSERT(width <= 32);
+				if (width == 32 && lane >= 4) {
+					lane -= 4;
+					IR_ASSERT(ctx->mflags & IR_X86_AVX);
+					|	vextractf128 xmm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+					op1_reg = def_reg;
+				}
+				if (lane == 0) {
+					if (def_reg != op1_reg) {
+						|	movaps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else if (lane == 1) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 85
+					} else {
+						if (def_reg != op1_reg) {
+							|	movaps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	shufps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 85
+					}
+				} else if (lane == 2) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vunpckhps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						if (def_reg != op1_reg) {
+							|	movaps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	unpckhps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else if (lane == 3) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 255
+					} else {
+						if (def_reg != op1_reg) {
+							|	movaps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+						|	shufps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 255
+					}
+				} else {
+					IR_ASSERT(0);
+				}
+			}
+		}
 	} else {
-		IR_ASSERT(op3_reg != IR_REG_NONE);
-		if (IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, insn->op3);
+		ir_reg op2_reg = ctx->regs[def][2];
+
+		IR_ASSERT(op2_reg != IR_REG_NONE);
+
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, element_type, op2_reg, insn->op2);
 		}
-		ir_emit_store_mem_int(ctx, type, mem, op3_reg);
-	}
-}

-static void ir_emit_vstore_fp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
-{
-	ir_insn *var_insn = &ctx->ir_base[insn->op2];
-	ir_ref type = ctx->ir_base[insn->op3].type;
-	ir_reg op3_reg = ctx->regs[ref][3];
-	ir_reg fp;
-	ir_mem mem;
+		/* extract through stack memory */
+		int offset = -width;
+		ir_mem mem = IR_MEM(IR_REG_RSP, offset, IR_REG_NONE, 1);
+		ir_mem mem2 = IR_MEM(IR_REG_RSP, offset, op2_reg, ir_type_size[element_type]);

-	IR_ASSERT(var_insn->op == IR_VAR);
-	fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-	mem = IR_MEM_BO(fp, IR_SPILL_POS_TO_OFFSET(var_insn->op3));
-	if ((op3_reg == IR_REG_NONE || IR_REG_SPILLED(op3_reg))
-	 && !IR_IS_CONST_REF(insn->op3)
-	 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
-	 && ir_is_same_mem_var(ctx, insn->op3, var_insn->op3)) {
-		return; // fake store
-	}
-	if (IR_IS_CONST_REF(insn->op3)) {
-		ir_emit_store_mem_fp_const(ctx, type, mem, insn->op3, IR_REG_NONE, op3_reg);
-	} else {
-		IR_ASSERT(op3_reg != IR_REG_NONE);
-		if (IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, insn->op3);
+		ir_emit_store_mem_fp(ctx, type, mem, op1_reg);
+
+#if IR_X86_I64
+|.if not X64
+		if (element_type == IR_I64 || element_type == IR_U64) {
+			ir_reg def_reg_hi = IR_REG_I64_HI(def_reg);
+
+			def_reg = IR_REG_I64_LO(def_reg);
+			if (def_reg == op2_reg) {
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, IR_MEM_I64_HI(mem2));
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg, mem2);
+			} else {
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg, mem2);
+				ir_emit_load_mem_int(ctx, IR_U32, def_reg_hi, IR_MEM_I64_HI(mem2));
+			}
+
+			if (IR_REG_SPILLED(ctx->regs[def][0])) {
+				ir_emit_store_i64_lo(ctx, def, def_reg);
+				ir_emit_store_i64_hi(ctx, def, def_reg_hi);
+			}
+			return;
 		}
-		ir_emit_store_mem_fp(ctx, type, mem, op3_reg);
+|.endif
+#endif
+		ir_emit_load_mem(ctx, element_type, def_reg, mem2);
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_load_int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vector_replace(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_ref type = insn->type;
-	ir_reg op2_reg = ctx->regs[def][2];
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op3_reg = ctx->regs[def][3];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_mem mem;
+	ir_reg tmp_reg = IR_REG_NONE;

-	if (ctx->use_lists[def].count == 1) {
-		/* dead load */
-		return;
-	}
-	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
-			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type) && type == ctx->ir_base[insn->op1].type);
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(element_type == ctx->ir_base[insn->op3].type ||
+		(IR_IS_TYPE_INT(element_type) &&
+		 IR_IS_TYPE_INT(ctx->ir_base[insn->op3].type) &&
+		 ir_type_size[element_type] == ir_type_size[ctx->ir_base[insn->op3].type]));
+	IR_ASSERT(def_reg != IR_REG_NONE && op3_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, insn->op1);
 		}
-		mem = IR_MEM_B(op2_reg);
-	} else if (IR_IS_CONST_REF(insn->op2)) {
-		mem = ir_fuse_addr_const(ctx, insn->op2);
+		if (op1_reg != def_reg) {
+			ir_emit_fp_mov(ctx, insn->type, def_reg, op1_reg);
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		ir_emit_load_imm_fp(ctx, insn->type, def_reg, insn->op1);
 	} else {
-		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
-		mem = ir_fuse_addr(ctx, def, insn->op2);
-		if (IR_REG_SPILLED(ctx->regs[def][0]) && ir_is_same_spill_slot(ctx, def, mem)) {
-			if (!ir_may_avoid_spill_load(ctx, def, def)) {
-				ir_emit_load_mem_int(ctx, type, def_reg, mem);
+		ir_mem mem;
+
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, mem);
+	}
+
+	if (IR_REG_SPILLED(op3_reg)) {
+		op3_reg = IR_REG_NUM(op3_reg);
+#if IR_X86_I64
+		if (element_type == IR_U64 || element_type == IR_I64) {
+			ir_reg op3_reg_hi = IR_REG_I64_HI(op3_reg);
+			ir_reg op3_reg_lo = IR_REG_I64_LO(op3_reg);
+			ir_emit_load_i64_lo(ctx, op3_reg_lo, insn->op3);
+			ir_emit_load_i64_hi(ctx, op3_reg_hi, insn->op3);
+		} else
+#endif
+		ir_emit_load(ctx, element_type, op3_reg, insn->op3);
+	}
+
+	if (IR_IS_CONST_REF(insn->op2)) {
+		uint32_t lane = ctx->ir_base[insn->op2].val.u32;
+
+		if (width <= 16) {
+			tmp_reg = op1_reg = def_reg;
+		} else if (width == 32) {
+			tmp_reg = ctx->tmp_regs[def];
+			if (lane * ir_type_size[element_type] < 16) {
+				op1_reg = def_reg;
+			} else {
+				op1_reg = tmp_reg;
+				IR_ASSERT(tmp_reg != IR_REG_NONE);
+				if (ctx->mflags & IR_X86_AVX2) {
+					// TODO: consider usage of "VPBLENDD" ???
+					|	vextracti128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+				} else {
+					IR_ASSERT(ctx->mflags & IR_X86_AVX);
+					|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+				}
 			}
-			/* avoid load to the same location (valid only when register is not reused) */
-			return;
+		} else {
+			IR_ASSERT(0 && "unsupprted vector wdith");
+		}
+
+		if (IR_IS_TYPE_INT(element_type)) {
+			if (element_type == IR_I8 || element_type == IR_U8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	movzx Rd(op3_reg), Rb(op3_reg)
+					|	vpinsrb xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), Rd(op3_reg), (lane%16)
+				} else if (ctx->mflags & IR_X86_SSE42) {
+					|	movzx Rd(op3_reg), Rb(op3_reg)
+					|	pinsrb xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op3_reg), (lane%16)
+				} else {
+					/* modify through stack memory */
+					int offset = width;
+
+					|	movdqu [Ra(IR_REG_RSP)-offset], xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	mov [Ra(IR_REG_RSP)-(offset-(lane%16))], Rb(op3_reg)
+					|	movdqu xmm(tmp_reg-IR_REG_FP_FIRST), [Ra(IR_REG_RSP)-offset]
+				}
+			} else if (element_type == IR_I16 || element_type == IR_U16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vpinsrw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), Rd(op3_reg), (lane%8)
+				} else {
+					|	pinsrw xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op3_reg), (lane%8)
+				}
+			} else if (element_type == IR_I32 || element_type == IR_U32) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vpinsrd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), Rd(op3_reg), (lane%4)
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					|	pinsrd xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op3_reg), (lane%4)
+				} else {
+					tmp_reg = ctx->tmp_regs[def];
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+
+					|	movd xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op3_reg)
+					if (lane % 4 == 0) {
+						|	movss xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else if (lane % 4 == 1) {
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 225
+						|	movss xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 225
+					} else if (lane % 4 == 2) {
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 198
+						|	movss xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 198
+					} else if (lane % 4 == 3) {
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 39
+						|	movss xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 39
+					}
+				}
+#if defined(IR_TARGET_X64)
+|.if X64
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				if (ctx->mflags & IR_X86_AVX) {
+					|	vpinsrq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), Rq(op3_reg), (lane%2)
+				} else if (ctx->mflags & IR_X86_SSE42) {
+					|	pinsrq xmm(tmp_reg-IR_REG_FP_FIRST), Rq(op3_reg), (lane%2)
+				} else {
+					tmp_reg = ctx->tmp_regs[def];
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+
+					|	movq xmm(tmp_reg-IR_REG_FP_FIRST), Rq(op3_reg)
+					if (lane % 2 == 0) {
+						|	movsd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				}
+|.endif
+#elif IR_X86_I64
+|.if not X64
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				ir_mem mem = IR_MEM(IR_REG_RSP, -8, IR_REG_NONE, 1);
+				ir_mem mem_hi = IR_MEM_I64_HI(mem);
+				ir_reg op3_reg_hi = IR_REG_I64_HI(op3_reg);
+
+				op3_reg = IR_REG_I64_LO(op3_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, mem, op3_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, mem_hi, op3_reg_hi);
+
+				if (ctx->mflags & IR_X86_AVX) {
+					if (lane % 2 == 0) {
+						|	ASM_TXT_TMEM_OP vmovsd, xmm(tmp_reg-IR_REG_FP_FIRST), qword, mem
+					} else {
+						|	ASM_TXT_TXT_TMEM_OP vmovhpd, xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), qword, mem
+					}
+				} else {
+					if (lane % 2 == 0) {
+						|	ASM_TXT_TMEM_OP movsd, xmm(tmp_reg-IR_REG_FP_FIRST), qword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP movhpd, xmm(tmp_reg-IR_REG_FP_FIRST), qword, mem
+					}
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
+			}
+		} else {
+			if (element_type == IR_DOUBLE) {
+				if (lane % 2 == 0) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovsd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movsd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vunpcklpd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST)
+					} else {
+						|	unpcklpd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST)
+					}
+				}
+			} else {
+				if (lane % 4 == 0) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vmovss xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movss xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST)
+					}
+				} else if (lane % 4 == 1) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vinsertps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST), 16
+					} else {
+						|	insertps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST), 16
+					}
+				} else if (lane % 4 == 2) {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vinsertps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST), 32
+					} else {
+						|	insertps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST), 32
+					}
+				} else {
+					if (ctx->mflags & IR_X86_AVX) {
+						|	vinsertps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST), 48
+					} else {
+						|	insertps xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op3_reg-IR_REG_FP_FIRST), 48
+					}
+				}
+			}
+		}
+
+		if (width == 32) {
+			int idx = lane * ir_type_size[element_type] >= 16;
+			if (ctx->mflags & IR_X86_AVX2) {
+				|	vinserti128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), idx
+			} else {
+				IR_ASSERT(ctx->mflags & IR_X86_AVX);
+				|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), idx
+			}
+		}
+	} else {
+		ir_reg op2_reg = ctx->regs[def][2];
+
+		IR_ASSERT(op2_reg != IR_REG_NONE);
+
+		if (IR_REG_SPILLED(op2_reg)) {
+			op3_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, element_type, op2_reg, insn->op2);
+		}
+
+		/* modify through stack memory */
+		int offset = -width;
+		ir_mem mem = IR_MEM(IR_REG_RSP, offset, IR_REG_NONE, 1);
+		ir_mem mem2 = IR_MEM(IR_REG_RSP, offset, op2_reg, ir_type_size[element_type]);
+
+		ir_emit_store_mem_fp(ctx, insn->type, mem, op1_reg);
+		if (IR_IS_TYPE_INT(element_type)) {
+#if IR_X86_I64
+|.if not X64
+			if (element_type == IR_I64 || element_type == IR_U64) {
+				ir_reg op3_reg_hi = IR_REG_I64_HI(op3_reg);
+
+				op3_reg = IR_REG_I64_LO(op3_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, mem2, op3_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, IR_MEM_I64_HI(mem), op3_reg_hi);
+			} else
+|.endif
+#endif
+			ir_emit_store_mem_int(ctx, element_type, mem2, op3_reg);
+		} else {
+			ir_emit_store_mem_fp(ctx, element_type, mem2, op3_reg);
 		}
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, mem);
 	}

-	ir_emit_load_mem_int(ctx, type, def_reg, mem);
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_load_fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vector_splat(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
-	ir_ref type = insn->type;
-	ir_reg op2_reg = ctx->regs[def][2];
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_mem mem;
+	ir_reg tmp_reg = ctx->regs[def][2];

-	if (ctx->use_lists[def].count == 1) {
-		/* dead load */
-		return;
-	}
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(element_type == ctx->ir_base[insn->op1].type ||
+		(IR_IS_TYPE_INT(element_type) &&
+		 IR_IS_TYPE_INT(ctx->ir_base[insn->op1].type) &&
+		 ir_type_size[element_type] == ir_type_size[ctx->ir_base[insn->op1].type]));
 	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
-			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+#if IR_X86_I64
+			if (element_type == IR_U64 || element_type == IR_I64) {
+				ir_reg op1_reg_hi = IR_REG_I64_HI(op1_reg);
+				ir_reg op1_reg_lo = IR_REG_I64_LO(op1_reg);
+				ir_emit_load_i64_lo(ctx, op1_reg_lo, insn->op1);
+				ir_emit_load_i64_hi(ctx, op1_reg_hi, insn->op1);
+			} else
+#endif
+			ir_emit_load(ctx, element_type, op1_reg, insn->op1);
 		}
-		mem = IR_MEM_B(op2_reg);
-	} else if (IR_IS_CONST_REF(insn->op2)) {
-		mem = ir_fuse_addr_const(ctx, insn->op2);
-	} else {
-		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
-		mem = ir_fuse_addr(ctx, def, insn->op2);
-		if (IR_REG_SPILLED(ctx->regs[def][0]) && ir_is_same_spill_slot(ctx, def, mem)) {
-			if (!ir_may_avoid_spill_load(ctx, def, def)) {
-				ir_emit_load_mem_fp(ctx, type, def_reg, mem);
+		if (IR_IS_TYPE_INT(element_type)) {
+			if (element_type == IR_I8 || element_type == IR_U8) {
+				|	movzx Rd(op1_reg), Rb(op1_reg)
+				if (ctx->mflags & IR_X86_AVX) {
+					IR_ASSERT(width <= 32);
+					|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (width <= 16) {
+							|	vpbroadcastb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpbroadcastb ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupprted vector wdith");
+						}
+					} else {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	vpshufb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						if (width == 32) {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSSE3) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					IR_ASSERT(width <= 16);
+					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	pshufb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(width <= 16);
+					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 0
+				}
+			} else if (element_type == IR_I16 || element_type == IR_U16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					IR_ASSERT(width <= 32);
+					|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (width <= 16) {
+							|	vpbroadcastw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpbroadcastw ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupprted vector wdith");
+						}
+					} else {
+						|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 0
+						|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 68
+						if (width == 32) {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(width <= 16);
+					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 0
+				}
+			} else if (element_type == IR_I32 || element_type == IR_U32) {
+				if (ctx->mflags & IR_X86_AVX) {
+					IR_ASSERT(width <= 32);
+					|	vmovd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (width <= 16) {
+							|	vpbroadcastd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpbroadcastd ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupprted vector wdith");
+						}
+					} else {
+						|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 0
+						if (width == 32) {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(width <= 16);
+					|	movd xmm(def_reg-IR_REG_FP_FIRST), Rd(op1_reg)
+					|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 0
+				}
+#if defined(IR_TARGET_X64)
+|.if X64
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				if (ctx->mflags & IR_X86_AVX) {
+					IR_ASSERT(width <= 32);
+					|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (width <= 16) {
+							|	vpbroadcastq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpbroadcastq ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupprted vector wdith");
+						}
+					} else {
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						if (width == 32) {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(width <= 16);
+					|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(op1_reg)
+					|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif IR_X86_I64
+|.if not X64
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				ir_mem mem = IR_MEM(IR_REG_RSP, -8, IR_REG_NONE, 1);
+				ir_mem mem_hi = IR_MEM_I64_HI(mem);
+				ir_reg op1_reg_hi = IR_REG_I64_HI(op1_reg);
+
+				op1_reg = IR_REG_I64_LO(op1_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, mem, op1_reg);
+				ir_emit_store_mem_int(ctx, IR_U32, mem_hi, op1_reg_hi);
+
+				if (ctx->mflags & IR_X86_AVX) {
+					IR_ASSERT(width <= 32);
+					|	ASM_TXT_TMEM_OP vmovq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (width <= 16) {
+							|	vpbroadcastq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpbroadcastq ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupprted vector wdith");
+						}
+					} else {
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						if (width == 32) {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(width <= 16);
+					|	ASM_TXT_TMEM_OP movq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
+			}
+		} else {
+			if (element_type == IR_DOUBLE) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (width <= 16) {
+						|	vmovddup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vbroadcastsd ymm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vmovddup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(0 && "unsupprted vector wdith");
+					}
+				} else if (ctx->mflags & IR_X86_SSE42) {
+					IR_ASSERT(width <= 16);
+					|	movddup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(width <= 16);
+					if (def_reg != op1_reg) {
+						|	movapd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	unpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(element_type == IR_FLOAT);
+				if (ctx->mflags & IR_X86_AVX) {
+					IR_ASSERT(width <= 32);
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (width <= 16) {
+							|	vbroadcastss xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vbroadcastss ymm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupprted vector wdith");
+						}
+					} else {
+						|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 0
+						if (width == 32) {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(width <= 16);
+					if (op1_reg != def_reg) {
+						|	movaps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	shufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 0
+				}
 			}
-			/* avoid load to the same location (valid only when register is not reused) */
-			return;
 		}
+	} else {
+		IR_ASSERT(0);
 	}

-	ir_emit_load_mem_fp(ctx, type, def_reg, mem);
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_store_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
-{
-	ir_insn *val_insn = &ctx->ir_base[insn->op3];
-	ir_ref type = val_insn->type;
-	ir_reg op2_reg = ctx->regs[ref][2];
-	ir_reg op3_reg = ctx->regs[ref][3];
-	ir_mem mem;

-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
-			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
-		}
-		mem = IR_MEM_B(op2_reg);
-	} else if (IR_IS_CONST_REF(insn->op2)) {
-		mem = ir_fuse_addr_const(ctx, insn->op2);
-	} else {
-		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
-		mem = ir_fuse_addr(ctx, ref, insn->op2);
-		if (!IR_IS_CONST_REF(insn->op3)
-		 && IR_REG_SPILLED(op3_reg)
-		 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
-		 && ir_is_same_spill_slot(ctx, insn->op3, mem)) {
-			if (!ir_may_avoid_spill_load(ctx, insn->op3, ref)) {
-				op3_reg = IR_REG_NUM(op3_reg);
-				ir_emit_load(ctx, type, op3_reg, insn->op3);
-			}
-			/* avoid store to the same location */
-			return;
+static ir_reg ir_emit_load_reg(ir_ctx *ctx, ir_type type, ir_reg reg, ir_ref ref)
+{
+	if (reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(reg)) {
+			reg = IR_REG_NUM(reg);
+			ir_emit_load(ctx, type, reg, ref);
 		}
 	}
+	return reg;
+}

-	if (IR_IS_CONST_REF(insn->op3)) {
-		ir_emit_store_mem_int_const(ctx, type, mem, insn->op3, op3_reg, 0);
-	} else {
-		IR_ASSERT(op3_reg != IR_REG_NONE);
-		if (IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, insn->op3);
+static void ir_emit_load_def_reg(ir_ctx *ctx, ir_type type, ir_reg def_reg, ir_reg use_reg, ir_ref use)
+{
+	if (def_reg != use_reg) {
+		if (use_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(use_reg)) {
+				use_reg = IR_REG_NUM(use_reg);
+				ir_emit_load(ctx, type, use_reg, use);
+			}
+			if (IR_IS_TYPE_INT(type)) {
+				ir_emit_mov(ctx, type, def_reg, use_reg);
+			} else {
+				ir_emit_fp_mov(ctx, type, def_reg, use_reg);
+			}
+		} else {
+			ir_emit_load(ctx, type, def_reg, use);
 		}
-		ir_emit_store_mem_int(ctx, type, mem, op3_reg);
 	}
 }

-static void ir_emit_cmp_and_store_int(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+static uint32_t ir_shufps_mask(ir_ctx *ctx, ir_ref ref)
 {
-	ir_reg addr_reg = ctx->regs[ref][2];
-	ir_mem mem;
-	ir_insn *cmp_insn = &ctx->ir_base[insn->op3];
-	ir_op op = cmp_insn->op;
-	ir_type type = ctx->ir_base[cmp_insn->op1].type;
-	ir_ref op1 = cmp_insn->op1;
-	ir_ref op2 = cmp_insn->op2;
-	ir_reg op1_reg = ctx->regs[insn->op3][1];
-	ir_reg op2_reg = ctx->regs[insn->op3][2];
+	int8_t *p;
+	uint32_t i, s, mask = 0;

-	if (addr_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(addr_reg)) {
-			addr_reg = IR_REG_NUM(addr_reg);
-			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
-			ir_emit_load(ctx, IR_ADDR, addr_reg, insn->op2);
-		}
-		mem = IR_MEM_B(addr_reg);
-	} else if (IR_IS_CONST_REF(insn->op2)) {
-		mem = ir_fuse_addr_const(ctx, insn->op2);
-	} else {
-		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
-		mem = ir_fuse_addr(ctx, ref, insn->op2);
-	}

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
-		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	for (i = 0; i < 4; i++) {
+		mask |= (IR_SHUFFLE_MASK(i) & 3) << (i * 2);
 	}
-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		if (op1 != op2) {
-			ir_emit_load(ctx, type, op2_reg, op2);
+	return mask;
+}
+
+static uint32_t ir_shufps_mask_12_0(ir_ctx *ctx, ir_ref ref)
+{
+	int8_t *p;
+	uint32_t i, i1, i2, s, mask1 = 0, mask2 = 0;
+
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	for (i = i1 = i2 = 0; i < 4; i++) {
+		if (!(IR_SHUFFLE_MASK(i) & 4)) {
+			if (i1 && (mask1 & 3) == (IR_SHUFFLE_MASK(i) & 3)) {
+				/* pass */
+			} else if (i1 == 2 && ((mask1 >> 2) & 3) == (IR_SHUFFLE_MASK(i) & 3)) {
+				mask2 |= 1 << (i * 2);
+			} else {
+				IR_ASSERT(i1 < 2);
+				mask1 |= (IR_SHUFFLE_MASK(i) & 3) << (i1 * 2);
+				mask2 |= i1 << (i * 2);
+				i1++;
+			}
+		} else {
+			if (i2 && ((mask1 >> 4) & 3) == (IR_SHUFFLE_MASK(i) & 3)) {
+				mask2 |= 2 << (i * 2);
+			} else if (i2 == 2 && ((mask1 >> 6) & 3) == (IR_SHUFFLE_MASK(i) & 3)) {
+				mask2 |= 3 << (i * 2);
+			} else {
+				IR_ASSERT(i2 < 2);
+				mask1 |= (IR_SHUFFLE_MASK(i) & 3) << ((i2 + 2) * 2);
+				mask2 |= (i2 + 2) << (i * 2);
+				i2++;
+			}
 		}
 	}
-
-	ir_emit_cmp_int_common(ctx, type, ref, cmp_insn, op1_reg, op1, op2_reg, op2);
-	_ir_emit_setcc_int_mem(ctx, op, mem);
+	return mask1 | (mask2 << 8);
 }

-static void ir_emit_store_fp(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+static uint32_t ir_shufps_mask_12_n(ir_ctx *ctx, ir_ref ref)
 {
-	ir_ref type = ctx->ir_base[insn->op3].type;
-	ir_reg op2_reg = ctx->regs[ref][2];
-	ir_reg op3_reg = ctx->regs[ref][3];
-	ir_mem mem;
+	int8_t *p;
+	uint32_t s, mask1 = 0, mask2 = 0;
+
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	IR_ASSERT((IR_SHUFFLE_MASK(0) & 4) != (IR_SHUFFLE_MASK(1) & 4));
+	if (!(IR_SHUFFLE_MASK(0) & 4)) {
+		mask1 |= (IR_SHUFFLE_MASK(0) & 3);
+		mask1 |= (IR_SHUFFLE_MASK(0) & 3) << 2;
+		mask1 |= (IR_SHUFFLE_MASK(1) & 3) << 4;
+		mask1 |= (IR_SHUFFLE_MASK(1) & 3) << 6;
+		mask2 |= 0x8;
+	} else {
+		mask1 |= (IR_SHUFFLE_MASK(1) & 3);
+		mask1 |= (IR_SHUFFLE_MASK(1) & 3) << 2;
+		mask1 |= (IR_SHUFFLE_MASK(0) & 3) << 4;
+		mask1 |= (IR_SHUFFLE_MASK(0) & 3) << 6;
+		mask2 |= 0x2;
+	}
+	mask2 |= (IR_SHUFFLE_MASK(2) & 3) << 4;
+	mask2 |= (IR_SHUFFLE_MASK(3) & 3) << 6;
+	return mask1 | (mask2 << 8);
+}

-	IR_ASSERT(op3_reg != IR_REG_NONE);
-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			IR_ASSERT(ctx->ir_base[insn->op2].type == IR_ADDR);
-			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
-		}
-		mem = IR_MEM_B(op2_reg);
-	} else if (IR_IS_CONST_REF(insn->op2)) {
-		mem = ir_fuse_addr_const(ctx, insn->op2);
+static uint32_t ir_shufps_mask_n_12(ir_ctx *ctx, ir_ref ref)
+{
+	int8_t *p;
+	uint32_t s, mask1 = 0, mask2 = 0;
+
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	IR_ASSERT((IR_SHUFFLE_MASK(2) & 4) != (IR_SHUFFLE_MASK(3) & 4));
+	mask2 |= (IR_SHUFFLE_MASK(0) & 3);
+	mask2 |= (IR_SHUFFLE_MASK(1) & 3) << 2;
+	if (!(IR_SHUFFLE_MASK(2) & 4)) {
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3);
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3) << 2;
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3) << 4;
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3) << 6;
+		mask2 |= 0x80;
 	} else {
-		IR_ASSERT(ir_rule(ctx, insn->op2) & IR_FUSED);
-		mem = ir_fuse_addr(ctx, ref, insn->op2);
-		if (!IR_IS_CONST_REF(insn->op3)
-		 && IR_REG_SPILLED(op3_reg)
-		 && ir_rule(ctx, insn->op3) != IR_STATIC_ALLOCA
-		 && ir_is_same_spill_slot(ctx, insn->op3, mem)) {
-			if (!ir_may_avoid_spill_load(ctx, insn->op3, ref)) {
-				op3_reg = IR_REG_NUM(op3_reg);
-				ir_emit_load(ctx, type, op3_reg, insn->op3);
-			}
-			/* avoid store to the same location */
-			return;
-		}
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3);
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3) << 2;
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3) << 4;
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3) << 6;
+		mask2 |= 0x20;
 	}
+	return mask1 | (mask2 << 8);
+}

-	if (IR_IS_CONST_REF(insn->op3)) {
-		ir_emit_store_mem_fp_const(ctx, type, mem, insn->op3, IR_REG_NONE, op3_reg);
+static uint32_t ir_shufps_mask_n_21(ir_ctx *ctx, ir_ref ref)
+{
+	int8_t *p;
+	uint32_t s, mask1 = 0, mask2 = 0;
+
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	IR_ASSERT((IR_SHUFFLE_MASK(2) & 4) != (IR_SHUFFLE_MASK(3) & 4));
+	mask2 |= (IR_SHUFFLE_MASK(0) & 3);
+	mask2 |= (IR_SHUFFLE_MASK(1) & 3) << 2;
+	if (!(IR_SHUFFLE_MASK(2) & 4)) {
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3);
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3) << 2;
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3) << 4;
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3) << 6;
+		mask2 |= 0x20;
 	} else {
-		IR_ASSERT(op3_reg != IR_REG_NONE);
-		if (IR_REG_SPILLED(op3_reg)) {
-			op3_reg = IR_REG_NUM(op3_reg);
-			ir_emit_load(ctx, type, op3_reg, insn->op3);
-		}
-		ir_emit_store_mem_fp(ctx, type, mem, op3_reg);
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3);
+		mask1 |= (IR_SHUFFLE_MASK(2) & 3) << 2;
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3) << 4;
+		mask1 |= (IR_SHUFFLE_MASK(3) & 3) << 6;
+		mask2 |= 0x80;
 	}
+	return mask1 | (mask2 << 8);
 }

-static void ir_emit_rload(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_shufpd(ir_ctx *ctx, ir_ref def_reg, ir_reg op2_reg, ir_ref ref)
 {
-	ir_reg src_reg = insn->op2;
-	ir_type type = insn->type;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	int8_t *p;
+	uint32_t s, mask = 0;

-	if (IR_REGSET_IN(IR_REGSET_UNION((ir_regset)ctx->fixed_regset, IR_REGSET_FIXED), src_reg)) {
-		if (ctx->vregs[def]
-		 && ctx->live_intervals[ctx->vregs[def]]
-		 && ctx->live_intervals[ctx->vregs[def]]->stack_spill_pos != -1) {
-			ir_emit_store(ctx, type, def, src_reg);
-		}
-	} else {
-		ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	mask |= (IR_SHUFFLE_MASK(0) & 1);
+	mask |= (IR_SHUFFLE_MASK(1) & 1) << 1;

-		if (def_reg == IR_REG_NONE) {
-			/* op3 is used as a flag that the value is already stored in memory.
-			 * If op3 is set we don't have to store the value once again (in case of spilling)
-			 */
-			if (!insn->op3 || !ir_is_same_spill_slot(ctx, def, IR_MEM_BO(ctx->spill_base, insn->op3))) {
-				ir_emit_store(ctx, type, def, src_reg);
-			}
-		} else {
-			if (src_reg != def_reg) {
-				if (IR_IS_TYPE_INT(type)) {
-					ir_emit_mov(ctx, type, def_reg, src_reg);
-				} else {
-					IR_ASSERT(IR_IS_TYPE_FP(type));
-					ir_emit_fp_mov(ctx, type, def_reg, src_reg);
-				}
-			}
-			if (IR_REG_SPILLED(ctx->regs[def][0])
-			 && (!insn->op3 || !ir_is_same_spill_slot(ctx, def,  IR_MEM_BO(ctx->spill_base, insn->op3)))) {
-				ir_emit_store(ctx, type, def, def_reg);
-			}
-		}
+	if (def_reg == op2_reg && mask == 2) {
+		/* pass */
+	} else if (mask == 0) {
+		|	unpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (mask == 3) {
+		|	unpckhpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else {
+		|	shufpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), mask
 	}
 }

-static void ir_emit_rstore(ir_ctx *ctx, ir_ref ref, ir_insn *insn)
+static void ir_emit_shufps(ir_ctx *ctx, ir_ref def_reg, ir_reg op2_reg, uint32_t mask)
 {
-	ir_ref type = ctx->ir_base[insn->op2].type;
-	ir_reg op2_reg = ctx->regs[ref][2];
-	ir_reg dst_reg = insn->op3;
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;

-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, insn->op2);
-		}
-		if (op2_reg != dst_reg) {
-			if (IR_IS_TYPE_INT(type)) {
-				ir_emit_mov(ctx, type, dst_reg, op2_reg);
-			} else {
-				IR_ASSERT(IR_IS_TYPE_FP(type));
-				ir_emit_fp_mov(ctx, type, dst_reg, op2_reg);
-			}
-		}
+	if (def_reg == op2_reg && mask == 0xe4) {
+		/* pass */
+	} else if (def_reg == op2_reg && mask == 0x50) {
+		|	unpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (def_reg == op2_reg && mask == 0xfa) {
+		|	unpckhps xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (def_reg == op2_reg && mask == 0xee) {
+		|	movhlps xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (mask == 0x44) {
+		|	movlhps xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
 	} else {
-		ir_emit_load_ex(ctx, type, dst_reg, insn->op2, ref);
+		|	shufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), mask
 	}
 }

-static void ir_emit_alloca(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_blendps(ir_ctx *ctx, ir_ref def_reg, ir_reg op2_reg, ir_ref ref)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	int8_t *p;
+	uint32_t s, mask = 0;
+
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	mask |= (IR_SHUFFLE_MASK(0) & 4) >> 2;
+	mask |= (IR_SHUFFLE_MASK(1) & 4) >> 1;
+	mask |= (IR_SHUFFLE_MASK(2) & 4);
+	mask |= (IR_SHUFFLE_MASK(3) & 4) << 1;
+
+	|	blendps xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), mask
+}
+
+static void ir_emit_vector_shuffle_sse(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg tmp_reg = IR_REG_NONE;
+	uint32_t mask;

-	if (ctx->use_lists[def].count == 1) {
-		/* dead alloca */
-		return;
+	if (op2_reg == IR_REG_NONE && insn->op1 == insn->op2) {
+		op2_reg = op1_reg;
 	}
-	if (IR_IS_CONST_REF(insn->op2)) {
-		ir_insn *val = &ctx->ir_base[insn->op2];
-		int32_t size = val->val.i32;
-
-		IR_ASSERT(IR_IS_TYPE_INT(val->type));
-		IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-		IR_ASSERT(IR_IS_TYPE_UNSIGNED(val->type) || val->val.i64 >= 0);
-		IR_ASSERT(IR_IS_SIGNED_32BIT(val->val.i64));
-
-		/* Stack must be 16 byte aligned */
-		size = IR_ALIGNED_SIZE(size, 16);
-		|	ASM_REG_IMM_OP sub, IR_ADDR, IR_REG_RSP, size
-		if (!(ctx->flags & IR_USE_FRAME_POINTER)) {
-			ctx->call_stack_size += size;
-		}
-	} else {
-		int32_t alignment = 16;
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_type type = ctx->ir_base[insn->op2].type;
+	IR_ASSERT(def_reg != IR_REG_NONE);

-		IR_ASSERT(ctx->flags & IR_FUNCTION);
-		IR_ASSERT(ctx->flags & IR_USE_FRAME_POINTER);
-		IR_ASSERT(def_reg != IR_REG_NONE);
-		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, insn->op2);
-		}
-		if (def_reg != op2_reg) {
-			if (op2_reg != IR_REG_NONE) {
-				ir_emit_mov(ctx, type, def_reg, op2_reg);
+	switch (rule) {
+		default:
+			IR_ASSERT(0 && "NIY shuffle rule");
+		case IR_SHUFPD_11:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			ir_emit_shufpd(ctx, def_reg, def_reg, insn->op3);
+			break;
+		case IR_SHUFPD_22:
+			ir_emit_load_def_reg(ctx, type, def_reg, op2_reg, insn->op2);
+			ir_emit_shufpd(ctx, def_reg, def_reg, insn->op3);
+			break;
+		case IR_SHUFPD_12:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			ir_emit_shufpd(ctx, def_reg, op2_reg, insn->op3);
+			break;
+		case IR_SHUFPD_21:
+			ir_emit_load_def_reg(ctx, type, def_reg, op2_reg, insn->op2);
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			ir_emit_shufpd(ctx, def_reg, op1_reg, insn->op3);
+			break;
+		case IR_MOVSD_12:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			|	movsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+			break;
+		case IR_SHUFPS_11:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, def_reg, mask);
+			break;
+		case IR_SHUFPS_22:
+			ir_emit_load_def_reg(ctx, type, def_reg, op2_reg, insn->op2);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, def_reg, mask);
+			break;
+		case IR_SHUFPS_12:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, op2_reg, mask);
+			break;
+		case IR_SHUFPS_21:
+			ir_emit_load_def_reg(ctx, type, def_reg, op2_reg, insn->op2);
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, op1_reg, mask);
+			break;
+		case IR_SHUFPS_12_0:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_12_0(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, op2_reg, mask & 0xff);
+			ir_emit_shufps(ctx, def_reg, def_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_12_1:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_12_n(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, op2_reg, mask & 0xff);
+			ir_emit_shufps(ctx, def_reg, op1_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_12_2:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			op1_reg = IR_REG_NUM(op1_reg);
+			mask = ir_shufps_mask_12_n(ctx, insn->op3);
+			ir_emit_shufps(ctx, def_reg, op2_reg, mask & 0xff);
+			ir_emit_shufps(ctx, def_reg, op2_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_1_21:
+			tmp_reg = ctx->tmp_regs[def];
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			if (tmp_reg == op1_reg) {
+				mask = ir_shufps_mask_n_12(ctx, insn->op3);
+				ir_emit_shufps(ctx, tmp_reg, op2_reg, mask & 0xff);
 			} else {
-				ir_emit_load(ctx, type, def_reg, insn->op2);
+				ir_emit_load_def_reg(ctx, type, tmp_reg, op2_reg, insn->op2);
+				mask = ir_shufps_mask_n_21(ctx, insn->op3);
+				ir_emit_shufps(ctx, tmp_reg, def_reg, mask & 0xff);
 			}
-		}
-
-		|	ASM_REG_IMM_OP add, IR_ADDR, def_reg, (alignment-1)
-		|	ASM_REG_IMM_OP and, IR_ADDR, def_reg, ~(alignment-1)
-		|	ASM_REG_REG_OP sub, IR_ADDR, IR_REG_RSP, def_reg
+			ir_emit_shufps(ctx, def_reg, tmp_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_2_12:
+			tmp_reg = ctx->tmp_regs[def];
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+			ir_emit_load_def_reg(ctx, type, def_reg, op2_reg, insn->op2);
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (tmp_reg == op2_reg) {
+				mask = ir_shufps_mask_n_21(ctx, insn->op3);
+				ir_emit_shufps(ctx, tmp_reg, op1_reg, mask & 0xff);
+			} else {
+				ir_emit_load_def_reg(ctx, type, tmp_reg, op1_reg, insn->op1);
+				mask = ir_shufps_mask_n_12(ctx, insn->op3);
+				ir_emit_shufps(ctx, tmp_reg, def_reg, mask & 0xff);
+			}
+			ir_emit_shufps(ctx, def_reg, tmp_reg, mask >> 8);
+			break;
+		case IR_BLENDPS_12:
+			ir_emit_load_def_reg(ctx, type, def_reg, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			ir_emit_blendps(ctx, def_reg, op2_reg, insn->op3);
+			break;
 	}
-	if (def_reg != IR_REG_NONE) {
-		|	mov Ra(def_reg), Ra(IR_REG_RSP)
-		if (IR_REG_SPILLED(ctx->regs[def][0])) {
-			ir_emit_store(ctx, insn->type, def, def_reg);
-		}
-	} else {
-		ir_emit_store(ctx, IR_ADDR, def, IR_REG_STACK_POINTER);
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_afree(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vshufpd(ir_ctx *ctx, ir_ref def_reg, ir_reg op1_reg, ir_reg op2_reg, ir_ref ref)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	int8_t *p;
+	uint32_t s, mask = 0;

-	if (IR_IS_CONST_REF(insn->op2)) {
-		ir_insn *val = &ctx->ir_base[insn->op2];
-		int32_t size = val->val.i32;
-
-		IR_ASSERT(IR_IS_TYPE_INT(val->type));
-		IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-		IR_ASSERT(IR_IS_TYPE_UNSIGNED(val->type) || val->val.i64 > 0);
-		IR_ASSERT(IR_IS_SIGNED_32BIT(val->val.i64));
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	mask |= (IR_SHUFFLE_MASK(0) & 1);
+	mask |= (IR_SHUFFLE_MASK(1) & 1) << 1;

-		/* Stack must be 16 byte aligned */
-		size = IR_ALIGNED_SIZE(size, 16);
-		|	ASM_REG_IMM_OP add, IR_ADDR, IR_REG_RSP, size
-		if (!(ctx->flags & IR_USE_FRAME_POINTER)) {
-			ctx->call_stack_size -= size;
+	if (op1_reg == op2_reg && mask == 2) {
+		if (def_reg != op1_reg) {
+			|	vmovapd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
 		}
+		/* pass */
+	} else if (mask == 0) {
+		|	vunpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (mask == 3) {
+		|	vunpckhpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
 	} else {
-//		int32_t alignment = 16;
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_type type = ctx->ir_base[insn->op2].type;
-
-		IR_ASSERT(ctx->flags & IR_FUNCTION);
-		if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, insn->op2);
-		}
-
-		// TODO: alignment ???
-
-		|	ASM_REG_REG_OP add, IR_ADDR, IR_REG_RSP, op2_reg
+		|	vshufpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), mask
 	}
 }

-static void ir_emit_block_begin(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vshufps(ir_ctx *ctx, ir_ref def_reg, ir_reg op1_reg, ir_reg op2_reg, uint32_t mask)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-
-	if (ctx->use_lists[def].count == 1) {
-		/* dead load */
-		return;
-	}
-	|	mov Ra(def_reg), Ra(IR_REG_RSP)

-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, IR_ADDR, def, def_reg);
+	if (op1_reg == op2_reg && mask == 0xe4) {
+		if (def_reg != op1_reg) {
+			|	vmovaps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+		}
+	} else if (op1_reg == op2_reg && mask == 0x50) {
+		|	vunpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (op1_reg == op2_reg && mask == 0xfa) {
+		|	vunpckhps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (op1_reg == op2_reg && mask == 0xee) {
+		|	vmovhlps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else if (mask == 0x44) {
+		|	vmovlhps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+	} else {
+		|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), mask
 	}
 }

-static void ir_emit_block_end(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vblendps(ir_ctx *ctx, ir_ref def_reg, ir_reg op1_reg, ir_reg op2_reg, ir_ref ref)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg op2_reg = ctx->regs[def][2];
-
-	IR_ASSERT(op2_reg != IR_REG_NONE);
-	if (IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
-	}
-
-	|	mov Ra(IR_REG_RSP), Ra(op2_reg)
+	int8_t *p;
+	uint32_t s, mask = 0;
+
+	IR_ASSERT(IR_IS_CONST_REF(ref) && IR_IS_TYPE_VECTOR(ctx->ir_base[ref].type));
+	p = ir_long_const_ptr(ctx, ref);
+	s = ir_type_size[IR_VECTOR_BASE_TYPE(ctx->ir_base[ref].type)];
+	mask |= (IR_SHUFFLE_MASK(0) & 4) >> 2;
+	mask |= (IR_SHUFFLE_MASK(1) & 4) >> 1;
+	mask |= (IR_SHUFFLE_MASK(2) & 4);
+	mask |= (IR_SHUFFLE_MASK(3) & 4) << 1;
+
+	|	vblendps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), mask
 }

-static void ir_emit_frame_addr(ir_ctx *ctx, ir_ref def)
+static void ir_emit_vector_shuffle_avx(ir_ctx *ctx, ir_ref def, ir_insn *insn, uint32_t rule)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg tmp_reg = IR_REG_NONE;
+	uint32_t mask;

-	if (ctx->flags & IR_USE_FRAME_POINTER) {
-		|	mov Ra(def_reg), Ra(IR_REG_RBP)
-	} else {
-		|	lea Ra(def_reg), [Ra(IR_REG_RSP)+(ctx->stack_frame_size + ctx->call_stack_size)]
+	if (op2_reg == IR_REG_NONE && insn->op1 == insn->op2) {
+		op2_reg = op1_reg;
+	}
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	switch (rule) {
+		default:
+			IR_ASSERT(0 && "NIY shuffle rule");
+		case IR_SHUFPD_11:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			ir_emit_vshufpd(ctx, def_reg, op1_reg, op1_reg, insn->op3);
+			break;
+		case IR_SHUFPD_22:
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			ir_emit_vshufpd(ctx, def_reg, op2_reg, op2_reg, insn->op3);
+			break;
+		case IR_SHUFPD_12:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			ir_emit_vshufpd(ctx, def_reg, op1_reg, op2_reg, insn->op3);
+			break;
+		case IR_SHUFPD_21:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			ir_emit_vshufpd(ctx, def_reg, op2_reg, op1_reg, insn->op3);
+			break;
+		case IR_MOVSD_12:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			|	vmovsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+			break;
+		case IR_SHUFPS_11:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op1_reg, op1_reg, mask);
+			break;
+		case IR_SHUFPS_22:
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op2_reg, op2_reg, mask);
+			break;
+		case IR_SHUFPS_12:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op1_reg, op2_reg, mask);
+			break;
+		case IR_SHUFPS_21:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op2_reg, op1_reg, mask);
+			break;
+		case IR_SHUFPS_12_0:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_12_0(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op1_reg, op2_reg, mask & 0xff);
+			ir_emit_vshufps(ctx, def_reg, def_reg, def_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_12_1:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_12_n(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op1_reg, op2_reg, mask & 0xff);
+			ir_emit_vshufps(ctx, def_reg, def_reg, op1_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_12_2:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_12_n(ctx, insn->op3);
+			ir_emit_vshufps(ctx, def_reg, op1_reg, op2_reg, mask & 0xff);
+			ir_emit_vshufps(ctx, def_reg, def_reg, op2_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_1_21:
+			tmp_reg = ctx->tmp_regs[def];
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_n_12(ctx, insn->op3);
+			ir_emit_vshufps(ctx, tmp_reg, op1_reg, op2_reg, mask & 0xff);
+			ir_emit_vshufps(ctx, def_reg, op1_reg, tmp_reg, mask >> 8);
+			break;
+		case IR_SHUFPS_2_12:
+			tmp_reg = ctx->tmp_regs[def];
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			mask = ir_shufps_mask_n_12(ctx, insn->op3);
+			ir_emit_vshufps(ctx, tmp_reg, op1_reg, op2_reg, mask & 0xff);
+			ir_emit_vshufps(ctx, def_reg, op2_reg, tmp_reg, mask >> 8);
+			break;
+		case IR_BLENDPS_12:
+			op1_reg = ir_emit_load_reg(ctx, type, op1_reg, insn->op1);
+			op2_reg = ir_emit_load_reg(ctx, type, op2_reg, insn->op2);
+			ir_emit_vblendps(ctx, def_reg, op1_reg, op2_reg, insn->op3);
+			break;
 	}
+
 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, IR_ADDR, def, def_reg);
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_va_start(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vector_shuffle(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
-	const ir_call_conv_dsc *cc = data->ra_data.cc;
 	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t element_size;
+	uint32_t width;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg op3_reg = ctx->regs[def][3];
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);

-	if (!cc->sysv_varargs) {
-		ir_reg fp;
-		int arg_area_offset;
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_reg tmp_reg = ctx->regs[def][3];
-		int32_t offset;
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type) && IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op3].type));

-		IR_ASSERT(tmp_reg != IR_REG_NONE);
-		if (op2_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op2_reg)) {
-				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
-			}
-			offset = 0;
-		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
-			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
-		}
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	element_size = ir_type_size[element_type];
+	width = IR_VECTOR_SIZE(type);

-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			fp = IR_REG_FRAME_POINTER;
-			arg_area_offset = sizeof(void*) * 2 + ctx->param_stack_size;
-		} else {
-			fp = IR_REG_STACK_POINTER;
-			arg_area_offset = ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*) + ctx->param_stack_size;
-		}
-		|	lea Ra(tmp_reg), aword [Ra(fp)+arg_area_offset]
-		|	mov aword [Ra(op2_reg)+offset], Ra(tmp_reg)
+	IR_ASSERT(IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op1].type)
+		&& element_type == IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op1].type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(ctx->ir_base[insn->op2].type)
+		&& element_type == IR_VECTOR_BASE_TYPE(ctx->ir_base[insn->op2].type));
+
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, insn->op1);
+	}
+
+	if (insn->op1 == insn->op2) {
+		op2_reg = op1_reg;
 	} else {
-		IR_ASSERT(sizeof(void*) == 8);
-#ifdef IR_TARGET_X64
-|.if X64
-		ir_reg fp;
-		int reg_save_area_offset;
-		int overflow_arg_area_offset;
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_reg tmp_reg = ctx->regs[def][3];
-		bool have_reg_save_area = 0;
-		int32_t offset;
+		IR_ASSERT(op2_reg != IR_REG_NONE);
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			ir_emit_load(ctx, type, op2_reg, insn->op2);
+		}
+	}

-		IR_ASSERT(tmp_reg != IR_REG_NONE);
-		if (op2_reg != IR_REG_NONE) {
+	if (1) {
+		/* shuffle through stack memory */
+		ir_type mask_type = ctx->ir_base[insn->op3].type;
+		ir_type mask_element_type = IR_VECTOR_BASE_TYPE(mask_type);
+		uint32_t mask_element_size = ir_type_size[mask_element_type];
+		uint32_t mask_len = IR_VECTOR_LENGTH(mask_type);
+		int dst_offset = -width;
+		int src_offset = dst_offset;
+		uint32_t mod = 0;
+		ir_reg tmp_reg = ctx->tmp_regs[def];
+
+		if (insn->op1 != insn->op2) {
 			if (IR_REG_SPILLED(op2_reg)) {
 				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+				ir_emit_load(ctx, type, op2_reg, insn->op2);
+			}
+			src_offset -= IR_VECTOR_SIZE(ctx->ir_base[insn->op2].type);
+			mod += IR_VECTOR_LENGTH(ctx->ir_base[insn->op2].type);
+			ir_emit_store_mem_fp(ctx, ctx->ir_base[insn->op2].type,
+				IR_MEM(IR_REG_RSP, src_offset, IR_REG_NONE, 1), op2_reg);
+		}
+		src_offset -= IR_VECTOR_SIZE(ctx->ir_base[insn->op1].type);
+		mod += IR_VECTOR_LENGTH(ctx->ir_base[insn->op1].type);
+		ir_emit_store_mem_fp(ctx, ctx->ir_base[insn->op1].type,
+			IR_MEM(IR_REG_RSP, src_offset, IR_REG_NONE, 1), op1_reg);
+
+		if (IR_IS_CONST_REF(insn->op3)) {
+			void *ptr = ir_long_const_ptr(ctx, insn->op3);
+			uint32_t i, j;
+
+			for (i = 0; i < mask_len; i++) {
+				ir_mem src, dst;
+
+				if (mask_element_size == 1) {
+					j = *(uint8_t*)ptr;
+					ptr = ((uint8_t*)ptr) + 1;
+				} else if (mask_element_size == 2) {
+					j = *(uint16_t*)ptr;
+					ptr = ((uint16_t*)ptr) + 1;
+				} else if (mask_element_size == 4) {
+					j = *(uint32_t*)ptr;
+					ptr = ((uint32_t*)ptr) + 1;
+				} else if (mask_element_size == 8) {
+					j = *(uint64_t*)ptr;
+					ptr = ((uint64_t*)ptr) + 1;
+				} else {
+					IR_ASSERT(0);
+					j = 0;
+				}
+				src = IR_MEM(IR_REG_RSP, src_offset + (j % mod) * element_size, IR_REG_NONE, 1);
+				dst = IR_MEM(IR_REG_RSP, dst_offset + i * element_size, IR_REG_NONE, 1);
+#if IR_X86_I64
+				if (element_type == IR_I64 || element_type == IR_U64) {
+					ir_emit_load_mem_int(ctx, IR_U32, tmp_reg, src);
+					ir_emit_store_mem_int(ctx, IR_U32, dst, tmp_reg);
+					src = IR_MEM_I64_HI(src);
+					dst = IR_MEM_I64_HI(dst);
+					ir_emit_load_mem_int(ctx, IR_U32, tmp_reg, src);
+					ir_emit_store_mem_int(ctx, IR_U32, dst, tmp_reg);
+				} else
+#endif
+				if (IR_IS_TYPE_INT(element_type)) {
+					ir_emit_load_mem_int(ctx, element_type, tmp_reg, src);
+					ir_emit_store_mem_int(ctx, element_type, dst, tmp_reg);
+				} else {
+					ir_emit_load_mem_fp(ctx, element_type, tmp_reg, src);
+					ir_emit_store_mem_fp(ctx, element_type, dst, tmp_reg);
+				}
 			}
-			offset = 0;
 		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
-			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
-		}
+			int mask_offset = src_offset - IR_VECTOR_SIZE(mask_type);
+			uint32_t i;
+			ir_reg idx_reg;
+
+			if (IR_IS_TYPE_FP(element_type)) {
+				idx_reg = IR_REG_RCX; // TODO: hardcoded registetrs ???
+#if IR_X86_I64
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				idx_reg = IR_REG_RCX; // TODO: hardcoded registetrs ???
+#endif
+			} else {
+				idx_reg = tmp_reg;
+			}

-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			fp = IR_REG_FRAME_POINTER;
-			reg_save_area_offset = -(ctx->stack_frame_size - ctx->locals_area_size);
-			overflow_arg_area_offset = sizeof(void*) * 2 + ctx->param_stack_size;
-		} else {
-			fp = IR_REG_STACK_POINTER;
-			reg_save_area_offset = ctx->locals_area_size + ctx->call_stack_size;
-			overflow_arg_area_offset = ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*) + ctx->param_stack_size;
+			IR_ASSERT(op3_reg != IR_REG_NONE);
+			if (IR_REG_SPILLED(op3_reg)) {
+				op3_reg = IR_REG_NUM(op3_reg);
+				ir_emit_load(ctx, type, op3_reg, insn->op3);
+			}
+			ir_emit_store_mem_fp(ctx, mask_type, IR_MEM(IR_REG_RSP, mask_offset, IR_REG_NONE, 1), op3_reg);
+
+			for (i = 0; i < mask_len; i++) {
+				ir_mem src, dst;
+
+				src = IR_MEM(IR_REG_RSP, mask_offset + i * mask_element_size, IR_REG_NONE, 1);
+#if IR_X86_I64
+				if (mask_element_type == IR_I64 || mask_element_type == IR_U64) {
+					/* ignore the high part of 64-bit index */
+					ir_emit_load_mem_int(ctx, IR_U32, idx_reg, src);
+				} else
+#endif
+				ir_emit_load_mem_int(ctx, mask_element_type, idx_reg, src);
+				IR_ASSERT(mod != 0 && ((mod - 1) & mod) == 0);
+				|	and Ra(idx_reg), (mod-1)
+				src = IR_MEM(IR_REG_RSP, src_offset, idx_reg, element_size);
+				dst = IR_MEM(IR_REG_RSP, dst_offset + i * element_size, IR_REG_NONE, 1);
+#if IR_X86_I64
+				if (element_type == IR_I64 || element_type == IR_U64) {
+					ir_emit_load_mem_int(ctx, IR_U32, tmp_reg, src);
+					ir_emit_store_mem_int(ctx, IR_U32, dst, tmp_reg);
+					src = IR_MEM_I64_HI(src);
+					dst = IR_MEM_I64_HI(dst);
+					ir_emit_load_mem_int(ctx, IR_U32, tmp_reg, src);
+					ir_emit_store_mem_int(ctx, IR_U32, dst, tmp_reg);
+				} else
+#endif
+				if (IR_IS_TYPE_INT(element_type)) {
+					ir_emit_load_mem_int(ctx, element_type, tmp_reg, src);
+					ir_emit_store_mem_int(ctx, element_type, dst, tmp_reg);
+				} else {
+					ir_emit_load_mem_fp(ctx, element_type, tmp_reg, src);
+					ir_emit_store_mem_fp(ctx, element_type, dst, tmp_reg);
+				}
+			}
 		}

-		if ((ctx->flags2 & (IR_HAS_VA_ARG_GP|IR_HAS_VA_COPY)) && ctx->gp_reg_params < cc->int_param_regs_count) {
-			|	lea Ra(tmp_reg), aword [Ra(fp)+reg_save_area_offset]
-			have_reg_save_area = 1;
-			/* Set va_list.gp_offset */
-			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], sizeof(void*) * ctx->gp_reg_params
-		} else {
-			reg_save_area_offset -= sizeof(void*) * cc->int_param_regs_count;
-			/* Set va_list.gp_offset */
-			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], sizeof(void*) * cc->int_param_regs_count
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, IR_MEM(IR_REG_RSP, dst_offset, IR_REG_NONE, 1));
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
+	}
+}
+
+static void ir_emit_vector_op(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+{
+	ir_backend_data *data = ctx->data;
+	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(type == ctx->ir_base[insn->op1].type);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, type, op1_reg, insn->op1);
 		}
-		if ((ctx->flags2 & (IR_HAS_VA_ARG_FP|IR_HAS_VA_COPY)) && ctx->fp_reg_params < cc->fp_param_regs_count) {
-			if (!have_reg_save_area) {
-				|	lea Ra(tmp_reg), aword [Ra(fp)+reg_save_area_offset]
-				have_reg_save_area = 1;
-			}
-			/* Set va_list.fp_offset */
-			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], sizeof(void*) * cc->int_param_regs_count + 16 * ctx->fp_reg_params
+
+		switch (insn->op) {
+			default:
+			case IR_ABS:
+				IR_ASSERT(0 && "NIY unary op");
+				break;
+			case IR_NEG:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (ctx->mflags & IR_X86_AVX) {
+						if (width <= 16) {
+							IR_ASSERT(def_reg != op1_reg);
+							|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, width, def_reg, def_reg, op1_reg
+						} else {
+							IR_ASSERT(width == 32);
+							if (ctx->mflags & IR_X86_AVX2) {
+								IR_ASSERT(def_reg != op1_reg);
+								|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+								|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, width, def_reg, def_reg, op1_reg
+							} else {
+								ir_reg tmp1_reg = ctx->regs[def][2];
+								ir_reg tmp2_reg = ctx->regs[def][3];
+
+								|	vpxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, 16, tmp2_reg, tmp1_reg, tmp2_reg
+								|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, 16, def_reg, tmp1_reg, op1_reg
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(def_reg != op1_reg);
+						IR_ASSERT(width <= 16);
+						|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	ASM_SSE_INT_VEC_REG_REG_OP psub, element_type, def_reg, op1_reg
+					}
+				} else {
+					IR_ASSERT(def_reg != op1_reg);
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					if (ctx->mflags & IR_X86_AVX) {
+						if (width <= 16) {
+							if (element_type == IR_FLOAT) {
+								|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(element_type == IR_DOUBLE);
+								|	vxorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							}
+						} else {
+							IR_ASSERT(width == 32);
+							if (element_type == IR_FLOAT) {
+								|	vxorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(element_type == IR_DOUBLE);
+								|	vxorpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							}
+						}
+						|	ASM_AVX_FP_VEC_REG_REG_REG_OP vsubp, element_type, width, def_reg, def_reg, op1_reg
+					} else {
+						IR_ASSERT(width <= 16);
+						if (element_type == IR_FLOAT) {
+							|	xorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(element_type == IR_DOUBLE);
+							|	xorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						}
+						|	ASM_SSE_FP_VEC_REG_REG_OP subp, element_type, def_reg, op1_reg
+					}
+				}
+				break;
+			case IR_NOT:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				IR_ASSERT(def_reg != op1_reg);
+				if (ctx->mflags & IR_X86_AVX) {
+					if (width <= 16) {
+						|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(width == 32);
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vxorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							|	vcmpps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 0xf
+							|	vxorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						}
+					}
+				} else {
+					IR_ASSERT(width <= 16);
+					|	pcmpeqd	xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				break;
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;
+
+		IR_ASSERT(width == 16);
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
 		} else {
-			/* Set va_list.fp_offset */
-			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], sizeof(void*) * cc->int_param_regs_count + 16 * cc->fp_param_regs_count
+			mem = ir_ref_spill_slot(ctx, insn->op1);
 		}
-		if (have_reg_save_area) {
-			/* Set va_list.reg_save_area */
-			|	mov qword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))], Ra(tmp_reg)
+
+		switch (insn->op) {
+			default:
+			case IR_ABS:
+				IR_ASSERT(0 && "NIY unary op");
+				break;
+			case IR_NEG:
+				IR_ASSERT(def_reg != op1_reg);
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (ctx->mflags & IR_X86_AVX) {
+						if (width == 16) {
+							|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, width, def_reg, def_reg, mem
+						} else {
+							IR_ASSERT(width == 32);
+							if (ctx->mflags & IR_X86_AVX2) {
+								|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+								|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, width, def_reg, def_reg, mem
+							} else {
+								ir_reg tmp_reg = ctx->regs[def][2];
+
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, 16, def_reg, tmp_reg, mem
+								|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, 16, tmp_reg, tmp_reg, IR_MEM_ADD(mem, 16)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(width == 16);
+						|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	ASM_SSE_INT_VEC_REG_MEM_OP psub, element_type, def_reg, mem
+					}
+				} else {
+					if (ctx->mflags & IR_X86_AVX) {
+						if (width == 16) {
+							if (element_type == IR_FLOAT) {
+								|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(element_type == IR_DOUBLE);
+								|	vxorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							}
+						} else {
+							IR_ASSERT(width == 32);
+							if (element_type == IR_FLOAT) {
+								|	vxorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(element_type == IR_DOUBLE);
+								|	vxorpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							}
+						}
+						|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vsubp, element_type, width, def_reg, def_reg, mem
+					} else {
+						IR_ASSERT(width == 16);
+						if (element_type == IR_FLOAT) {
+							|	vxorps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(element_type == IR_DOUBLE);
+							|	vxorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						}
+						|	ASM_SSE_FP_VEC_REG_MEM_OP subp, element_type, def_reg, mem
+					}
+				}
+				break;
+			case IR_NOT:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				IR_ASSERT(def_reg != op1_reg);
+				if (ctx->mflags & IR_X86_AVX) {
+					if (width == 16) {
+						|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	ASM_TXT_TXT_TMEM_OP vpxor, xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else {
+						IR_ASSERT(width == 32);
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							|	ASM_TXT_TXT_TMEM_OP vpxor, ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword, mem
+						} else {
+							|	vxorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							|	vcmpps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 0xf
+							|	ASM_TXT_TXT_TMEM_OP vxorps, ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword, mem
+						}
+					}
+				} else {
+					IR_ASSERT(width == 16);
+					|	pcmpeqd	xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	ASM_TXT_TMEM_OP pxor, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+				}
+				break;
 		}
-		|	lea Ra(tmp_reg), aword [Ra(fp)+overflow_arg_area_offset]
-		/* Set va_list.overflow_arg_area */
-		|	mov qword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
-|.endif
-#endif
+	}
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_va_copy(ir_ctx *ctx, ir_ref def, ir_insn *insn)
+static void ir_emit_vector_binop_sse2(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
-	const ir_call_conv_dsc *cc = data->ra_data.cc;
 	dasm_State **Dst = &data->dasm_state;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp_reg = ctx->regs[def][3];
+	int shift_count = 0;

-	if (!cc->sysv_varargs) {
-		ir_reg tmp_reg = ctx->regs[def][1];
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_reg op3_reg = ctx->regs[def][3];
-		int32_t op2_offset, op3_offset;
+	(void)width;

-		IR_ASSERT(tmp_reg != IR_REG_NONE);
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+		type = ctx->ir_base[op1].type;
+		IR_ASSERT(type == ctx->ir_base[op2].type);
+	} else if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+		IR_ASSERT(type == ctx->ir_base[op1].type);
+		IR_ASSERT(insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR);
 		if (op2_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op2_reg)) {
 				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+				ir_emit_load(ctx, ctx->ir_base[op2].type, op2_reg, op2);
 			}
-			op2_offset = 0;
-		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
-			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			op2_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
-		}
-		if (op3_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op3_reg)) {
-				op3_reg = IR_REG_NUM(op3_reg);
-				ir_emit_load(ctx, IR_ADDR, op3_reg, insn->op3);
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+#if IR_X86_I64
+			if (ctx->ir_base[op2].type == IR_I64 || ctx->ir_base[op2].type == IR_U64) {
+				/* ignore the high part of the register */
+				op2_reg = IR_REG_I64_LO(op2_reg);
 			}
-			op3_offset = 0;
+#endif
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vmovd xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op2_reg)
+			} else {
+				|	movd xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op2_reg)
+			}
+			op2_reg = tmp_reg;
 		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op3) == IR_STATIC_ALLOCA);
-			op3_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			op3_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op3]);
+			IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+			shift_count = ctx->ir_base[insn->op2].val.i32;
 		}
-		|	mov Ra(tmp_reg), aword [Ra(op3_reg)+op3_offset]
-		|	mov aword [Ra(op2_reg)+op2_offset], Ra(tmp_reg)
 	} else {
-		IR_ASSERT(sizeof(void*) == 8);
-#ifdef IR_TARGET_X64
-|.if X64
-		ir_reg tmp_reg = ctx->regs[def][1];
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_reg op3_reg = ctx->regs[def][3];
-		int32_t op2_offset, op3_offset;
+		IR_ASSERT(type == ctx->ir_base[op1].type && type == ctx->ir_base[op2].type);
+	}

-		IR_ASSERT(tmp_reg != IR_REG_NONE);
-		if (op2_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op2_reg)) {
-				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
-			}
-			op2_offset = 0;
-		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
-			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			op2_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);
+
+	IR_ASSERT(width <= 16);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+		if (op1 == op2) {
+			op2_reg = op1_reg;
 		}
-		if (op3_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op3_reg)) {
-				op3_reg = IR_REG_NUM(op3_reg);
-				ir_emit_load(ctx, IR_ADDR, op3_reg, insn->op3);
+	}
+
+	if (insn->op < IR_LT || insn->op > IR_UGT) {
+		if (def_reg != op1_reg) {
+			if (op1_reg != IR_REG_NONE) {
+				ir_emit_fp_mov(ctx, type, def_reg, op1_reg);
+			} else {
+				ir_emit_load(ctx, type, def_reg, op1);
 			}
-			op3_offset = 0;
-		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op3) == IR_STATIC_ALLOCA);
-			op3_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			op3_offset = ir_local_offset(ctx, &ctx->ir_base[insn->op3]);
 		}
-		|	mov Rd(tmp_reg), dword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))]
-		|	mov dword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], Rd(tmp_reg)
-		|	mov Rd(tmp_reg), dword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))]
-		|	mov aword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], Ra(tmp_reg)
-		|	mov Ra(tmp_reg), aword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
-		|	mov aword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
-		|	mov Ra(tmp_reg), aword [Ra(op3_reg)+(op3_offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))]
-		|	mov aword [Ra(op2_reg)+(op2_offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))], Ra(tmp_reg)
-|.endif
-#endif
 	}
-}
-
-static void ir_emit_va_arg(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	const ir_call_conv_dsc *cc = data->ra_data.cc;
-	dasm_State **Dst = &data->dasm_state;

-	if (!cc->sysv_varargs) {
-		ir_type type = insn->type;
-		ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_reg tmp_reg = ctx->regs[def][3];
-		int32_t offset;
-
-		IR_ASSERT((def_reg != IR_REG_NONE || ctx->use_lists[def].count == 1) && tmp_reg != IR_REG_NONE);
-		if (op2_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op2_reg)) {
-				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load(ctx, type, op2_reg, op2);
 			}
-			offset = 0;
-		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
-			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
 		}
-		|	mov Ra(tmp_reg), aword [Ra(op2_reg)+offset]
-		if (!cc->pass_struct_by_val || !insn->op3) {
-			if (def_reg != IR_REG_NONE) {
-				ir_emit_load_mem(ctx, type, def_reg, IR_MEM_B(tmp_reg));
-			}
-			|	add Ra(tmp_reg), IR_MAX(ir_type_size[type], sizeof(void*))
-		} else {
-			int size = IR_VA_ARG_SIZE(insn->op3);
-
-			if (def_reg != IR_REG_NONE) {
-				IR_ASSERT(type == IR_ADDR);
-				int align = IR_VA_ARG_ALIGN(insn->op3);
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_SSE_INT_VEC_REG_REG_OP padd, element_type, def_reg, op2_reg
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_OP addp, element_type, def_reg, op2_reg
+				}
+				break;
+			case IR_SUB:
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_SSE_INT_VEC_REG_REG_OP psub, element_type, def_reg, op2_reg
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_OP subp, element_type, def_reg, op2_reg
+				}
+				break;
+			case IR_MUL:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pmullw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pmulld xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pmuludq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 232
+							|	pmuludq xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+							|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_OP mulp, element_type, def_reg, op2_reg
+				}
+				break;
+			case IR_DIV:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_SSE_FP_VEC_REG_REG_OP divp, element_type, def_reg, op2_reg
+				break;
+			case IR_MIN:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8) {
+						|	pminsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_U8) {
+						|	pminub xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16) {
+						|	pminsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_U16) {
+						|	pminuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32) {
+						|	pminsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_U32) {
+						|	pminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_OP minp, element_type, def_reg, op2_reg
+				}
+				break;
+			case IR_MAX:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8) {
+						|	pmaxsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_U8) {
+						|	pmaxub xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16) {
+						|	pmaxsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_U16) {
+						|	pmaxuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32) {
+						|	pmaxsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_U32) {
+						|	pmaxud xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_OP maxp, element_type, def_reg, op2_reg
+				}
+				break;
+			case IR_AND:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				break;
+			case IR_OR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				break;
+			case IR_XOR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				break;
+			case IR_EQ:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pshufd	xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 177
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0);
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 0
+				}
+				break;
+			case IR_NE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pshufd	xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 177
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0);
+					}
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	ASM_SSE_INT_VEC_REG_REG_OP pcmpeq, element_type, def_reg, tmp_reg
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 4
+				}
+				break;
+			case IR_LT:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (def_reg != op2_reg) {
+						IR_ASSERT(def_reg != op1_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					}
+					if ((ctx->mflags & IR_X86_SSE42) || (element_type != IR_I64 && element_type != IR_U64)) {
+						/* pcmpgtq requires SSE4.2 */
+						|	ASM_SSE_INT_VEC_REG_REG_OP pcmpgt, element_type, def_reg, op1_reg
+					} else {
+						ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+						|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+						|	psrlq xmm(tmp_reg-IR_REG_FP_FIRST), 32
+						|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+						|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+						|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					if (def_reg != op1_reg) {
+						IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 1
+				}
+				break;
+			case IR_GT:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (def_reg != op1_reg) {
+						IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if ((ctx->mflags & IR_X86_SSE42) || (element_type != IR_I64 && element_type != IR_U64)) {
+						/* pcmpgtq requires SSE4.2 */
+						|	ASM_SSE_INT_VEC_REG_REG_OP pcmpgt, element_type, def_reg, op2_reg
+					} else {
+						ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+						|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+						|	psrlq xmm(tmp_reg-IR_REG_FP_FIRST), 32
+						|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+						|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+						|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+						|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					if (def_reg != op2_reg) {
+						IR_ASSERT(def_reg != op1_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					}
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op1_reg, 1
+				}
+				break;
+			case IR_LE:
+				if (def_reg != op1_reg) {
+					IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpgtb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pminsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32|| element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpgtd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						if (ctx->mflags & IR_X86_SSE42) {
+							|	pcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	psrlq xmm(tmp_reg-IR_REG_FP_FIRST), 32
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+							|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 2
+				}
+				break;
+			case IR_GE:
+				if (def_reg != op2_reg) {
+					IR_ASSERT(def_reg != op1_reg || op1_reg == op2_reg);
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				}
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpgtb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pminsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32|| element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpgtd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						if (ctx->mflags & IR_X86_SSE42) {
+							|	pcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	psrlq xmm(tmp_reg-IR_REG_FP_FIRST), 32
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+							|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op1_reg, 2
+				}
+				break;
+			case IR_ULT:
+				if (def_reg != op2_reg) {
+					IR_ASSERT(def_reg != op1_reg || op1_reg == op2_reg);
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				}
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	psubusb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	psubusw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32|| element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						 } else {
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						 }
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE42) {
+							|	pcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+							|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op1_reg, 6
+				}
+				break;
+			case IR_UGT:
+				if (def_reg != op1_reg) {
+					IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	psubusb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	psubusw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32|| element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE42) {
+							|	pcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+							|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 6
+				}
+				break;
+			case IR_ULE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (def_reg != op1_reg) {
+						IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	pminub xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpeqw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllw xmm(tmp_reg-IR_REG_FP_FIRST), 15
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpgtw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I32|| element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						if (ctx->mflags & IR_X86_SSE42) {
+							|	pcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	pcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+							|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					if (def_reg != op2_reg) {
+						IR_ASSERT(def_reg != op1_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					}
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op1_reg, 5
+				}
+				break;
+			case IR_UGE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (def_reg != op2_reg) {
+						IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					}
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	pminub xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpeqw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllw xmm(tmp_reg-IR_REG_FP_FIRST), 15
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpgtw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I32|| element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllw xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						if (ctx->mflags & IR_X86_SSE42) {
+							|	pcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	psllq xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	pcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							ir_reg tmp2_reg = ctx->tmp_regs[def];
+
+							|	pcmpeqd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pslld xmm(tmp_reg-IR_REG_FP_FIRST), 31
+							|	pxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	movdqa xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	pcmpgtd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 160
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 245
+							|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					if (def_reg != op1_reg) {
+						IR_ASSERT(def_reg != op2_reg || op1_reg == op2_reg);
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 5
+				}
+				break;
+			case IR_ORDERED:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 7
+				break;
+			case IR_UNORDERED:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_SSE_FP_VEC_REG_REG_TXT_OP cmpp, element_type, def_reg, op2_reg, 3
+				break;
+			case IR_SHL:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (element_type == IR_I16 || element_type == IR_U16) {
+					|	psllw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					|	pslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					|	psllq xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(0 && "unsupported vector type");
+				}
+				break;
+			case IR_SHR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (element_type == IR_I16 || element_type == IR_U16) {
+					|	psrlw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					|	psrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					|	psrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(0 && "unsupported vector type");
+				}
+				break;
+			case IR_SAR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (element_type == IR_I16 || element_type == IR_U16) {
+					|	psraw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					|	psrad xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(0 && "unsupported vector type");
+				}
+				break;
+		}
+	} else if (IR_IS_CONST_REF(op2)) {
+		int label = ir_get_const_label(ctx, op2);

-				if (align > (int)sizeof(void*)) {
-					|	add Ra(tmp_reg), (align-1)
-					|	and Ra(tmp_reg), ~(align-1)
+		IR_ASSERT(width == 16);
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_SSE_INT_VEC_REG_TXT_OP padd, element_type, def_reg, [=>label]
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_TXT_OP addp, element_type, def_reg, [=>label]
 				}
-				|	mov Ra(def_reg), Ra(tmp_reg)
-			}
-			|	add Ra(tmp_reg), IR_ALIGNED_SIZE(size, sizeof(void*))
-		}
-		|	mov aword [Ra(op2_reg)+offset], Ra(tmp_reg)
-		if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
-			ir_emit_store(ctx, type, def, def_reg);
+				break;
+			case IR_SUB:
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_SSE_INT_VEC_REG_TXT_OP psub, element_type, def_reg, [=>label]
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_TXT_OP subp, element_type, def_reg, [=>label]
+				}
+				break;
+			case IR_MUL:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pmullw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pmulld xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	pmuludq xmm(tmp_reg-IR_REG_FP_FIRST), [=>label]
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 232
+							|	pmuludq xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+							|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_TXT_OP mulp, element_type, def_reg, [=>label]
+				}
+				break;
+			case IR_DIV:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_SSE_FP_VEC_REG_TXT_OP divp, element_type, def_reg, [=>label]
+				break;
+			case IR_MIN:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8) {
+						|	pminsb xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_U8) {
+						|	pminub xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I16) {
+						|	pminsw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_U16) {
+						|	pminuw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I32) {
+						|	pminsd xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_U32) {
+						|	pminud xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_TXT_OP minp, element_type, def_reg, [=>label]
+				}
+				break;
+			case IR_MAX:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8) {
+						|	pmaxsb xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_U8) {
+						|	pmaxub xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I16) {
+						|	pmaxsw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_U16) {
+						|	pmaxuw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I32) {
+						|	pmaxsd xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_U32) {
+						|	pmaxud xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_TXT_OP maxp, element_type, def_reg, [=>label]
+				}
+				break;
+			case IR_AND:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	pand xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+				break;
+			case IR_OR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	por xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+				break;
+			case IR_XOR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	pxor xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+				break;
+			case IR_EQ:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+							|	pshufd	xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 177
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0);
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+//					|	ASM_SSE_FP_VEC_REG_TXT_TXT_OP cmpp, element_type, def_reg, [=>label], 0
+					IR_ASSERT(0 && "DynAsm limitation (IP-relative address followed by immediate operand) ???");
+				}
+				break;
+			case IR_NE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	pcmpeqb xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	pcmpeqw xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	pcmpeqq xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							|	pcmpeqd xmm(def_reg-IR_REG_FP_FIRST), [=>label]
+							|	pshufd	xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 177
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0);
+					}
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	ASM_SSE_INT_VEC_REG_REG_OP pcmpeq, element_type, def_reg, tmp_reg
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+//					|	ASM_SSE_FP_VEC_REG_TXT_TXT_OP cmpp, element_type, def_reg, [=>label], 4
+					IR_ASSERT(0 && "DynAsm limitation (IP-relative address followed by immediate operand) ???");
+				}
+				break;
+			case IR_SHL:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (element_type == IR_I16 || element_type == IR_U16) {
+					|	psllw xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					|	pslld xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					|	psllq xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else {
+					IR_ASSERT(0 && "unsupported vector type");
+				}
+				break;
+			case IR_SHR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (element_type == IR_I16 || element_type == IR_U16) {
+					|	psrlw xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					|	psrld xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else if (element_type == IR_I64 || element_type == IR_U64) {
+					|	psrlq xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else {
+					IR_ASSERT(0 && "unsupported vector type");
+				}
+				break;
+			case IR_SAR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (element_type == IR_I16 || element_type == IR_U16) {
+					|	psraw xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else if (element_type == IR_I32 || element_type == IR_U32) {
+					|	psrad xmm(def_reg-IR_REG_FP_FIRST), shift_count
+				} else {
+					IR_ASSERT(0 && "unsupported vector type");
+				}
+				break;
 		}
 	} else {
-		IR_ASSERT(sizeof(void*) == 8);
-#ifdef IR_TARGET_X64
-|.if X64
-		ir_type type = insn->type;
-		ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-		ir_reg op2_reg = ctx->regs[def][2];
-		ir_reg tmp_reg = ctx->regs[def][3];
-		int32_t offset;
+		ir_mem mem;

-		IR_ASSERT((def_reg != IR_REG_NONE || ctx->use_lists[def].count == 1) && tmp_reg != IR_REG_NONE);
-		if (op2_reg != IR_REG_NONE) {
-			if (IR_REG_SPILLED(op2_reg)) {
-				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
-			}
-			offset = 0;
+		IR_ASSERT(width == 16);
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
 		} else {
-			IR_ASSERT(ir_rule(ctx, insn->op2) == IR_STATIC_ALLOCA);
-			op2_reg = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-			offset = ir_local_offset(ctx, &ctx->ir_base[insn->op2]);
+			mem = ir_ref_spill_slot(ctx, op2);
 		}
-		if (insn->op3) {
-			/* long struct arguemnt */
-			IR_ASSERT(type == IR_ADDR);
-			int align = IR_VA_ARG_ALIGN(insn->op3);
-			int size = IR_VA_ARG_SIZE(insn->op3);
-
-			|	mov Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
-			if (align > (int)sizeof(void*)) {
-				|	add Ra(tmp_reg), (align-1)
-				|	and Ra(tmp_reg), ~(align-1)
-			}
-			if (def_reg != IR_REG_NONE) {
-				|	mov Ra(def_reg), Ra(tmp_reg)
-			}
-			|	add Ra(tmp_reg), IR_ALIGNED_SIZE(size, sizeof(void*))
-			|	mov aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
-		} else if (IR_IS_TYPE_INT(type)) {
-			|	mov Rd(tmp_reg), dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))]
-			|	cmp Rd(tmp_reg), sizeof(void*) * cc->int_param_regs_count
-			|	jge >1
-			|	add Rd(tmp_reg), sizeof(void*)
-			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, gp_offset))], Rd(tmp_reg)
-			|	add Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))]
-			|	jmp >2
-			|1:
-			|	mov Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
-			|	add Ra(tmp_reg), sizeof(void*)
-			|	mov aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
-			|2:
-			if (def_reg != IR_REG_NONE) {
-				if (ir_type_size[type] == 8) {
-					|	mov Rq(def_reg), qword [Ra(tmp_reg)-sizeof(void*)]
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_SSE_INT_VEC_REG_MEM_OP padd, element_type, def_reg, mem
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_OP addp, element_type, def_reg, mem
+				}
+				break;
+			case IR_SUB:
+				if (IR_IS_TYPE_INT(element_type)) {
+					|	ASM_SSE_INT_VEC_REG_MEM_OP psub, element_type, def_reg, mem
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_OP subp, element_type, def_reg, mem
+				}
+				break;
+			case IR_MUL:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	ASM_TXT_TMEM_OP pmullw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	ASM_TXT_TMEM_OP pmulld, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+						} else {
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 245
+							|	ASM_TXT_TMEM_OP pmuludq, xmm(tmp_reg-IR_REG_FP_FIRST), oword, mem
+							|	pshufd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 232
+							|	ASM_TXT_TMEM_OP pmuludq, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+							|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+							|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_OP mulp, element_type, def_reg, mem
+				}
+				break;
+			case IR_DIV:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_SSE_FP_VEC_REG_MEM_OP divp, element_type, def_reg, mem
+				break;
+			case IR_MIN:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8) {
+						|	ASM_TXT_TMEM_OP pminsb, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_U8) {
+						|	ASM_TXT_TMEM_OP pminub, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I16) {
+						|	ASM_TXT_TMEM_OP pminsw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_U16) {
+						|	ASM_TXT_TMEM_OP pminuw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I32) {
+						|	ASM_TXT_TMEM_OP pminsd, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_U32) {
+						|	ASM_TXT_TMEM_OP pminud, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_OP minp, element_type, def_reg, mem
+				}
+				break;
+			case IR_MAX:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8) {
+						|	ASM_TXT_TMEM_OP pmaxsb, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_U8) {
+						|	ASM_TXT_TMEM_OP pmaxub, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I16) {
+						|	ASM_TXT_TMEM_OP pmaxsw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_U16) {
+						|	ASM_TXT_TMEM_OP pmaxuw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I32) {
+						|	ASM_TXT_TMEM_OP pmaxsd, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_U32) {
+						|	ASM_TXT_TMEM_OP pmaxud, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
 				} else {
-					|	mov Rd(def_reg), dword [Ra(tmp_reg)-sizeof(void*)]
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_OP maxp, element_type, def_reg, mem
 				}
-			}
-		} else {
-			|	mov Rd(tmp_reg), dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))]
-			|	cmp Rd(tmp_reg), sizeof(void*) * cc->int_param_regs_count + 16 * cc->fp_param_regs_count
-			|	jge >1
-			|	add Rd(tmp_reg), 16
-			|	mov dword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, fp_offset))], Rd(tmp_reg)
-			|	add Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, reg_save_area))]
-			if (def_reg != IR_REG_NONE) {
-				ir_emit_load_mem_fp(ctx, type, def_reg, IR_MEM_BO(tmp_reg, -16));
-			}
-			|	jmp >2
-			|1:
-			|	mov Ra(tmp_reg), aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))]
-			if (def_reg != IR_REG_NONE) {
-				ir_emit_load_mem_fp(ctx, type, def_reg, IR_MEM_BO(tmp_reg, 0));
-			}
-			|	add Ra(tmp_reg), 8
-			|	mov aword [Ra(op2_reg)+(offset+offsetof(ir_x86_64_sysv_va_list, overflow_arg_area))], Ra(tmp_reg)
-			|2:
-		}
-		if (def_reg != IR_REG_NONE && IR_REG_SPILLED(ctx->regs[def][0])) {
-			ir_emit_store(ctx, type, def, def_reg);
+				break;
+			case IR_AND:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	ASM_TXT_TMEM_OP pand, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+				break;
+			case IR_OR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	ASM_TXT_TMEM_OP por, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+				break;
+			case IR_XOR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				|	ASM_TXT_TMEM_OP pxor, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+				break;
+			case IR_EQ:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	ASM_TXT_TMEM_OP pcmpeqb, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	ASM_TXT_TMEM_OP pcmpeqw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	ASM_TXT_TMEM_OP pcmpeqd, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	ASM_TXT_TMEM_OP pcmpeqq, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+						} else {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							|	ASM_TXT_TMEM_OP pcmpeqd, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+							|	pshufd	xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 177
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0);
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_TXT_OP cmpp, element_type, def_reg, mem, 0
+				}
+				break;
+			case IR_NE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (element_type == IR_I8 || element_type == IR_U8) {
+						|	ASM_TXT_TMEM_OP pcmpeqb, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I16 || element_type == IR_U16) {
+						|	ASM_TXT_TMEM_OP pcmpeqw, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	ASM_TXT_TMEM_OP pcmpeqd, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						if (ctx->mflags & IR_X86_SSE41) {
+							|	ASM_TXT_TMEM_OP pcmpeqq, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+						} else {
+							IR_ASSERT(tmp_reg != IR_REG_NONE);
+							|	ASM_TXT_TMEM_OP pcmpeqd, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+							|	pshufd	xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 177
+							|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(0);
+					}
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	ASM_SSE_INT_VEC_REG_REG_OP pcmpeq, element_type, def_reg, tmp_reg
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_SSE_FP_VEC_REG_MEM_TXT_OP cmpp, element_type, def_reg, mem, 4
+				}
+				break;
 		}
-|.endif
-#endif
+	}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
 }

-static void ir_emit_switch(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn)
+static void ir_emit_vector_binop_avx(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type;
-	ir_block *bb;
-	ir_insn *use_insn, *val;
-	uint32_t n, *p, use_block;
-	int i;
-	int label, default_label = 0;
-	int count = 0;
-	ir_val min, max;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t width;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
 	ir_reg op2_reg = ctx->regs[def][2];
 	ir_reg tmp_reg = ctx->regs[def][3];
-	bool has_case_range = 0;
+	ir_reg tmp2_reg = ctx->tmp_regs ? ctx->tmp_regs[def] : IR_REG_NONE;
+	int shift_count = 0;

-	type = ctx->ir_base[insn->op2].type;
-	IR_ASSERT(tmp_reg != IR_REG_NONE);
-	if (IR_IS_TYPE_SIGNED(type)) {
-		min.u64 = 0x7fffffffffffffff;
-		max.u64 = 0x8000000000000000;
-	} else {
-		min.u64 = 0xffffffffffffffff;
-		max.u64 = 0x0;
-	}
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);

-	bb = &ctx->cfg_blocks[b];
-	p = &ctx->cfg_edges[bb->successors];
-	for (n = bb->successors_count; n != 0; p++, n--) {
-		use_block = *p;
-		use_insn = &ctx->ir_base[ctx->cfg_blocks[use_block].start];
-		if (use_insn->op == IR_CASE_VAL) {
-			val = &ctx->ir_base[use_insn->op2];
-			IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-			if (IR_IS_TYPE_SIGNED(type)) {
-				IR_ASSERT(IR_IS_TYPE_SIGNED(val->type));
-				min.i64 = IR_MIN(min.i64, val->val.i64);
-				max.i64 = IR_MAX(max.i64, val->val.i64);
-			} else {
-				IR_ASSERT(!IR_IS_TYPE_SIGNED(val->type));
-				min.u64 = (int64_t)IR_MIN(min.u64, val->val.u64);
-				max.u64 = (int64_t)IR_MAX(max.u64, val->val.u64);
+	if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+		type = ctx->ir_base[op1].type;
+		IR_ASSERT(type == ctx->ir_base[op2].type);
+	} else if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+		IR_ASSERT(type == ctx->ir_base[op1].type);
+		IR_ASSERT(insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR);
+		if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				ir_emit_load(ctx, type, op2_reg, op2);
 			}
-			count++;
-		} else if (use_insn->op == IR_CASE_RANGE) {
-			has_case_range = 1;
-			val = &ctx->ir_base[use_insn->op2];
-			IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-			ir_insn *val2 = &ctx->ir_base[use_insn->op3];
-			IR_ASSERT(!IR_IS_SYM_CONST(val2->op));
-			if (IR_IS_TYPE_SIGNED(type)) {
-				IR_ASSERT(IR_IS_TYPE_SIGNED(val->type));
-				min.i64 = IR_MIN(min.i64, val->val.i64);
-				max.i64 = IR_MAX(max.i64, val2->val.i64);
+#if IR_X86_I64
+			if (ctx->ir_base[op2].type == IR_I64 || ctx->ir_base[op2].type == IR_U64) {
+				/* ignore the high part of the register */
+				op2_reg = IR_REG_I64_LO(op2_reg);
+			}
+#endif
+			IR_ASSERT(tmp_reg != IR_REG_NONE);
+			if (ctx->mflags & IR_X86_AVX) {
+				|	vmovd xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op2_reg)
 			} else {
-				IR_ASSERT(!IR_IS_TYPE_SIGNED(val->type));
-				min.u64 = (int64_t)IR_MIN(min.u64, val->val.u64);
-				max.u64 = (int64_t)IR_MAX(max.u64, val2->val.u64);
+				|	movd xmm(tmp_reg-IR_REG_FP_FIRST), Rd(op2_reg)
 			}
+			op2_reg = tmp_reg;
 		} else {
-			IR_ASSERT(use_insn->op == IR_CASE_DEFAULT);
-			default_label = ir_skip_empty_target_blocks(ctx, use_block);
+			IR_ASSERT(IR_IS_CONST_REF(insn->op2));
+			shift_count = ctx->ir_base[insn->op2].val.i32;
 		}
+	} else {
+		IR_ASSERT(type == ctx->ir_base[op1].type && type == ctx->ir_base[op2].type);
 	}

-	IR_ASSERT(op2_reg != IR_REG_NONE);
-	if (IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		ir_emit_load(ctx, type, op2_reg, insn->op2);
-	}
-
-	/* Generate a table jmp or a seqence of calls */
-	if (!has_case_range && count > 2 && (max.i64-min.i64) < count * 8) {
-		int *labels = ir_mem_malloc(sizeof(int) * (size_t)(max.i64 - min.i64 + 1));
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	width = IR_VECTOR_SIZE(type);

-		for (i = 0; i <= (max.i64 - min.i64); i++) {
-			labels[i] = default_label;
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+	if (IR_REG_SPILLED(op1_reg)) {
+		op1_reg = IR_REG_NUM(op1_reg);
+		ir_emit_load(ctx, type, op1_reg, op1);
+		if (op1 == op2) {
+			op2_reg = op1_reg;
 		}
-		p = &ctx->cfg_edges[bb->successors];
-		for (n = bb->successors_count; n != 0; p++, n--) {
-			use_block = *p;
-			use_insn = &ctx->ir_base[ctx->cfg_blocks[use_block].start];
-			if (use_insn->op == IR_CASE_VAL) {
-				val = &ctx->ir_base[use_insn->op2];
-				IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-				label = ir_skip_empty_target_blocks(ctx, use_block);
-				labels[val->val.i64 - min.i64] = label;
+	}
+	if (op2_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op2_reg)) {
+			op2_reg = IR_REG_NUM(op2_reg);
+			if (op1 != op2) {
+				ir_emit_load(ctx, type, op2_reg, op2);
 			}
 		}
-
-		switch (ir_type_size[type]) {
+		switch (insn->op) {
 			default:
-				IR_ASSERT(0 && "Unsupported type size");
-			case 1:
-				if (IR_IS_TYPE_SIGNED(type)) {
-					|	movsx Ra(op2_reg), Rb(op2_reg)
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpadd, element_type, 16, tmp_reg, tmp_reg, tmp2_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpadd, element_type, 16, def_reg, op1_reg, op2_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpadd, element_type, width, def_reg, op1_reg, op2_reg
+					}
 				} else {
-					|	movzx Ra(op2_reg), Rb(op2_reg)
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_OP vaddp, element_type, width, def_reg, op1_reg, op2_reg
+				}
+				break;
+			case IR_SUB:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, 16, tmp_reg, tmp_reg, tmp2_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, 16, def_reg, op1_reg, op2_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpsub, element_type, width, def_reg, op1_reg, op2_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_OP vsubp, element_type, width, def_reg, op1_reg, op2_reg
+				}
+				break;
+			case IR_MUL:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width <= 16) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpmullw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpmulld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpmullw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vpmulld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+							if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpmullw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpmullw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vpmulld xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpmulld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_OP vmulp, element_type, width, def_reg, op1_reg, op2_reg
+				}
+				break;
+			case IR_DIV:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_AVX_FP_VEC_REG_REG_REG_OP vdivp, element_type, width, def_reg, op1_reg, op2_reg
+				break;
+			case IR_MIN:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width <= 16) {
+						if (element_type == IR_I8) {
+							|	vpminsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U8) {
+							|	vpminub xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16) {
+							|	vpminsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U16) {
+							|	vpminuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32) {
+							|	vpminsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U32) {
+							|	vpminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (element_type == IR_I8) {
+							|	vpminsb ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U8) {
+							|	vpminub ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16) {
+							|	vpminsw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U16) {
+							|	vpminuw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32) {
+							|	vpminsd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U32) {
+							|	vpminud ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_OP vminp, element_type, width, def_reg, op1_reg, op2_reg
+				}
+				break;
+			case IR_MAX:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width <= 16) {
+						if (element_type == IR_I8) {
+							|	vpmaxsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U8) {
+							|	vpmaxub xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16) {
+							|	vpmaxsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U16) {
+							|	vpmaxuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32) {
+							|	vpmaxsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U32) {
+							|	vpmaxud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (element_type == IR_I8) {
+							|	vpmaxsb ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U8) {
+							|	vpmaxub ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16) {
+							|	vpmaxsw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U16) {
+							|	vpmaxuw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32) {
+							|	vpmaxsd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_U32) {
+							|	vpmaxud ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_OP vmaxp, element_type, width, def_reg, op1_reg, op2_reg
+				}
+				break;
+			case IR_AND:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width <= 16) {
+					|	vpand xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpand ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	vpand xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vpand xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_OR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width <= 16) {
+					|	vpor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpor ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	vpor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vpor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_XOR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width <= 16) {
+					|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_EQ:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp_reg, tmp_reg, tmp2_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, op1_reg, op2_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, op1_reg, op2_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 0
+				}
+				break;
+			case IR_NE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp2_reg, tmp_reg, tmp2_reg
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp2_reg, tmp2_reg, tmp_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, op1_reg, op2_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, def_reg, tmp_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, op1_reg, op2_reg
+						if (width <= 16) {
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector width");
+						}
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, def_reg, tmp_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 4
+				}
+				break;
+			case IR_LT:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, tmp_reg, tmp2_reg, tmp_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, def_reg, op2_reg, op1_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, width, def_reg, op2_reg, op1_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 1
+				}
+				break;
+			case IR_GT:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, tmp_reg, tmp_reg, tmp2_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, def_reg, op1_reg, op2_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, width, def_reg, op1_reg, op2_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 14
+				}
+				break;
+			case IR_LE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, tmp2_reg, tmp_reg, tmp2_reg
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp2_reg, tmp2_reg, tmp_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, def_reg, op1_reg, op2_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, def_reg, tmp_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+					} else {
+						IR_ASSERT(tmp_reg != IR_REG_NONE);
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, width, def_reg, op1_reg, op2_reg
+						if (width <= 16) {
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector width");
+						}
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, def_reg, tmp_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 2
 				}
 				break;
-			case 2:
-				if (IR_IS_TYPE_SIGNED(type)) {
-					|	movsx Ra(op2_reg), Rw(op2_reg)
+			case IR_GE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, tmp2_reg, tmp2_reg, tmp_reg
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp2_reg, tmp2_reg, tmp_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, 16, def_reg, op2_reg, op1_reg
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, def_reg, tmp_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpgt, element_type, width, def_reg, op2_reg, op1_reg
+						if (width <= 16) {
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector width");
+						}
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, def_reg, tmp_reg
+					}
 				} else {
-					|	movzx Ra(op2_reg), Rw(op2_reg)
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 13
 				}
 				break;
-			case 4:
-|.if X64
-				if (IR_IS_TYPE_SIGNED(type)) {
-					if (op2_reg == IR_REG_RAX) {
-						|	cdqe
+			case IR_ULT:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (width <= 16) {
+						if (element_type == IR_I8 || element_type == IR_U8) {
+							|	vpsubusb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsubusw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32|| element_type == IR_U32) {
+							|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpsllq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vpsubusb ymm(def_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpsubusw ymm(def_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I32|| element_type == IR_U32) {
+								|	vpminud ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vpcmpeqq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpsllq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), 63
+								|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpsubusb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpsubusb xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpsubusw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpsubusw xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I32|| element_type == IR_U32) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 63
+
+								|	vpxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						}
 					} else {
-						|	movsxd Ra(op2_reg), Rd(op2_reg)
+						IR_ASSERT(0 && "unsupported vector width");
 					}
 				} else {
-					|	mov Rd(op2_reg), Rd(op2_reg)
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op2_reg, op1_reg, 6
 				}
 				break;
-||			case 8:
-|.endif
+			case IR_UGT:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (width <= 16) {
+						if (element_type == IR_I8 || element_type == IR_U8) {
+							|	vpsubusb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsubusw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32|| element_type == IR_U32) {
+							|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpsllq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vpsubusb ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpsubusw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I32|| element_type == IR_U32) {
+								|	vpminud ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vpcmpeqq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpsllq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), 63
+								|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpsubusb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpsubusb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpsubusw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpsubusw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I32|| element_type == IR_U32) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpminud xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1
+
+								|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 63
+
+								|	vpxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 6
+				}
 				break;
-		}
+			case IR_ULE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (width <= 16) {
+						if (element_type == IR_I8 || element_type == IR_U8) {
+							|	vpminub xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpminuw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32|| element_type == IR_U32) {
+							|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpsllq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vpminub ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpminuw ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vpminud ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vpcmpeqq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpsllq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), 63
+								|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-		if (min.i64 != 0) {
-			int64_t offset = -min.i64;
+								|	vpminub xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)

-			if (IR_IS_SIGNED_32BIT(offset)) {
-				|	lea Ra(tmp_reg), [Ra(op2_reg)+(int32_t)offset]
-			} else {
-				IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-				|	mov64 Rq(tmp_reg), offset
-				|	add Ra(tmp_reg), Ra(op2_reg)
-|.endif
-			}
-			if (default_label) {
-				offset = max.i64 - min.i64;
+								|	vpminub xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)

-				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
-				|	cmp Ra(tmp_reg), (int32_t)offset
-				|	ja =>default_label
-			}
-|.if X64
-			if (ctx->code_buffer
-			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->start)
-			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->end)) {
-				|	jmp aword [Ra(tmp_reg)*8+>1]
-			} else {
-				int64_t offset = -min.i64;
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
-				offset *= 8;
-				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
-				|	lea Ra(tmp_reg), aword [>1]
-				|	jmp aword [Ra(tmp_reg)+Ra(op2_reg)*8+offset]
-			}
-|.else
-			|	jmp aword [Ra(tmp_reg)*4+>1]
-|.endif
-		} else {
-			if (default_label) {
-				int64_t offset = max.i64;
+								|	vpminuw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)

-				IR_ASSERT(IR_IS_SIGNED_32BIT(offset));
-				|	cmp Ra(op2_reg), (int32_t)offset
-				|	ja =>default_label
-			}
-|.if X64
-			if (ctx->code_buffer
-			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->start)
-			 && IR_IS_SIGNED_32BIT((char*)ctx->code_buffer->end)) {
-				|	jmp aword [Ra(op2_reg)*8+>1]
-			} else {
-				|	lea Ra(tmp_reg), aword [>1]
-				|	jmp aword [Ra(tmp_reg)+Ra(op2_reg)*8]
-			}
-|.else
-			|	jmp aword [Ra(op2_reg)*4+>1]
-|.endif
-		}
+								|	vpminuw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)

-		|.jmp_table
-		if (!data->jmp_table_label) {
-			data->jmp_table_label = ctx->cfg_blocks_count + ctx->consts_count + 3;
-			|=>data->jmp_table_label:
-		}
-		|.align aword
-		|1:
-		for (i = 0; i <= (max.i64 - min.i64); i++) {
-			int b = labels[i];
-			if (b) {
-				ir_block *bb = &ctx->cfg_blocks[b];
-				ir_insn *insn = &ctx->ir_base[bb->end];
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-				if (insn->op == IR_IJMP && IR_IS_CONST_REF(insn->op2)) {
-					ir_ref prev = ctx->prev_ref[bb->end];
-					if (prev != bb->start && ctx->ir_base[prev].op == IR_SNAPSHOT) {
-						prev = ctx->prev_ref[prev];
-					}
-					if (prev == bb->start) {
-						void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op2]);
+								|	vpminud xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)

-						|	.aword &addr
-						if (ctx->ir_base[bb->start].op1 == def
-						 && ctx->ir_base[bb->start].op != IR_CASE_DEFAULT) {
-							bb->flags |= IR_BB_EMPTY;
-						}
-						continue;
-					}
-				}
-				|	.aword =>b
-			} else {
-				|	.aword 0
-			}
-		}
-		|.code
-		ir_mem_free(labels);
-	} else {
-		p = &ctx->cfg_edges[bb->successors];
-		for (n = bb->successors_count; n != 0; p++, n--) {
-			use_block = *p;
-			use_insn = &ctx->ir_base[ctx->cfg_blocks[use_block].start];
-			if (use_insn->op == IR_CASE_VAL) {
-				val = &ctx->ir_base[use_insn->op2];
-				IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-				label = ir_skip_empty_target_blocks(ctx, use_block);
-				if (val->val.u64 == 0) {
-					|	ASM_REG_REG_OP test, type, op2_reg, op2_reg
-				} else if (IR_IS_32BIT(type, val->val)) {
-					|	ASM_REG_IMM_OP cmp, type, op2_reg, val->val.i32
-				} else {
-					IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-					|	mov64 Ra(tmp_reg), val->val.i64
-					|	ASM_REG_REG_OP cmp, type, op2_reg, tmp_reg
-|.endif
-				}
-				|	je =>label
-			} else if (use_insn->op == IR_CASE_RANGE) {
-				val = &ctx->ir_base[use_insn->op2];
-				IR_ASSERT(!IR_IS_SYM_CONST(val->op));
-				label = ir_skip_empty_target_blocks(ctx, use_block);
-				if (IR_IS_32BIT(type, val->val)) {
-					|	ASM_REG_IMM_OP cmp, type, op2_reg, val->val.i32
-				} else {
-					IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-					|	mov64 Ra(tmp_reg), val->val.i64
-					|	ASM_REG_REG_OP cmp, type, op2_reg, tmp_reg
-|.endif
-				}
-				if (IR_IS_TYPE_SIGNED(type)) {
-					|	jl >1
-				} else {
-					|	jb >1
-				}
-				val = &ctx->ir_base[use_insn->op3];
-				IR_ASSERT(!IR_IS_SYM_CONST(val->op3));
-				label = ir_skip_empty_target_blocks(ctx, use_block);
-				if (IR_IS_32BIT(type, val->val)) {
-					|	ASM_REG_IMM_OP cmp, type, op2_reg, val->val.i32
-				} else {
-					IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-					|	mov64 Ra(tmp_reg), val->val.i64
-					|	ASM_REG_REG_OP cmp, type, op2_reg, tmp_reg
-|.endif
-				}
-				if (IR_IS_TYPE_SIGNED(type)) {
-					|	jle =>label
-				} else {
-					|	jbe =>label
-				}
-				|1:
-			}
-		}
-		if (default_label) {
-			|	jmp =>default_label
-		}
-	}
-}
+								|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)

-static int32_t ir_call_used_stack(ir_ctx *ctx, ir_insn *insn, const ir_call_conv_dsc *cc, int *copy_stack_ptr)
-{
-	int j, n;
-	ir_type type;
-	int int_param = 0;
-	int fp_param = 0;
-	int32_t used_stack = 0;
-	int32_t copy_stack = 0;
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-	n = insn->inputs_count;
-	for (j = 3; j <= n; j++) {
-		ir_insn *arg = &ctx->ir_base[ir_insn_op(insn, j)];
-		type = arg->type;
-		if (IR_IS_TYPE_INT(type)) {
-			if (arg->op == IR_ARGVAL) {
-				int size = arg->op2;
-				int align = arg->op3;
+								|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 63

-				if (!cc->pass_struct_by_val) {
-					copy_stack += size;
-					align = IR_MAX((int)sizeof(void*), align);
-					copy_stack = IR_ALIGNED_SIZE(copy_stack, align);
-					type = IR_ADDR;
+								|	vpxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
 				} else {
-					align = IR_MAX((int)sizeof(void*), align);
-					used_stack = IR_ALIGNED_SIZE(used_stack, align);
-					used_stack += size;
-					used_stack = IR_ALIGNED_SIZE(used_stack, sizeof(void*));
-					continue;
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op2_reg, op1_reg, 5
 				}
-			}
-			if (int_param >= cc->int_param_regs_count) {
-				used_stack += IR_MAX(sizeof(void*), ir_type_size[type]);
-			}
-			int_param++;
-			if (cc->shadow_param_regs) {
-				fp_param++;
-			}
-		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
-			if (fp_param >= cc->fp_param_regs_count) {
-				used_stack += IR_MAX(sizeof(void*), ir_type_size[type]);
-			}
-			fp_param++;
-			if (cc->shadow_param_regs) {
-				int_param++;
-			}
-		}
-	}
+				break;
+			case IR_UGE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (width <= 16) {
+						if (element_type == IR_I8 || element_type == IR_U8) {
+							|	vpminub xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpminuw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32|| element_type == IR_U32) {
+							|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpcmpeqq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpsllq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 63
+							|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+							|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vpminub ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpminuw ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vpminud ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vpcmpeqq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpsllq ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), 63
+								|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							if (element_type == IR_I8 || element_type == IR_U8) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-	/* Reserved "home space" or "shadow store" for register arguments (used in Windows64 ABI) */
-	used_stack += cc->shadow_store_size;
+								|	vpminub xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+
+								|	vpminub xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqb xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)

-	copy_stack = IR_ALIGNED_SIZE(copy_stack, 16);
-	used_stack += copy_stack;
-	*copy_stack_ptr = copy_stack;
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-	return used_stack;
-}
+								|	vpminuw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)

-static int32_t ir_emit_arguments(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_proto_t *proto, const ir_call_conv_dsc *cc, ir_reg tmp_reg)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	int j, n;
-	ir_ref arg;
-	ir_insn *arg_insn;
-	uint8_t type;
-	ir_reg src_reg, dst_reg;
-	int int_param = 0;
-	int fp_param = 0;
-	int count = 0;
-	int32_t used_stack, copy_stack = 0, stack_offset = cc->shadow_store_size;
-	ir_copy *copies;
-	bool do_pass3 = 0;
-	/* For temporaries we may use any scratch registers except for registers used for parameters */
-	ir_reg tmp_fp_reg = IR_REG_FP_LAST; /* Temporary register for FP loads and swap */
+								|	vpminuw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)

-	n = insn->inputs_count;
-	if (n < 3) {
-		return 0;
-	}
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-	if (tmp_reg == IR_REG_NONE) {
-		tmp_reg = IR_REG_RAX;
-	}
+								|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)

-	if (insn->op == IR_CALL
-	 && (ctx->flags2 & IR_PREALLOCATED_STACK)
-	 && !cc->cleanup_stack_by_callee) {
-		if (!cc->pass_struct_by_val) {
-			used_stack = ir_call_used_stack(ctx, insn, cc, &copy_stack);
-		} else {
-			used_stack = 0;
-		}
-	} else {
-		used_stack = ir_call_used_stack(ctx, insn, cc, &copy_stack);
-		if (cc->shadow_store_size
-		 && insn->op == IR_TAILCALL
-		 && used_stack == cc->shadow_store_size) {
-			used_stack = 0;
-		}
-		if (ctx->fixed_call_stack_size
-		 && used_stack <= ctx->fixed_call_stack_size
-		 && !cc->cleanup_stack_by_callee) {
-			used_stack = 0;
-		} else {
-			/* Stack must be 16 byte aligned */
-			int32_t aligned_stack = IR_ALIGNED_SIZE(used_stack, 16);
-			ctx->call_stack_size += aligned_stack;
-			if (aligned_stack) {
-				|	sub Ra(IR_REG_RSP), aligned_stack
-			}
-		}
-	}
+								|	vpminud xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)

-	if (copy_stack) {
-		/* Copy struct arguments */
-		IR_ASSERT(sizeof(void*) == 8);
-|.if X64
-		int copy_stack_offset = 0;
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else if (element_type == IR_I64 || element_type == IR_U64) {
+								|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+								|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST), 1

-		for (j = 3; j <= n; j++) {
-			arg = ir_insn_op(insn, j);
-			src_reg = ir_get_alocated_reg(ctx, def, j);
-			arg_insn = &ctx->ir_base[arg];
-			type = arg_insn->type;
+								|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 63

-			if (arg_insn->op == IR_ARGVAL) {
-				/* make a stack copy */
-				int size = arg_insn->op2;
-				int align = arg_insn->op3;
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)

-				copy_stack_offset += size;
-				align = IR_MAX((int)sizeof(void*), align);
-				copy_stack_offset = IR_ALIGNED_SIZE(copy_stack_offset, align);
-				src_reg = ctx->regs[arg][1];
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+								|	vpcmpgtq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+								|	vpcmpeqq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)

-				|	lea	rdi, [rsp + (used_stack - copy_stack_offset)]
-				if (src_reg != IR_REG_NONE) {
-					if (IR_REG_SPILLED(src_reg)) {
-						src_reg = IR_REG_NUM(src_reg);
-						ir_emit_load(ctx, IR_ADDR, src_reg, arg_insn->op1);
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
 					}
-					|	mov rsi, Ra(src_reg)
 				} else {
-					ir_emit_load(ctx, IR_ADDR, IR_REG_RSI, arg_insn->op1);
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 5
 				}
-				ir_emit_load_imm_int(ctx, IR_ADDR, IR_REG_RCX, size);
-				|	rep; movsb
-			}
+				break;
+			case IR_ORDERED:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 7
+				break;
+			case IR_UNORDERED:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_AVX_FP_VEC_REG_REG_REG_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, op2_reg, 3
+				break;
+			case IR_SHL:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width <= 16) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	vpsllw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsllw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpslld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsllq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsllw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsllw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpslld xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsllq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_SHR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width <= 16) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	vpsrlw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsrlw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsrlq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsrlw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsrlw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrld xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsrlq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_SAR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width <= 16) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	vpsraw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	vpsrad xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsraw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrad ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsraw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsraw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrad xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+							|	vpsrad xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op2_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
 		}
-|.endif
-	}
+	} else if (IR_IS_CONST_REF(op2)) {
+		int label = ir_get_const_label(ctx, op2);

-	/* 1. move all register arguments that should be passed through stack
-	 *    and collect arguments that should be passed through registers */
-	copies = ir_mem_malloc((n - 2) * sizeof(ir_copy));
-	for (j = 3; j <= n; j++) {
-		arg = ir_insn_op(insn, j);
-		src_reg = ir_get_alocated_reg(ctx, def, j);
-		arg_insn = &ctx->ir_base[arg];
-		type = arg_insn->type;
-		if (IR_IS_TYPE_INT(type)) {
-			if (arg_insn->op == IR_ARGVAL && cc->pass_struct_by_val) {
-				int size = arg_insn->op2;
-				int align = arg_insn->op3;
-				align = IR_MAX((int)sizeof(void*), align);
-				stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
-				if (size) {
-					src_reg = ctx->regs[arg][1];
-					if (src_reg != IR_REG_NONE) {
-						if (IR_REG_SPILLED(src_reg)) {
-							src_reg = IR_REG_NUM(src_reg);
-							ir_emit_load(ctx, IR_ADDR, src_reg, arg_insn->op1);
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpadd, element_type, 16, tmp_reg, tmp_reg, [=>label+16]
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpadd, element_type, 16, def_reg, op1_reg, [=>label]
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpadd, element_type, width, def_reg, op1_reg, [=>label]
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_TXT_OP vaddp, element_type, width, def_reg, op1_reg, [=>label]
+				}
+				break;
+			case IR_SUB:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpsub, element_type, 16, tmp_reg, tmp_reg, [=>label+16]
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpsub, element_type, 16, def_reg, op1_reg, [=>label]
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpsub, element_type, width, def_reg, op1_reg, [=>label]
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_TXT_OP vsubp, element_type, width, def_reg, op1_reg, [=>label]
+				}
+				break;
+			case IR_MUL:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpmullw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpmulld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
 						}
-						if (src_reg != IR_REG_RSI) {
-							|.if X64
-							|	mov rsi, Ra(src_reg)
-							|.else
-							|	mov	esi, Ra(src_reg)
-							|.endif
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpmullw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vpmulld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							if (element_type == IR_I16 || element_type == IR_U16) {
+								|	vpmullw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), [=>label+16]
+								|	vpmullw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	vpmulld xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), [=>label+16]
+								|	vpmulld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_TXT_OP vmulp, element_type, width, def_reg, op1_reg, [=>label]
+				}
+				break;
+			case IR_DIV:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_AVX_FP_VEC_REG_REG_TXT_OP vdivp, element_type, width, def_reg, op1_reg, [=>label]
+				break;
+			case IR_MIN:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						if (element_type == IR_I8) {
+							|	vpminsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U8) {
+							|	vpminub xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I16) {
+							|	vpminsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U16) {
+							|	vpminuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I32) {
+							|	vpminsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U32) {
+							|	vpminud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (element_type == IR_I8) {
+							|	vpminsb ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U8) {
+							|	vpminub ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I16) {
+							|	vpminsw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U16) {
+							|	vpminuw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I32) {
+							|	vpminsd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U32) {
+							|	vpminud ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_TXT_OP vminp, element_type, width, def_reg, op1_reg, [=>label]
+				}
+				break;
+			case IR_MAX:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						if (element_type == IR_I8) {
+							|	vpmaxsb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U8) {
+							|	vpmaxub xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I16) {
+							|	vpmaxsw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U16) {
+							|	vpmaxuw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I32) {
+							|	vpmaxsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U32) {
+							|	vpmaxud xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (element_type == IR_I8) {
+							|	vpmaxsb ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U8) {
+							|	vpmaxub ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I16) {
+							|	vpmaxsw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U16) {
+							|	vpmaxuw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_I32) {
+							|	vpmaxsd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else if (element_type == IR_U32) {
+							|	vpmaxud ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_TXT_OP vmaxp, element_type, width, def_reg, op1_reg, [=>label]
+				}
+				break;
+			case IR_AND:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width == 16) {
+					|	vpand xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpand ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vpand xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), [=>label + 16]
+						|	vpand xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_OR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width == 16) {
+					|	vpor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpor ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vpor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), [=>label+16]
+						|	vpor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_XOR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width == 16) {
+					|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpxor ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [=>label]
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), [=>label+16]
+						|	vpxor xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [=>label]
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_EQ:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpcmpeq, element_type, 16, tmp_reg, tmp_reg, [=>label+16]
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpcmpeq, element_type, 16, def_reg, op1_reg, [=>label]
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpcmpeq, element_type, width, def_reg, op1_reg, [=>label]
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+//					|	ASM_AVX_FP_VEC_REG_REG_TXT_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, [=>label], 0
+					IR_ASSERT(0 && "DynAsm limitation (IP-relative address followed by immediate operand) ???");
+				}
+				break;
+			case IR_NE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpcmpeq, element_type, 16, tmp2_reg, tmp_reg, [=>label+16]
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp2_reg, tmp2_reg, tmp_reg
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpcmpeq, element_type, 16, def_reg, op1_reg, [=>label]
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, def_reg, tmp_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_TXT_OP vpcmpeq, element_type, width, def_reg, op1_reg, [=>label]
+						if (width == 16) {
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector width");
+						}
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, def_reg, tmp_reg
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+//					|	ASM_AVX_FP_VEC_REG_REG_TXT_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, [=>label], 4
+					IR_ASSERT(0 && "DynAsm limitation (IP-relative address followed by immediate operand) ???");
+				}
+				break;
+			case IR_SHL:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width == 16) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	vpsllw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsllw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpslld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsllq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsllw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsllw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpslld xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsllq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_SHR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width == 16) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	vpsrlw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else if (element_type == IR_I64 || element_type == IR_U64) {
+						|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else {
+						IR_ASSERT(0 && "unsupported vector type");
+					}
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsrlw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsrlq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsrlw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsrlw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrld xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I64 || element_type == IR_U64) {
+							|	vpsrlq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
 						}
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					}
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
+				}
+				break;
+			case IR_SAR:
+				IR_ASSERT(IR_IS_TYPE_INT(element_type));
+				if (width == 16) {
+					if (element_type == IR_I16 || element_type == IR_U16) {
+						|	vpsraw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+					} else if (element_type == IR_I32 || element_type == IR_U32) {
+						|	vpsrad xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
 					} else {
-						ir_emit_load(ctx, IR_ADDR, IR_REG_RSI, arg_insn->op1);
+						IR_ASSERT(0 && "unsupported vector type");
 					}
-					if (stack_offset == 0) {
-						|.if X64
-						|	mov	rdi, rsp
-						|.else
-						|	mov	edi, esp
-						|.endif
+				} else if (width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsraw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrad ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
 					} else {
-						|.if X64
-						|	lea	rdi, [rsp+stack_offset]
-						|.else
-						|	lea	edi, [esp+stack_offset]
-						|.endif
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	vpsraw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsraw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	vpsrad xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), shift_count
+							|	vpsrad xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), shift_count
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
 					}
-					|.if X64
-					|	mov rcx, size
-					|	rep; movsb
-					|.else
-					|	mov ecx, size
-					|	rep; movsb
-					|.endif
+				} else {
+					IR_ASSERT(0 && "unsupported vector width");
 				}
-				stack_offset += size;
-				stack_offset = IR_ALIGNED_SIZE(stack_offset, sizeof(void*));
-				continue;
-			}
-			if (int_param < cc->int_param_regs_count) {
-				dst_reg = cc->int_param_regs[int_param];
-			} else {
-				dst_reg = IR_REG_NONE; /* pass argument through stack */
-			}
-			int_param++;
-			if (cc->shadow_param_regs) {
-				fp_param++;
-			}
-			if (arg_insn->op == IR_ARGVAL && !cc->pass_struct_by_val) {
-				do_pass3 = 3;
-				continue;
-			}
-		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(type));
-			if (fp_param < cc->fp_param_regs_count) {
-				dst_reg = cc->fp_param_regs[fp_param];
-			} else {
-				dst_reg = IR_REG_NONE; /* pass argument through stack */
-			}
-			fp_param++;
-			if (cc->shadow_param_regs) {
-				int_param++;
-			}
+				break;
 		}
-		if (dst_reg != IR_REG_NONE) {
-			if (IR_IS_CONST_REF(arg) ||
-			    src_reg == IR_REG_NONE ||
-			    (IR_REG_SPILLED(src_reg) && !IR_REGSET_IN(cc->preserved_regs, IR_REG_NUM(src_reg)))) {
-				/* delay CONST->REG and MEM->REG moves to third pass */
-				do_pass3 = 1;
-			} else {
-				if (IR_REG_SPILLED(src_reg)) {
-					src_reg = IR_REG_NUM(src_reg);
-					ir_emit_load(ctx, type, src_reg, arg);
-				}
-				if (src_reg != dst_reg) {
-					/* delay REG->REG moves to second pass */
-					copies[count].type = type;
-					copies[count].from = src_reg;
-					copies[count].to = dst_reg;
-					count++;
-				}
-			}
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, op2) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, op2);
 		} else {
-			/* Pass register arguments to stack (REG->MEM moves) */
-			if (!IR_IS_CONST_REF(arg) && src_reg != IR_REG_NONE && !IR_REG_SPILLED(src_reg)) {
-				ir_emit_store_mem(ctx, type, IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset), src_reg);
-			} else {
-				do_pass3 = 1;
-			}
-			stack_offset += IR_MAX(sizeof(void*), ir_type_size[type]);
+			mem = ir_ref_spill_slot(ctx, op2);
 		}
-	}
-
-	/* 2. move all arguments that should be passed from one register to another (REG->REG movs) */
-	if (count) {
-		ir_parallel_copy(ctx, copies, count, tmp_reg, tmp_fp_reg);
-	}
-	ir_mem_free(copies);
-
-	/* 3. move the remaining memory and immediate values */
-	if (do_pass3) {
-		int copy_stack_offset = 0;
-
-		stack_offset = cc->shadow_store_size;
-		int_param = 0;
-		fp_param = 0;
-		for (j = 3; j <= n; j++) {
-			arg = ir_insn_op(insn, j);
-			src_reg = ir_get_alocated_reg(ctx, def, j);
-			arg_insn = &ctx->ir_base[arg];
-			type = arg_insn->type;
-			if (IR_IS_TYPE_INT(type)) {
-				if (arg_insn->op == IR_ARGVAL) {
-					int size = arg_insn->op2;
-					int align = arg_insn->op3;
-
-					if (cc->pass_struct_by_val) {
-						align = IR_MAX((int)sizeof(void*), align);
-						stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
-						stack_offset += size;
-						stack_offset = IR_ALIGNED_SIZE(stack_offset, sizeof(void*));
-						continue;
+		switch (insn->op) {
+			default:
+				IR_ASSERT(0 && "NIY binary op");
+			case IR_ADD:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpadd, element_type, 16, tmp_reg, tmp_reg, IR_MEM_ADD(mem, 16)
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpadd, element_type, 16, def_reg, op1_reg, mem
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
 					} else {
-						/* pass pointer to the copy on stack */
-						copy_stack_offset += size;
-						align = IR_MAX((int)sizeof(void*), align);
-						copy_stack_offset = IR_ALIGNED_SIZE(copy_stack_offset, align);
-						if (int_param < cc->int_param_regs_count) {
-							dst_reg = cc->int_param_regs[int_param];
-							|	lea Ra(dst_reg), [r4 + (used_stack - copy_stack_offset)]
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpadd, element_type, width, def_reg, op1_reg, mem
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vaddp, element_type, width, def_reg, op1_reg, mem
+				}
+				break;
+			case IR_SUB:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, 16, tmp_reg, tmp_reg, IR_MEM_ADD(mem, 16)
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, 16, def_reg, op1_reg, mem
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpsub, element_type, width, def_reg, op1_reg, mem
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vsubp, element_type, width, def_reg, op1_reg, mem
+				}
+				break;
+			case IR_MUL:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						if (element_type == IR_I16 || element_type == IR_U16) {
+							|	ASM_TXT_TXT_TMEM_OP vpmullw, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_I32 || element_type == IR_U32) {
+							|	ASM_TXT_TXT_TMEM_OP vpmulld, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
 						} else {
-							|	lea Ra(tmp_reg), [r4 + (used_stack - copy_stack_offset)]
-							ir_emit_store_mem_int(ctx, IR_ADDR, IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset), tmp_reg);
-							stack_offset += sizeof(void*);
+							IR_ASSERT(0 && "unsupported vector type");
 						}
-						int_param++;
-						if (cc->shadow_param_regs) {
-							fp_param++;
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (element_type == IR_I16 || element_type == IR_U16) {
+								|	ASM_TXT_TXT_TMEM_OP vpmullw, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	ASM_TXT_TXT_TMEM_OP vpmulld, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+						} else {
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							if (element_type == IR_I16 || element_type == IR_U16) {
+								|	ASM_TXT_TXT_TMEM_OP vpmullw, xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), oword, IR_MEM_ADD(mem, 16)
+								|	ASM_TXT_TXT_TMEM_OP vpmullw, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+							} else if (element_type == IR_I32 || element_type == IR_U32) {
+								|	ASM_TXT_TXT_TMEM_OP vpmulld, xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), oword, IR_MEM_ADD(mem, 16)
+								|	ASM_TXT_TXT_TMEM_OP vpmulld, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+							} else {
+								IR_ASSERT(0 && "unsupported vector type");
+							}
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
 						}
-						continue;
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
 					}
-				}
-				if (int_param < cc->int_param_regs_count) {
-					dst_reg = cc->int_param_regs[int_param];
 				} else {
-					dst_reg = IR_REG_NONE; /* argument already passed through stack */
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vmulp, element_type, width, def_reg, op1_reg, mem
 				}
-				int_param++;
-				if (cc->shadow_param_regs) {
-					fp_param++;
+				break;
+			case IR_DIV:
+				IR_ASSERT(IR_IS_TYPE_FP(element_type));
+				|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vdivp, element_type, width, def_reg, op1_reg, mem
+				break;
+			case IR_MIN:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						if (element_type == IR_I8) {
+							|	ASM_TXT_TXT_TMEM_OP vpminsb, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_U8) {
+							|	ASM_TXT_TXT_TMEM_OP vpminub, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_I16) {
+							|	ASM_TXT_TXT_TMEM_OP vpminsw, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_U16) {
+							|	ASM_TXT_TXT_TMEM_OP vpminuw, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_I32) {
+							|	ASM_TXT_TXT_TMEM_OP vpminsd, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_U32) {
+							|	ASM_TXT_TXT_TMEM_OP vpminud, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (element_type == IR_I8) {
+							|	ASM_TXT_TXT_TMEM_OP vpminsb, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_U8) {
+							|	ASM_TXT_TXT_TMEM_OP vpminub, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_I16) {
+							|	ASM_TXT_TXT_TMEM_OP vpminsw, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_U16) {
+							|	ASM_TXT_TXT_TMEM_OP vpminuw, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_I32) {
+							|	ASM_TXT_TXT_TMEM_OP vpminsd, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_U32) {
+							|	ASM_TXT_TXT_TMEM_OP vpminud, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vminp, element_type, width, def_reg, op1_reg, mem
 				}
-			} else {
-				IR_ASSERT(IR_IS_TYPE_FP(type));
-				if (fp_param < cc->fp_param_regs_count) {
-					dst_reg = cc->fp_param_regs[fp_param];
+				break;
+			case IR_MAX:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						if (element_type == IR_I8) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxsb, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_U8) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxub, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_I16) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxsw, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_U16) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxuw, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_I32) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxsd, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else if (element_type == IR_U32) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxud, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else if (width == 32) {
+						if (element_type == IR_I8) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxsb, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_U8) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxub, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_I16) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxsw, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_U16) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxuw, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_I32) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxsd, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else if (element_type == IR_U32) {
+							|	ASM_TXT_TXT_TMEM_OP vpmaxud, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else {
+							IR_ASSERT(0 && "unsupported vector type");
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
 				} else {
-					dst_reg = IR_REG_NONE; /* argument already passed through stack */
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vmaxp, element_type, width, def_reg, op1_reg, mem
 				}
-				fp_param++;
-				if (cc->shadow_param_regs) {
-					int_param++;
+				break;
+			case IR_AND:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						|	ASM_TXT_TXT_TMEM_OP vpand, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	ASM_TXT_TXT_TMEM_OP vpand, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else {
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	ASM_TXT_TXT_TMEM_OP vpand, xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), oword, IR_MEM_ADD(mem, 16)
+							|	ASM_TXT_TXT_TMEM_OP vpand, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
+					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vandp, element_type, width, def_reg, op1_reg, mem
 				}
-			}
-			if (dst_reg != IR_REG_NONE) {
-				if (IR_IS_CONST_REF(arg) ||
-				    src_reg == IR_REG_NONE ||
-				    (IR_REG_SPILLED(src_reg) && !IR_REGSET_IN(cc->preserved_regs, IR_REG_NUM(src_reg)))) {
-					if (IR_IS_TYPE_INT(type)) {
-						if (IR_IS_CONST_REF(arg)) {
-							if (type == IR_I8 || type == IR_I16) {
-								type = IR_I32;
-							} else if (type == IR_U8 || type == IR_U16) {
-								type = IR_U32;
-							}
-							ir_emit_load(ctx, type, dst_reg, arg);
-						} else if (ctx->vregs[arg]) {
-							ir_mem mem = ir_ref_spill_slot(ctx, arg);
-
-							if (ir_type_size[type] > 2) {
-								ir_emit_load_mem_int(ctx, type, dst_reg, mem);
-							} else if (ir_type_size[type] == 2) {
-								if (type == IR_I16) {
-									|	ASM_TXT_TMEM_OP movsx, Rd(dst_reg), word, mem
-								} else {
-									|	ASM_TXT_TMEM_OP movzx, Rd(dst_reg), word, mem
-								}
-							} else {
-								IR_ASSERT(ir_type_size[type] == 1);
-								if (type == IR_I8) {
-									|	ASM_TXT_TMEM_OP movsx, Rd(dst_reg), byte, mem
-								} else {
-									|	ASM_TXT_TMEM_OP movzx, Rd(dst_reg), byte, mem
-								}
-							}
+				break;
+			case IR_OR:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						|	ASM_TXT_TXT_TMEM_OP vpor, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	ASM_TXT_TXT_TMEM_OP vpor, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
 						} else {
-							ir_load_local_addr(ctx, dst_reg, arg);
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	ASM_TXT_TXT_TMEM_OP vpor, xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), oword, IR_MEM_ADD(mem, 16)
+							|	ASM_TXT_TXT_TMEM_OP vpor, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
 						}
 					} else {
-						ir_emit_load(ctx, type, dst_reg, arg);
+						IR_ASSERT(0 && "unsupported vector width");
 					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vorp, element_type, width, def_reg, op1_reg, mem
 				}
-			} else {
-				ir_mem mem = IR_MEM_BO(IR_REG_STACK_POINTER, stack_offset);
-
-				if (IR_IS_TYPE_INT(type)) {
-					if (IR_IS_CONST_REF(arg)) {
-						ir_emit_store_mem_int_const(ctx, type, mem, arg, tmp_reg, 1);
-					} else if (src_reg == IR_REG_NONE) {
-						IR_ASSERT(tmp_reg != IR_REG_NONE);
-						ir_emit_load(ctx, type, tmp_reg, arg);
-						ir_emit_store_mem_int(ctx, type, mem, tmp_reg);
-					} else if (IR_REG_SPILLED(src_reg)) {
-						src_reg = IR_REG_NUM(src_reg);
-						ir_emit_load(ctx, type, src_reg, arg);
-						ir_emit_store_mem_int(ctx, type, mem, src_reg);
+				break;
+			case IR_XOR:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (width == 16) {
+						|	ASM_TXT_TXT_TMEM_OP vpxor, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+					} else if (width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	ASM_TXT_TXT_TMEM_OP vpxor, ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), yword, mem
+						} else {
+							|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	ASM_TXT_TXT_TMEM_OP vpxor, xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), oword, IR_MEM_ADD(mem, 16)
+							|	ASM_TXT_TXT_TMEM_OP vpxor, xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), oword, mem
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(0 && "unsupported vector width");
 					}
 				} else {
-					if (IR_IS_CONST_REF(arg)) {
-						ir_emit_store_mem_fp_const(ctx, type, mem, arg, tmp_reg, tmp_fp_reg);
-					} else if (src_reg == IR_REG_NONE) {
-						IR_ASSERT(tmp_fp_reg != IR_REG_NONE);
-						ir_emit_load(ctx, type, tmp_fp_reg, arg);
-						ir_emit_store_mem_fp(ctx, IR_DOUBLE, mem, tmp_fp_reg);
-					} else if (IR_REG_SPILLED(src_reg)) {
-						src_reg = IR_REG_NUM(src_reg);
-						ir_emit_load(ctx, type, src_reg, arg);
-						ir_emit_store_mem_fp(ctx, type, mem, src_reg);
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_OP vxorp, element_type, width, def_reg, op1_reg, mem
+				}
+				break;
+			case IR_EQ:
+				if (IR_IS_TYPE_INT(element_type)) {
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpcmpeq, element_type, 16, tmp_reg, tmp_reg, IR_MEM_ADD(mem, 16)
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpcmpeq, element_type, 16, def_reg, op1_reg, mem
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpcmpeq, element_type, width, def_reg, op1_reg, mem
 					}
+				} else {
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, mem, 0
 				}
-				stack_offset += IR_MAX(sizeof(void*), ir_type_size[type]);
-			}
-		}
-	}
-
-	/* WIN64 calling convention requires duplcation of parameters passed in FP register into GP ones */
-	if (proto && (proto->flags & IR_VARARG_FUNC) && cc->shadow_param_regs) {
-		n = IR_MIN(n, IR_MIN(cc->int_param_regs_count, cc->fp_param_regs_count) + 2);
-		for (j = 3; j <= n; j++) {
-			arg = ir_insn_op(insn, j);
-			arg_insn = &ctx->ir_base[arg];
-			type = arg_insn->type;
-			if (IR_IS_TYPE_FP(type)) {
-				src_reg = cc->fp_param_regs[j-3];
-				dst_reg = cc->int_param_regs[j-3];
-|.if X64
-				if (ctx->mflags & IR_X86_AVX) {
-					|	vmovd Rq(dst_reg), xmm(src_reg-IR_REG_FP_FIRST)
+				break;
+			case IR_NE:
+				if (IR_IS_TYPE_INT(element_type)) {
+					IR_ASSERT(tmp_reg != IR_REG_NONE);
+					if (!(ctx->mflags & IR_X86_AVX2) && width == 32) {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpcmpeq, element_type, 16, tmp2_reg, tmp_reg, IR_MEM_ADD(mem, 16)
+						|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, tmp2_reg, tmp2_reg, tmp_reg
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpcmpeq, element_type, 16, def_reg, op1_reg, mem
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, 16, def_reg, def_reg, tmp_reg
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+					} else {
+						|	ASM_AVX_INT_VEC_REG_REG_MEM_OP vpcmpeq, element_type, width, def_reg, op1_reg, mem
+						if (width == 16) {
+							|	vpxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						} else if (width == 32) {
+							|	vpxor ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST)
+						} else {
+							IR_ASSERT(0 && "unsupported vector width");
+						}
+						|	ASM_AVX_INT_VEC_REG_REG_REG_OP vpcmpeq, element_type, width, def_reg, def_reg, tmp_reg
+					}
 				} else {
-					|	movd Rq(dst_reg), xmm(src_reg-IR_REG_FP_FIRST)
+					IR_ASSERT(IR_IS_TYPE_FP(element_type));
+					|	ASM_AVX_FP_VEC_REG_REG_MEM_TXT_OP vcmpp, element_type, width, def_reg, op1_reg, mem, 4
 				}
-|.endif
-			}
+				break;
 		}
 	}
-
-	if (insn->op == IR_CALL && (ctx->flags2 & IR_PREALLOCATED_STACK)) {
-		used_stack = 0;
-	}
-
-	if (proto && (proto->flags & IR_VARARG_FUNC) && cc->fp_varargs_reg != IR_REG_NONE) {
-		/* set hidden argument to specify the number of vector registers used */
-		fp_param = IR_MIN(fp_param, cc->fp_param_regs_count);
-		if (fp_param) {
-			|	mov Rd(cc->fp_varargs_reg), fp_param
-		} else {
-			|	xor Rd(cc->fp_varargs_reg), Rd(cc->fp_varargs_reg)
-		}
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, insn->type, def, def_reg);
 	}
-
-	return used_stack;
 }

-static void ir_emit_call_ex(ir_ctx *ctx, ir_ref def, ir_insn *insn, const ir_proto_t *proto, const ir_call_conv_dsc *cc, int32_t used_stack)
+static void ir_emit_vector_binop_expand(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg def_reg;
-	ir_ref func = insn->op2;
+	ir_type type = insn->type;
+	ir_type element_type;
+	uint32_t element_size, width, count, i;
+	ir_ref op1 = insn->op1;
+	ir_ref op2 = insn->op2;
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg op2_reg = ctx->regs[def][2];
+	ir_reg tmp1_reg = ctx->regs[def][3];
+#if IR_X86_I64
+	ir_reg tmp1_reg_hi = IR_REG_NONE;
+#endif
+	ir_reg def_reg = ctx->regs[def][0];
+	int offset = 0;
+	ir_mem src1_mem, src2_mem, dst_mem;
+	bool scalar_shift = 0;
+	int shift_count = 0;
+	void *ptr1 = NULL;
+	void *ptr2 = NULL;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(type));
+	if (insn->op >= IR_EQ && insn->op <= IR_UNORDERED) {
+		type = ctx->ir_base[op1].type;
+	} else {
+		IR_ASSERT(type == ctx->ir_base[op1].type);
+	}

-	if (!IR_IS_CONST_REF(func) && ctx->rules[func] == (IR_FUSED | IR_SIMPLE | IR_PROTO)) {
-		func = ctx->ir_base[func].op1;
+	element_type = IR_VECTOR_BASE_TYPE(type);
+	element_size = ir_type_size[element_type];
+	width = IR_VECTOR_SIZE(type);
+	count = IR_VECTOR_LENGTH(type);
+
+	if (def_reg == IR_REG_NONE) {
+		dst_mem = ir_ref_spill_slot(ctx, def);
+	} else if (IR_REG_SPILLED(def_reg)) {
+		dst_mem = ir_ref_spill_slot(ctx, def);
+		def_reg = IR_REG_NUM(def_reg);
+	} else {
+		offset = -width;
+		dst_mem = IR_MEM(IR_REG_RSP, offset, IR_REG_NONE, 1);
 	}
-	if (IR_IS_CONST_REF(func)) {
-		void *addr = ir_call_addr(ctx, insn, &ctx->ir_base[func]);

-		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
-			|	call aword &addr
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			src1_mem = ir_ref_spill_slot(ctx, op1);
+			ir_emit_load(ctx, type, op1_reg, op1);
 		} else {
-|.if X64
-||			ir_reg tmp_reg = cc->int_ret_reg;
-||
-||			if (proto && (proto->flags & IR_VARARG_FUNC) && tmp_reg == cc->fp_varargs_reg) {
-||				tmp_reg = IR_REG_R11; // TODO: avoid usage of hardcoded temporary register ???
-||			}
-||			if (IR_IS_SIGNED_32BIT(addr)) {
-				|	mov Rq(tmp_reg), ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
-||			} else {
-				|	mov64 Rq(tmp_reg), ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
-||			}
-			|	call Rq(tmp_reg)
-|.endif
+			if (offset != 0) {
+				src1_mem = dst_mem;
+			} else {
+				offset = -width;
+				src1_mem = IR_MEM(IR_REG_RSP, offset, IR_REG_NONE, 1);
+			}
+			ir_emit_store_mem_fp(ctx, type, src1_mem, op1_reg);
 		}
-    } else {
-		ir_reg op2_reg = ctx->regs[def][2];
+	} else if (IR_IS_CONST_REF(op1)) {
+		src1_mem = IR_MEM_NONE;
+		ptr1 = ir_long_const_ptr(ctx, op1);
+	} else {
+		if (ir_rule(ctx, op1) & IR_FUSED) {
+			src1_mem = ir_fuse_load(ctx, def, op1);
+		} else {
+			src1_mem = ir_ref_spill_slot(ctx, op1);
+		}
+	}

+	if (IR_IS_TYPE_INT(ctx->ir_base[op2].type)) {
+		ir_type shift_type;
+
+		IR_ASSERT(insn->op == IR_SHL || insn->op == IR_SHR || insn->op == IR_SAR);
+		src2_mem = IR_MEM_NONE;
+		scalar_shift = 1;
+		shift_type = ctx->ir_base[op2].type;
 		if (op2_reg != IR_REG_NONE) {
 			if (IR_REG_SPILLED(op2_reg)) {
 				op2_reg = IR_REG_NUM(op2_reg);
-				ir_emit_load(ctx, IR_ADDR, op2_reg, func);
+				ir_emit_load(ctx, shift_type, op2_reg, op2);
+			}
+#if IR_X86_I64
+			if (ctx->ir_base[op2].type == IR_I64 || ctx->ir_base[op2].type == IR_U64) {
+				/* ignore the high part of the register */
+				op2_reg = IR_REG_I64_LO(op2_reg);
+				if (op2_reg != IR_REG_RCX) {
+					ir_emit_mov(ctx, IR_U32, IR_REG_RCX, op2_reg);
+				}
+			} else
+#endif
+			if (op2_reg != IR_REG_RCX) {
+				ir_emit_mov(ctx, shift_type, IR_REG_RCX, op2_reg);
+				op2_reg = IR_REG_RCX;
 			}
-			|	call Ra(op2_reg)
+		} else if (IR_IS_CONST_REF(op2)) {
+			shift_count = ctx->ir_base[op2].val.i32;
 		} else {
 			ir_mem mem;

-			if (ir_rule(ctx, func) & IR_FUSED) {
-				mem = ir_fuse_load(ctx, def, func);
+			if (ir_rule(ctx, op2) & IR_FUSED) {
+				mem = ir_fuse_load(ctx, def, op2);
 			} else {
-				mem = ir_ref_spill_slot(ctx, func);
+				mem = ir_ref_spill_slot(ctx, op2);
 			}
-
-			|	ASM_TMEM_OP call, aword, mem
+#if IR_X86_I64
+			if (ctx->ir_base[op2].type == IR_I64 || ctx->ir_base[op2].type == IR_U64) {
+				/* ignore the high part of the register */
+				ir_emit_load_mem(ctx, IR_U32, IR_REG_RCX, mem);
+			} else
+#endif
+			ir_emit_load_mem(ctx, shift_type, IR_REG_RCX, mem);
+			op2_reg = IR_REG_RCX;
 		}
-    }
-
-	if (used_stack) {
-		int32_t aligned_stack = IR_ALIGNED_SIZE(used_stack, 16);
-
-		ctx->call_stack_size -= aligned_stack;
-		if (cc->cleanup_stack_by_callee) {
-			aligned_stack -= used_stack;
-			if (aligned_stack) {
-				|	add Ra(IR_REG_RSP), aligned_stack
+	} else {
+		IR_ASSERT(type == ctx->ir_base[op2].type);
+		if (op1 == op2) {
+			op2_reg = op1_reg;
+			src2_mem = src1_mem;
+			ptr2 = ptr1;
+		} else if (op2_reg != IR_REG_NONE) {
+			if (IR_REG_SPILLED(op2_reg)) {
+				op2_reg = IR_REG_NUM(op2_reg);
+				src2_mem = ir_ref_spill_slot(ctx, op2);
+				ir_emit_load(ctx, type, op2_reg, op2);
+			} else {
+				offset -= width;
+				src2_mem = IR_MEM(IR_REG_RSP, offset, IR_REG_NONE, 1);
+				ir_emit_store_mem_fp(ctx, type, src2_mem, op2_reg);
 			}
+		} else if (IR_IS_CONST_REF(op2)) {
+			src2_mem = IR_MEM_NONE;
+			ptr2 = ir_long_const_ptr(ctx, op2);
 		} else {
-			|	add Ra(IR_REG_RSP), aligned_stack
+			if (ir_rule(ctx, op2) & IR_FUSED) {
+				src2_mem = ir_fuse_load(ctx, def, op2);
+			} else {
+				src2_mem = ir_ref_spill_slot(ctx, op2);
+			}
 		}
 	}

-	if (insn->type != IR_VOID) {
-		if (IR_IS_TYPE_INT(insn->type)) {
-			def_reg = IR_REG_NUM(ctx->regs[def][0]);
-			if (def_reg != IR_REG_NONE) {
-				if (def_reg != cc->int_ret_reg) {
-					ir_emit_mov(ctx, insn->type, def_reg, cc->int_ret_reg);
+	if (tmp1_reg == IR_REG_NONE) {
+		/* hardcoded temporary for integer vectors */
+		tmp1_reg = IR_REG_RAX;
+#if IR_X86_I64
+		if (element_type == IR_I64 || element_type == IR_U64) {
+			tmp1_reg_hi = IR_REG_RDX;
+		}
+#endif
+	}
+
+	IR_ASSERT(IR_IS_TYPE_INT(element_type));
+
+	for (i = 0; i < count; i++) {
+		if (IR_IS_CONST_REF(op1)) {
+			int64_t val;
+
+			if (element_size == 8) {
+				val = ((int64_t*)ptr1)[i];
+			} else if (element_size == 4) {
+				if (IR_IS_TYPE_SIGNED(element_type)) {
+					val = ((int32_t*)ptr1)[i];
+				} else {
+					val = ((uint32_t*)ptr1)[i];
 				}
-				if (IR_REG_SPILLED(ctx->regs[def][0])) {
-					ir_emit_store(ctx, insn->type, def, def_reg);
+			} else if (element_size == 2) {
+				if (IR_IS_TYPE_SIGNED(element_type)) {
+					val = ((int16_t*)ptr1)[i];
+				} else {
+					val = ((uint16_t*)ptr1)[i];
+				}
+			} else {
+				IR_ASSERT(element_size == 1);
+				if (IR_IS_TYPE_SIGNED(element_type)) {
+					val = ((int8_t*)ptr1)[i];
+				} else {
+					val = ((uint8_t*)ptr1)[i];
 				}
-			} else if (ctx->use_lists[def].count > 1) {
-				ir_emit_store(ctx, insn->type, def, cc->int_ret_reg);
 			}
+#if IR_X86_I64
+			if (element_type == IR_I64 || element_type == IR_U64) {
+				ir_emit_load_imm_int(ctx, IR_U32, tmp1_reg, (uint32_t)(val & 0xffffffff));
+				ir_emit_load_imm_int(ctx, IR_U32, tmp1_reg_hi, (uint32_t)(val >> 32));
+			} else
+#endif
+			ir_emit_load_imm_int(ctx, element_type, tmp1_reg, val);
+#if IR_X86_I64
+		} else if (element_type == IR_I64 || element_type == IR_U64) {
+			ir_emit_load_mem(ctx, IR_U32, tmp1_reg, IR_MEM_ADD(src1_mem, element_size * i));
+			ir_emit_load_mem(ctx, IR_U32, tmp1_reg_hi, IR_MEM_ADD(src1_mem, element_size * i + 4));
+#endif
 		} else {
-			IR_ASSERT(IR_IS_TYPE_FP(insn->type));
-			def_reg = IR_REG_NUM(ctx->regs[def][0]);
-			if (cc->fp_ret_reg != IR_REG_NONE) {
-				if (def_reg != IR_REG_NONE) {
-					if (def_reg != cc->fp_ret_reg) {
-						ir_emit_fp_mov(ctx, insn->type, def_reg, cc->fp_ret_reg);
+			ir_emit_load_mem(ctx, element_type, tmp1_reg, IR_MEM_ADD(src1_mem, element_size * i));
+		}
+
+		if (IR_IS_CONST_REF(op2)) {
+			int val = 0;
+#if IR_X86_I64
+			uint32_t val_hi = 0;
+			void *addr = NULL;
+#endif
+			ir_reg tmp2_reg = IR_REG_NONE;
+
+			if (!scalar_shift) {
+				if (element_size == 8) {
+					if (IR_IS_SIGNED_32BIT(((int64_t*)ptr2)[i])) {
+						val = ((int64_t*)ptr2)[i];
+					} else {
+#if IR_X86_I64
+						val_hi = ((uint64_t*)ptr2)[i] >> 32;
+#else
+						tmp2_reg = IR_REG_RCX;
+						ir_emit_load_imm_int(ctx, element_type, tmp2_reg, ((int64_t*)ptr2)[i]);
+#endif
 					}
-					if (IR_REG_SPILLED(ctx->regs[def][0])) {
-						ir_emit_store(ctx, insn->type, def, def_reg);
+				} else if (element_size == 4) {
+					val = ((int32_t*)ptr2)[i];
+				} else if (element_size == 2) {
+					if (IR_IS_TYPE_SIGNED(element_type)) {
+						val = ((int16_t*)ptr2)[i];
+					} else {
+						val = ((uint16_t*)ptr2)[i];
 					}
-				} else if (ctx->use_lists[def].count > 1) {
-					ir_emit_store(ctx, insn->type, def, cc->fp_ret_reg);
+				} else {
+					IR_ASSERT(element_size == 1);
+					if (IR_IS_TYPE_SIGNED(element_type)) {
+						val = ((int8_t*)ptr2)[i];
+					} else {
+						val = ((uint8_t*)ptr2)[i];
+					}
+				}
+#if IR_X86_I64
+				if (element_type == IR_I64 || element_type == IR_U64) {
+					/* pass */
+			    } else
+#endif
+				if (tmp2_reg == IR_REG_NONE
+				 && (insn->op == IR_DIV || insn->op == IR_MOD || (insn->op == IR_MUL && element_size == 1))) {
+					tmp2_reg = IR_REG_RCX;
+					ir_emit_load_imm_int(ctx, element_type, tmp2_reg, val);
+			    }
+			} else {
+				val = shift_count;
+			}
+
+
+			if (tmp2_reg != IR_REG_NONE) {
+				switch (insn->op) {
+					default:
+						IR_ASSERT(0 && "NIY binary op");
+					case IR_ADD:
+						|	ASM_REG_REG_OP add, element_type, tmp1_reg, tmp2_reg
+						break;
+					case IR_SUB:
+						|	ASM_REG_REG_OP sub, element_type, tmp1_reg, tmp2_reg
+						break;
+					case IR_MUL:
+						if (element_size != 1) {
+							|	ASM_REG_REG_MUL imul, element_type, tmp1_reg, tmp2_reg
+						} else {
+							if (IR_IS_TYPE_SIGNED(element_type)) {
+								|	imul Rb(tmp2_reg)
+							} else {
+								|	mul Rb(tmp2_reg);
+							}
+						}
+						break;
+					case IR_DIV:
+					case IR_MOD:
+						if (IR_IS_TYPE_SIGNED(element_type)) {
+							if (element_size == 8) {
+								|	cqo
+							} else if (element_size == 4) {
+								|	cdq
+							} else if (element_size == 2) {
+								|	cwd
+							} else {
+								|	cbw
+							}
+						} else if (element_size == 1) {
+							|	movzx ax, al
+						} else {
+							|	ASM_REG_REG_OP xor, element_type, IR_REG_RDX, IR_REG_RDX
+						}
+						if (IR_IS_TYPE_SIGNED(element_type)) {
+							|	ASM_REG_OP idiv, element_type, tmp2_reg
+						} else {
+							|	ASM_REG_OP div, element_type, tmp2_reg
+						}
+						if (insn->op == IR_MOD) {
+							if (element_size == 1) {
+								|	mov al, ah
+							} else {
+								ir_emit_mov(ctx, element_type, tmp1_reg, IR_REG_RDX);
+							}
+						}
+						break;
+					case IR_OR:
+						|	ASM_REG_IMM_OP or, element_type, tmp1_reg, val
+						break;
+					case IR_AND:
+						|	ASM_REG_IMM_OP and, element_type, tmp1_reg, val
+						break;
+					case IR_XOR:
+						|	ASM_REG_IMM_OP xor, element_type, tmp1_reg, val
+						break;
+					case IR_SHL:
+						|	ASM_REG_IMM_OP shl, element_type, tmp1_reg, val
+						break;
+					case IR_SHR:
+						|	ASM_REG_IMM_OP shr, element_type, tmp1_reg, val
+						break;
+					case IR_SAR:
+						|	ASM_REG_IMM_OP sar, element_type, tmp1_reg, val
+						break;
+					case IR_EQ:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	sete Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_NE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setne Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_LT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setl Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_LE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setle Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_GE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setge Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_GT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setg Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_ULT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setb Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_ULE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setbe Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_UGE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setae Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_UGT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	seta Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+				}
+#if IR_X86_I64
+|.if not X64
+			} else if (element_type == IR_I64 || element_type == IR_U64) {
+				switch (insn->op) {
+					default:
+						IR_ASSERT(0 && "NIY binary op");
+					case IR_MUL:
+						if (val_hi && val) {
+							|	imul edx, val
+							|	imul ecx, eax, val_hi
+							|	add ecx, edx
+							|	mov edx, val
+							|	mul edx
+							|	add edx, ecx
+						} else if (val_hi && !val) {
+							|	imul edx, eax, val_hi
+							|	xor eax, eax
+						} else {
+							|	imul ecx, edx, val
+							|	mov edx, val
+							|	mul edx
+							|	add edx, ecx
+						}
+						break;
+					case IR_DIV:
+					case IR_MOD:
+						|	sub esp, (width*3)+12
+						|	push val_hi
+						|	push val
+						|	push Rd(tmp1_reg_hi)
+						|	push Rd(tmp1_reg)
+						if (insn->op == IR_DIV) {
+							if (element_type == IR_I64) {
+								addr = __divdi3;
+							} else {
+								IR_ASSERT(element_type == IR_U64);
+								addr = __udivdi3;
+							}
+						} else {
+							IR_ASSERT(insn->op == IR_MOD);
+							if (element_type == IR_I64) {
+								addr = __moddi3;
+							} else {
+								IR_ASSERT(element_type == IR_U64);
+								addr = __umoddi3;
+							}
+						}
+						|	call aword &addr
+#ifdef _WIN32
+						|	add esp, (width*3)+12
+#else
+						|	add esp, (width*3)+28
+#endif
+						break;
+					case IR_SHL:
+						|	shld Rd(tmp1_reg_hi), Rd(tmp1_reg), val
+						|	shl	Rd(tmp1_reg), val
+						if (val & 32) {
+							|	mov Rd(tmp1_reg_hi), Rd(tmp1_reg)
+							|	xor	Rd(tmp1_reg), Rd(tmp1_reg)
+						}
+						break;
+					case IR_SHR:
+						|	shrd Rd(tmp1_reg), Rd(tmp1_reg_hi), val
+						|	shr	Rd(tmp1_reg_hi), val
+						if (val & 32) {
+							|	mov Rd(tmp1_reg), Rd(tmp1_reg_hi)
+							|	xor	Rd(tmp1_reg_hi), Rd(tmp1_reg_hi)
+						}
+						break;
+					case IR_SAR:
+						|	shrd Rd(tmp1_reg), Rd(tmp1_reg_hi), val
+						|	sar	Rd(tmp1_reg_hi), val
+						if (val & 32) {
+							|	mov Rd(tmp1_reg), Rd(tmp1_reg_hi)
+							|	sar	Rd(tmp1_reg_hi), 31
+						}
+						break;
+				}
+|.endif
+#endif
+			} else {
+				switch (insn->op) {
+					default:
+						IR_ASSERT(0 && "NIY binary op");
+					case IR_ADD:
+						|	ASM_REG_IMM_OP add, element_type, tmp1_reg, val
+						break;
+					case IR_SUB:
+						|	ASM_REG_IMM_OP sub, element_type, tmp1_reg, val
+						break;
+					case IR_MUL:
+						IR_ASSERT(element_size != 1);
+						|	ASM_REG_IMM_MUL imul, element_type, tmp1_reg, val
+						break;
+					case IR_OR:
+						|	ASM_REG_IMM_OP or, element_type, tmp1_reg, val
+						break;
+					case IR_AND:
+						|	ASM_REG_IMM_OP and, element_type, tmp1_reg, val
+						break;
+					case IR_XOR:
+						|	ASM_REG_IMM_OP xor, element_type, tmp1_reg, val
+						break;
+					case IR_SHL:
+						|	ASM_REG_IMM_OP shl, element_type, tmp1_reg, val
+						break;
+					case IR_SHR:
+						|	ASM_REG_IMM_OP shr, element_type, tmp1_reg, val
+						break;
+					case IR_SAR:
+						|	ASM_REG_IMM_OP sar, element_type, tmp1_reg, val
+						break;
+					case IR_EQ:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	sete Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_NE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setne Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_LT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setl Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_LE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setle Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_GE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setge Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_GT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setg Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_ULT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setb Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_ULE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setbe Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_UGE:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	setae Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
+					case IR_UGT:
+						|	ASM_REG_IMM_OP cmp, element_type, tmp1_reg, val
+						|	seta Rb(tmp1_reg)
+						if (element_size != 1) {
+							|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+						}
+						|	ASM_REG_OP neg, element_type, tmp1_reg
+						break;
 				}
 			}
-#ifdef IR_TARGET_X86
-			if (ctx->use_lists[def].count > 1 && cc->fp_ret_reg == IR_REG_NONE) {
-				int32_t offset;
-				ir_reg fp;
+#if IR_X86_I64
+|.if not X64
+		} else if (element_type == IR_I64 || element_type == IR_U64) {
+			ir_mem mem = IR_MEM_NONE;
+			void *addr = NULL;

-				if (def_reg == IR_REG_NONE) {
-					offset = ir_ref_spill_slot_offset(ctx, def, &fp);
-					if (insn->type == IR_DOUBLE) {
-						|	fstp qword [Ra(fp)+offset]
+			if (!scalar_shift) {
+				mem = IR_MEM_ADD(src2_mem, element_size * i);
+		    }
+
+			switch (insn->op) {
+				default:
+					IR_ASSERT(0 && "NIY binary op");
+				case IR_MUL:
+					|	ASM_TXT_TMEM_OP mov, ecx, dword, IR_MEM_I64_HI(mem)
+					|	ASM_TXT_TMEM_OP imul, edx, dword, mem
+					|	imul ecx, eax
+					|	add ecx, edx
+					|	ASM_TMEM_OP mul, dword, mem
+					|	add edx, ecx
+					break;
+				case IR_DIV:
+				case IR_MOD:
+					|	sub esp, (width*3)+12
+					if (IR_MEM_BASE(mem) == IR_REG_RSP) {
+						mem = IR_MEM_ADD(mem, ((width*3)+16));
+						|	ASM_MEM_PUSH_OP push, IR_U32, mem
+						|	ASM_MEM_PUSH_OP push, IR_U32, mem
 					} else {
-						IR_ASSERT(insn->type == IR_FLOAT);
-						|	fstp dword [Ra(fp)+offset]
+						|	ASM_MEM_PUSH_OP push, IR_U32, IR_MEM_ADD(mem, 4)
+						|	ASM_MEM_PUSH_OP push, IR_U32, mem
 					}
-				} else {
-					offset = ctx->ret_slot;
-					IR_ASSERT(offset != -1);
-					offset = IR_SPILL_POS_TO_OFFSET(offset);
-					fp = (ctx->flags & IR_USE_FRAME_POINTER) ? IR_REG_FRAME_POINTER : IR_REG_STACK_POINTER;
-					if (insn->type == IR_DOUBLE) {
-						|	fstp qword [Ra(fp)+offset]
+					|	push Rd(tmp1_reg_hi)
+					|	push Rd(tmp1_reg)
+					if (insn->op == IR_DIV) {
+						if (element_type == IR_I64) {
+							addr = __divdi3;
+						} else {
+							IR_ASSERT(element_type == IR_U64);
+							addr = __udivdi3;
+						}
 					} else {
-						IR_ASSERT(insn->type == IR_FLOAT);
-						|	fstp dword [Ra(fp)+offset]
-					}
-					ir_emit_load_mem_fp(ctx, insn->type, def_reg, IR_MEM_BO(fp, offset));
-					if (IR_REG_SPILLED(ctx->regs[def][0])) {
-						ir_emit_store(ctx, insn->type, def, def_reg);
+						IR_ASSERT(insn->op == IR_MOD);
+						if (element_type == IR_I64) {
+							addr = __moddi3;
+						} else {
+							IR_ASSERT(element_type == IR_U64);
+							addr = __umoddi3;
+						}
 					}
-				}
-			}
+					|	call aword &addr
+#ifdef _WIN32
+					|	add esp, (width*3)+12
+#else
+					|	add esp, (width*3)+28
 #endif
-		}
-	}
-}
-
-static void ir_emit_call(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	const ir_proto_t *proto = ir_call_proto(ctx, insn);
-	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
-	int32_t used_stack = ir_emit_arguments(ctx, def, insn, proto, cc, ctx->regs[def][1]);
-	ir_emit_call_ex(ctx, def, insn, proto, cc, used_stack);
-}
-
-static void ir_emit_tailcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	const ir_proto_t *proto = ir_call_proto(ctx, insn);
-	const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
-	int32_t used_stack = ir_emit_arguments(ctx, def, insn, proto, cc, ctx->regs[def][1]);
-	ir_ref func = insn->op2;
-
-	if (used_stack != 0) {
-		ir_emit_call_ex(ctx, def, insn, proto, cc, used_stack);
-		ir_emit_return_void(ctx);
-		return;
-	}
-
-	/* Move op2 to a tmp register before epilogue if it's in
-	 * used_preserved_regs, because it will be overridden. */
-
-	ir_reg op2_reg = IR_REG_NONE;
-	ir_mem mem = IR_MEM_B(IR_REG_NONE);
-	if (!IR_IS_CONST_REF(func) && ctx->rules[func] == (IR_FUSED | IR_SIMPLE | IR_PROTO)) {
-		func = ctx->ir_base[func].op1;
-	}
-	if (!IR_IS_CONST_REF(func)) {
-		op2_reg = ctx->regs[def][2];
-
-		ir_regset preserved_regs = (ir_regset)ctx->used_preserved_regs | IR_REGSET(IR_REG_STACK_POINTER);
-		if (ctx->flags & IR_USE_FRAME_POINTER) {
-			preserved_regs |= IR_REGSET(IR_REG_FRAME_POINTER);
-		}
-
-		bool is_spill_slot = op2_reg != IR_REG_NONE
-			&& IR_REG_SPILLED(op2_reg)
-			&& ctx->vregs[func];
-
-		if (op2_reg != IR_REG_NONE && !is_spill_slot) {
-			if (IR_REGSET_IN(preserved_regs, IR_REG_NUM(op2_reg))) {
-				ir_ref orig_op2_reg = op2_reg;
-				op2_reg = IR_REG_RAX;
-
-				if (IR_REG_SPILLED(orig_op2_reg)) {
-					ir_emit_load(ctx, IR_ADDR, op2_reg, func);
-				} else {
-					ir_type type = ctx->ir_base[func].type;
-					| ASM_REG_REG_OP mov, type, op2_reg, IR_REG_NUM(orig_op2_reg)
-				}
-			} else {
-				op2_reg = IR_REG_NUM(op2_reg);
-			}
-		} else {
-			if (ir_rule(ctx, func) & IR_FUSED) {
-				IR_ASSERT(op2_reg == IR_REG_NONE);
-				mem = ir_fuse_load(ctx, def, func);
-			} else {
-				mem = ir_ref_spill_slot(ctx, func);
-			}
-			ir_reg base = IR_MEM_BASE(mem);
-			ir_reg index = IR_MEM_INDEX(mem);
-			if ((base != IR_REG_NONE && IR_REGSET_IN(preserved_regs, base)) ||
-					(index != IR_REG_NONE && IR_REGSET_IN(preserved_regs, index))) {
-				op2_reg = IR_REG_RAX;
-
-				ir_type type = ctx->ir_base[func].type;
-				ir_emit_load_mem_int(ctx, type, op2_reg, mem);
-			} else {
-				op2_reg = IR_REG_NONE;
+					break;
+				case IR_SHL:
+					if (!scalar_shift) {
+						ir_emit_load_mem_int(ctx, IR_U32, IR_REG_RCX, mem);
+					}
+					|	shld Rd(tmp1_reg_hi), Rd(tmp1_reg), cl
+					|	shl	Rd(tmp1_reg), cl
+					|	test cl, 32
+					|	je	>1
+					|	mov Rd(tmp1_reg_hi), Rd(tmp1_reg)
+					|	xor	Rd(tmp1_reg), Rd(tmp1_reg)
+					|1:
+					break;
+				case IR_SHR:
+					if (!scalar_shift) {
+						ir_emit_load_mem_int(ctx, IR_U32, IR_REG_RCX, mem);
+					}
+					|	shrd Rd(tmp1_reg), Rd(tmp1_reg_hi), cl
+					|	shr	Rd(tmp1_reg_hi), cl
+					|	test cl, 32
+					|	je	>1
+					|	mov Rd(tmp1_reg), Rd(tmp1_reg_hi)
+					|	xor	Rd(tmp1_reg_hi), Rd(tmp1_reg_hi)
+					|1:
+					break;
+				case IR_SAR:
+					if (!scalar_shift) {
+						ir_emit_load_mem_int(ctx, IR_U32, IR_REG_RCX, mem);
+					}
+					|	shrd Rd(tmp1_reg), Rd(tmp1_reg_hi), cl
+					|	sar	Rd(tmp1_reg_hi), cl
+					|	test cl, 32
+					|	je	>1
+					|	mov Rd(tmp1_reg), Rd(tmp1_reg_hi)
+					|	sar	Rd(tmp1_reg_hi), 31
+					|1:
+					break;
 			}
-		}
-	}
-
-	ir_emit_epilogue(ctx);
-
-	if (IR_IS_CONST_REF(func)) {
-		void *addr = ir_call_addr(ctx, insn, &ctx->ir_base[func]);
-
-		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
-			|	jmp aword &addr
-		} else {
-|.if X64
-||			ir_reg tmp_reg = cc->int_ret_reg;
-||
-||			if (proto && (proto->flags & IR_VARARG_FUNC) && tmp_reg == cc->fp_varargs_reg) {
-||				tmp_reg = IR_REG_R11; // TODO: avoid usage of hardcoded temporary register ???
-||			}
-||			if (IR_IS_SIGNED_32BIT(addr)) {
-				|	mov Rq(tmp_reg), ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
-||			} else {
-				|	mov64 Rq(tmp_reg), ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
-||			}
-			|	jmp Rq(tmp_reg)
 |.endif
-		}
-    } else {
-		if (op2_reg != IR_REG_NONE) {
-			IR_ASSERT(!IR_REGSET_IN((ir_regset)ctx->used_preserved_regs, op2_reg));
-			|	jmp Ra(op2_reg)
+#endif
 		} else {
-			|	ASM_TMEM_OP jmp, aword, mem
-		}
-    }
-}
-
-static void ir_emit_ijmp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_reg op2_reg = ctx->regs[def][2];
-
-	if (IR_IS_CONST_REF(insn->op2)) {
-		if (ctx->ir_base[insn->op2].op == IR_LABEL) {
-			if (!data->resolved_label_syms) {
-				data->resolved_label_syms = 1;
-				ir_resolve_label_syms(ctx);
-			}
-
-			uint32_t target = ctx->ir_base[insn->op2].val.u32_hi;
-			target = ir_skip_empty_target_blocks(ctx, target);
+			ir_mem mem = IR_MEM_NONE;

-			|	jmp =>target
-			return;
-		}
-
-		void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op2]);
+			if (!scalar_shift) {
+				mem = IR_MEM_ADD(src2_mem, element_size * i);
+		    }

-		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
-			|	jmp aword &addr
-		} else {
-|.if X64
-			if (IR_IS_SIGNED_32BIT(addr)) {
-				|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
-			} else {
-				|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+			switch (insn->op) {
+				default:
+					IR_ASSERT(0 && "NIY binary op");
+				case IR_ADD:
+					|	ASM_REG_MEM_OP add, element_type, tmp1_reg, mem
+					break;
+				case IR_SUB:
+					|	ASM_REG_MEM_OP sub, element_type, tmp1_reg, mem
+					break;
+				case IR_MUL:
+					if (element_size != 1) {
+						|	ASM_REG_MEM_MUL imul, element_type, tmp1_reg, mem
+					} else {
+						if (IR_IS_TYPE_SIGNED(element_type)) {
+							|	ASM_MEM_OP imul, IR_I8, mem
+						} else {
+							|	ASM_MEM_OP mul, IR_U8, mem
+						}
+					}
+					break;
+				case IR_DIV:
+				case IR_MOD:
+					if (IR_IS_TYPE_SIGNED(element_type)) {
+						if (element_size == 8) {
+							|	cqo
+						} else if (element_size == 4) {
+							|	cdq
+						} else if (element_size == 2) {
+							|	cwd
+						} else {
+							|	cbw
+						}
+					} else if (element_size == 1) {
+						|	movzx ax, al
+					} else {
+						|	ASM_REG_REG_OP xor, element_type, IR_REG_RDX, IR_REG_RDX
+					}
+					if (IR_IS_TYPE_SIGNED(element_type)) {
+						|	ASM_MEM_OP idiv, element_type, mem
+					} else {
+						|	ASM_MEM_OP div, element_type, mem
+					}
+					if (insn->op == IR_MOD) {
+						if (element_size == 1) {
+							|	mov al, ah
+						} else {
+							ir_emit_mov(ctx, element_type, tmp1_reg, IR_REG_RDX);
+						}
+					}
+					break;
+				case IR_OR:
+					|	ASM_REG_MEM_OP or, element_type, tmp1_reg, mem
+					break;
+				case IR_AND:
+					|	ASM_REG_MEM_OP and, element_type, tmp1_reg, mem
+					break;
+				case IR_XOR:
+					|	ASM_REG_MEM_OP xor, element_type, tmp1_reg, mem
+					break;
+				case IR_SHL:
+					if (!scalar_shift) {
+						ir_emit_load_mem_int(ctx, element_type, IR_REG_RCX, mem);
+					}
+					|	ASM_REG_TXT_OP shl, element_type, tmp1_reg, cl
+					break;
+				case IR_SHR:
+					if (!scalar_shift) {
+						ir_emit_load_mem_int(ctx, element_type, IR_REG_RCX, mem);
+					}
+					|	ASM_REG_TXT_OP shr, element_type, tmp1_reg, cl
+					break;
+				case IR_SAR:
+					if (!scalar_shift) {
+						ir_emit_load_mem_int(ctx, element_type, IR_REG_RCX, mem);
+					}
+					|	ASM_REG_TXT_OP sar, element_type, tmp1_reg, cl
+					break;
+				case IR_EQ:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	sete Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_NE:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setne Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_LT:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setl Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_LE:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setle Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_GE:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setge Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_GT:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setg Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_ULT:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setb Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_ULE:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setbe Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_UGE:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	setae Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
+				case IR_UGT:
+					|	ASM_REG_MEM_OP cmp, element_type, tmp1_reg, mem
+					|	seta Rb(tmp1_reg)
+					if (element_size != 1) {
+						|	ASM_REG_TXT_MUL movzx, element_type, tmp1_reg, Rb(tmp1_reg)
+					}
+					|	ASM_REG_OP neg, element_type, tmp1_reg
+					break;
 			}
-			|	jmp rax
-|.endif
-		}
-	} else if (ir_rule(ctx, insn->op2) & IR_FUSED) {
-	    ir_mem mem = ir_fuse_load(ctx, def, insn->op2);
-		|	ASM_TMEM_OP jmp, aword, mem
-	} else if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, IR_ADDR, op2_reg, insn->op2);
 		}
-		|	jmp Ra(op2_reg)
-	} else {
-		ir_mem mem = ir_ref_spill_slot(ctx, insn->op2);

-		|	ASM_TMEM_OP jmp, aword, mem
+#if IR_X86_I64
+		if (element_type == IR_I64 || element_type == IR_U64) {
+			ir_emit_store_mem(ctx, IR_U32, IR_MEM_ADD(dst_mem, element_size * i), tmp1_reg);
+			ir_emit_store_mem(ctx, IR_U32, IR_MEM_ADD(dst_mem, element_size * i + 4), tmp1_reg_hi);
+		} else
+#endif
+		ir_emit_store_mem(ctx, element_type, IR_MEM_ADD(dst_mem, element_size * i), tmp1_reg);
+	}
+
+	if (def_reg != IR_REG_NONE) {
+		ir_emit_load_mem_fp(ctx, insn->type, def_reg, dst_mem);
 	}
 }

-static bool ir_emit_guard_jcc(ir_ctx *ctx, uint32_t b, ir_ref def, uint32_t next_block, uint8_t op, void *addr, bool int_cmp, bool after_op)
+static void ir_emit_vector_ext(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *next_insn = &ctx->ir_base[def + 1];
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_type dst_element_type, src_element_type;
+	uint32_t src_width, dst_width, src_size, dst_size;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][2];

-	if (next_insn->op == IR_END || next_insn->op == IR_LOOP_END) {
-		ir_block *bb = &ctx->cfg_blocks[b];
-		uint32_t target;
+	(void)src_width;

-		if (!(bb->flags & IR_BB_DESSA_MOVES)) {
-			target = ctx->cfg_edges[bb->successors];
-			if (UNEXPECTED(bb->successors_count == 2)) {
-				if (ctx->cfg_blocks[target].flags & IR_BB_ENTRY) {
-					target = ctx->cfg_edges[bb->successors + 1];
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type) && IR_IS_TYPE_VECTOR(dst_type));
+	src_element_type = IR_VECTOR_BASE_TYPE(src_type);
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_element_type = IR_VECTOR_BASE_TYPE(dst_type);
+	dst_width = IR_VECTOR_SIZE(dst_type);
+
+	IR_ASSERT(IR_IS_TYPE_INT(src_element_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_element_type));
+	src_size = ir_type_size[src_element_type];
+	dst_size = ir_type_size[dst_element_type];
+	IR_ASSERT(src_size < dst_size);
+	IR_ASSERT(def_reg != IR_REG_NONE);
+
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (src_size == 1) {
+			if (dst_size == 2) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxbw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else {
+								|	vpmovsxbw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxbw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+								|	vpmovzxbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	vpmovsxbw xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+								|	vpmovsxbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxbw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxbw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width <= 16);
+					if (insn->op == IR_ZEXT) {
+						|	pmovzxbw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	pmovsxbw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
 				} else {
-					IR_ASSERT(ctx->cfg_blocks[ctx->cfg_edges[bb->successors + 1]].flags & IR_BB_ENTRY);
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (insn->op == IR_ZEXT) {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrlw xmm(def_reg-IR_REG_FP_FIRST), 8
+					} else {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psraw xmm(def_reg-IR_REG_FP_FIRST), 8
+					}
+				}
+			} else if (dst_size == 4) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxbd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else {
+								|	vpmovsxbd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxbd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	vpmovsxbd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width <= 16);
+					if (insn->op == IR_ZEXT) {
+						|	pmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	pmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (insn->op == IR_ZEXT) {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrld xmm(def_reg-IR_REG_FP_FIRST), 24
+					} else {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 24
+					}
 				}
 			} else {
-				IR_ASSERT(bb->successors_count == 1);
+				IR_ASSERT(dst_size == 8);
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxbq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else {
+								|	vpmovsxbq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxbq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+								|	vpmovzxbq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	vpmovsxbq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+								|	vpmovsxbq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxbq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxbq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width <= 16);
+					if (insn->op == IR_ZEXT) {
+						|	pmovzxbq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	pmovsxbq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (insn->op == IR_ZEXT) {
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpgtd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 24
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				}
 			}
-			target = ir_skip_empty_target_blocks(ctx, target);
-			if (target != next_block) {
-				if (int_cmp) {
-					switch (op) {
-						default:
-							IR_ASSERT(0 && "NIY binary op");
-						case IR_EQ:
-							|	jne =>target
-							break;
-						case IR_NE:
-							|	je =>target
-							break;
-						case IR_LT:
-							if (after_op) {
-								|	jns =>target
+		} else if (src_size == 2) {
+			if (dst_size == 4) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxwd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
 							} else {
-								|	jge =>target
+								|	vpmovsxwd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
 							}
-							break;
-						case IR_GE:
-							if (after_op) {
-								|	js =>target
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxwd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+								|	vpmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
 							} else {
-								|	jl =>target
+								|	vpmovsxwd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+								|	vpmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
 							}
-							break;
-						case IR_LE:
-							|	jg =>target
-							break;
-						case IR_GT:
-							|	jle =>target
-							break;
-						case IR_ULT:
-							|	jae =>target
-							break;
-						case IR_UGE:
-							|	jb =>target
-							break;
-						case IR_ULE:
-							|	ja =>target
-							break;
-						case IR_UGT:
-							|	jbe =>target
-							break;
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width <= 16);
+					if (insn->op == IR_ZEXT) {
+						|	pmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	pmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
 					}
 				} else {
-					switch (op) {
-						default:
-							IR_ASSERT(0 && "NIY binary op");
-						case IR_EQ:
-							|	jne =>target
-							|	jp =>target
-							break;
-						case IR_NE:
-							|	jp &addr
-							|	je =>target
-							break;
-						case IR_LT:
-							|	jae =>target
-							break;
-						case IR_GE:
-							|	jp &addr
-							|	jb =>target
-							break;
-						case IR_LE:
-							|	ja =>target
-							break;
-						case IR_GT:
-							|	jp &addr
-							|	jbe =>target
-							break;
-						case IR_ULT:
-							|	jp =>target
-							|	jae =>target
-							break;
-						case IR_UGE:
-							|	jb =>target
-							break;
-						case IR_ULE:
-							|	jp =>target
-							|	ja =>target
-							break;
-						case IR_UGT:
-							|	jbe =>target
-							break;
-						case IR_ORDERED:
-							|	jnp =>target
-							break;
-						case IR_UNORDERED:
-							|	jp =>target
-							break;
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (insn->op == IR_ZEXT) {
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrld xmm(def_reg-IR_REG_FP_FIRST), 16
+					} else {
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 16
+					}
+				}
+			} else {
+				IR_ASSERT(dst_size == 8);
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxwq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							} else {
+								|	vpmovsxwq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	vpmovzxwq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpmovzxwq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	vpmovsxwq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+								|	vpmovsxwq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxwq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxwq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width <= 16);
+					if (insn->op == IR_ZEXT) {
+						|	pmovzxwq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	pmovsxwq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					if (insn->op == IR_ZEXT) {
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpgtd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 16
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				}
+			}
+		} else {
+			IR_ASSERT(src_size == 4);
+			IR_ASSERT(dst_size == 8);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (dst_width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxdq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxdq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						if (insn->op == IR_ZEXT) {
+							|	vpmovzxdq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+							|	vpmovzxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vpmovsxdq xmm(tmp_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+							|	vpmovsxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (insn->op == IR_ZEXT) {
+						|	vpmovzxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	vpmovsxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				}
+			} else if (ctx->mflags & IR_X86_SSE41) {
+				IR_ASSERT(dst_width <= 16);
+				if (insn->op == IR_ZEXT) {
+					|	pmovzxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					|	pmovsxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(dst_width <= 16);
+				if (def_reg != op1_reg) {
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				if (insn->op == IR_ZEXT) {
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+				} else {
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	pcmpgtd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+				}
+			}
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;
+
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
+		} else {
+			mem = ir_ref_spill_slot(ctx, insn->op1);
+		}
+
+		if (src_size == 1) {
+			if (dst_size == 2) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						IR_ASSERT(src_width == 16);
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxbw, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxbw, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxbw, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+								|	ASM_TXT_TMEM_OP vpmovzxbw, xmm(tmp_reg-IR_REG_FP_FIRST), qword, IR_MEM_ADD(mem, 8)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxbw, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+								|	ASM_TXT_TMEM_OP vpmovsxbw, xmm(tmp_reg-IR_REG_FP_FIRST), qword, IR_MEM_ADD(mem, 8)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width == 16);
+						IR_ASSERT(src_width == 8);
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxbw, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+						} else {
+							|	ASM_TXT_TMEM_OP vpmovsxbw, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width == 16);
+					IR_ASSERT(src_width == 8);
+					if (insn->op == IR_ZEXT) {
+						|	ASM_TXT_TMEM_OP pmovzxbw, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP pmovsxbw, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					ir_emit_load_mem_fp(ctx, src_type, def_reg, mem);
+					if (insn->op == IR_ZEXT) {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrlw xmm(def_reg-IR_REG_FP_FIRST), 8
+					} else {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psraw xmm(def_reg-IR_REG_FP_FIRST), 8
 					}
 				}
-				|	jmp &addr
-				return 1;
-			}
-		}
-	} else if (next_insn->op == IR_IJMP && IR_IS_CONST_REF(next_insn->op2)) {
-		void *target_addr = ir_jmp_addr(ctx, next_insn, &ctx->ir_base[next_insn->op2]);
-
-		if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, target_addr)) {
-			if (int_cmp) {
-				switch (op) {
-					default:
-						IR_ASSERT(0 && "NIY binary op");
-					case IR_EQ:
-						|	jne &target_addr
-						break;
-					case IR_NE:
-						|	je &target_addr
-						break;
-					case IR_LT:
-						if (after_op) {
-							|	jns &target_addr
+			} else if (dst_size == 4) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						IR_ASSERT(src_width == 8);
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxbd, ymm(def_reg-IR_REG_FP_FIRST), qword, mem
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxbd, ymm(def_reg-IR_REG_FP_FIRST), qword, mem
+							}
 						} else {
-							|	jge &target_addr
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxbd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+								|	ASM_TXT_TMEM_OP vpmovzxbd, xmm(tmp_reg-IR_REG_FP_FIRST), dword, IR_MEM_ADD(mem, 4)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxbd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+								|	ASM_TXT_TMEM_OP vpmovsxbd, xmm(tmp_reg-IR_REG_FP_FIRST), dword, IR_MEM_ADD(mem, 4)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							}
 						}
-						break;
-					case IR_GE:
-						if (after_op) {
-							|	js &target_addr
+					} else {
+						IR_ASSERT(dst_width == 16);
+						IR_ASSERT(src_width == 4);
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxbd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
 						} else {
-							|	jl &target_addr
+							|	ASM_TXT_TMEM_OP vpmovsxbd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
 						}
-						break;
-					case IR_LE:
-						|	jg &target_addr
-						break;
-					case IR_GT:
-						|	jle &target_addr
-						break;
-					case IR_ULT:
-						|	jae &target_addr
-						break;
-					case IR_UGE:
-						|	jb &target_addr
-						break;
-					case IR_ULE:
-						|	ja &target_addr
-						break;
-					case IR_UGT:
-						|	jbe &target_addr
-						break;
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width == 16);
+					IR_ASSERT(src_width == 4);
+					if (insn->op == IR_ZEXT) {
+						|	ASM_TXT_TMEM_OP pmovzxbd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP pmovsxbd, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					ir_emit_load_mem_fp(ctx, src_type, def_reg, mem);
+					if (insn->op == IR_ZEXT) {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrld xmm(def_reg-IR_REG_FP_FIRST), 24
+					} else {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 24
+					}
 				}
 			} else {
-				switch (op) {
-					default:
-						IR_ASSERT(0 && "NIY binary op");
-					case IR_EQ:
-						|	jne &target_addr
-						|	jp &target_addr
-						break;
-					case IR_NE:
-						|	jp &addr
-						|	je &target_addr
-						break;
-					case IR_LT:
-						|	jae &target_addr
-						break;
-					case IR_GE:
-						|	jp &addr
-						|	jb &target_addr
-						break;
-					case IR_LE:
-						|	ja &target_addr
-						break;
-					case IR_GT:
-						|	jp &addr
-						|	jbe &target_addr
-						break;
-					case IR_ULT:
-						|	jp &target_addr
-						|	jae &target_addr
-						break;
-					case IR_UGE:
-						|	jb &target_addr
-						break;
-					case IR_ULE:
-						|	jp &target_addr
-						|	ja &target_addr
-						break;
-					case IR_UGT:
-						|	jbe &target_addr
-						break;
-					case IR_ORDERED:
-						|	jnp &target_addr
-						break;
-					case IR_UNORDERED:
-						|	jp &target_addr
-						break;
-				}
-			}
-			|	jmp &addr
-			return 1;
-		}
-	}
-
-	if (int_cmp) {
-		switch (op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_EQ:
-				|	je &addr
-				break;
-			case IR_NE:
-				|	jne &addr
-				break;
-			case IR_LT:
-				if (after_op) {
-					|	js &addr
+				IR_ASSERT(dst_size == 8);
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						IR_ASSERT(src_width == 4);
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxbq, ymm(def_reg-IR_REG_FP_FIRST), dword, mem
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxbq, ymm(def_reg-IR_REG_FP_FIRST), dword, mem
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxbq, xmm(def_reg-IR_REG_FP_FIRST), word, mem
+								|	ASM_TXT_TMEM_OP vpmovzxbq, xmm(tmp_reg-IR_REG_FP_FIRST), word, IR_MEM_ADD(mem, 2)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxbq, xmm(def_reg-IR_REG_FP_FIRST), word, mem
+								|	ASM_TXT_TMEM_OP vpmovsxbq, xmm(tmp_reg-IR_REG_FP_FIRST), word, IR_MEM_ADD(mem, 2)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width == 16);
+						IR_ASSERT(src_width == 2);
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxbq, xmm(def_reg-IR_REG_FP_FIRST), word, mem
+						} else {
+							|	ASM_TXT_TMEM_OP vpmovsxbq, xmm(def_reg-IR_REG_FP_FIRST), word, mem
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width == 16);
+					IR_ASSERT(src_width == 2);
+					if (insn->op == IR_ZEXT) {
+						|	ASM_TXT_TMEM_OP pmovzxbq, xmm(def_reg-IR_REG_FP_FIRST), word, mem
+					} else {
+						|	ASM_TXT_TMEM_OP pmovsxbq, xmm(def_reg-IR_REG_FP_FIRST), word, mem
+					}
 				} else {
-					|	jl &addr
+					IR_ASSERT(dst_width <= 16);
+					ir_emit_load_mem_fp(ctx, src_type, def_reg, mem);
+					if (insn->op == IR_ZEXT) {
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	punpcklbw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpgtd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 24
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
 				}
-				break;
-			case IR_GE:
-				if (after_op) {
-					|	jns &addr
+			}
+		} else if (src_size == 2) {
+			if (dst_size == 4) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						IR_ASSERT(src_width == 16);
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxwd, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxwd, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxwd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+								|	ASM_TXT_TMEM_OP vpmovzxwd, xmm(tmp_reg-IR_REG_FP_FIRST), qword, IR_MEM_ADD(mem, 8)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxwd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+								|	ASM_TXT_TMEM_OP vpmovsxwd, xmm(tmp_reg-IR_REG_FP_FIRST), qword, IR_MEM_ADD(mem, 8)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width == 16);
+						IR_ASSERT(src_width == 8);
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxwd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+						} else {
+							|	ASM_TXT_TMEM_OP vpmovsxwd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width == 16);
+					IR_ASSERT(src_width == 8);
+					if (insn->op == IR_ZEXT) {
+						|	ASM_TXT_TMEM_OP pmovzxwd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP pmovsxwd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					}
 				} else {
-					|	jge &addr
+					IR_ASSERT(dst_width <= 16);
+					ir_emit_load_mem_fp(ctx, src_type, def_reg, mem);
+					if (insn->op == IR_ZEXT) {
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrld xmm(def_reg-IR_REG_FP_FIRST), 16
+					} else {
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 16
+					}
 				}
-				break;
-			case IR_LE:
-				|	jle &addr
-				break;
-			case IR_GT:
-				|	jg &addr
-				break;
-			case IR_ULT:
-				|	jb &addr
-				break;
-			case IR_UGE:
-				|	jae &addr
-				break;
-			case IR_ULE:
-				|	jbe &addr
-				break;
-			case IR_UGT:
-				|	ja &addr
-				break;
-		}
-	} else {
-		switch (op) {
-			default:
-				IR_ASSERT(0 && "NIY binary op");
-			case IR_EQ:
-				|	jp >1
-				|	je &addr
-				|1:
-				break;
-			case IR_NE:
-				|	jne &addr
-				|	jp &addr
-				break;
-			case IR_LT:
-				|	jp >1
-				|	jb &addr
-				|1:
-				break;
-			case IR_GE:
-				|	jae &addr
-				break;
-			case IR_LE:
-				|	jp >1
-				|	jbe &addr
-				|1:
-				break;
-			case IR_GT:
-				|	ja &addr
-				break;
-			case IR_ULT:
-				|	jb &addr
-				break;
-			case IR_UGE:
-				|	jp &addr
-				|	jae &addr
-				break;
-			case IR_ULE:
-				|	jbe &addr
-				break;
-			case IR_UGT:
-				|	jp &addr
-				|	ja &addr
-				break;
-			case IR_ORDERED:
-				|	jp &addr
-				break;
-			case IR_UNORDERED:
-				|	jnp &addr
-				break;
-		}
-	}
-	return 0;
-}
-
-static bool ir_emit_guard(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_reg op2_reg = ctx->regs[def][2];
-	ir_type type = ctx->ir_base[insn->op2].type;
-	void *addr;
-
-	IR_ASSERT(IR_IS_TYPE_INT(type));
-	if (IR_IS_CONST_REF(insn->op2)) {
-		bool is_true = ir_ref_is_true(ctx, insn->op2);
-
-		if ((insn->op == IR_GUARD && !is_true) || (insn->op == IR_GUARD_NOT && is_true)) {
-			addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
-			if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
-				|	jmp aword &addr
 			} else {
-|.if X64
-				if (IR_IS_SIGNED_32BIT(addr)) {
-					|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+				IR_ASSERT(dst_size == 8);
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						IR_ASSERT(src_width == 8);
+						if (ctx->mflags & IR_X86_AVX2) {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxwq, ymm(def_reg-IR_REG_FP_FIRST), qword, mem
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxwq, ymm(def_reg-IR_REG_FP_FIRST), qword, mem
+							}
+						} else {
+							if (insn->op == IR_ZEXT) {
+								|	ASM_TXT_TMEM_OP vpmovzxwq, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+								|	ASM_TXT_TMEM_OP vpmovzxwq, xmm(tmp_reg-IR_REG_FP_FIRST), dword, IR_MEM_ADD(mem, 4)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							} else {
+								|	ASM_TXT_TMEM_OP vpmovsxwq, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+								|	ASM_TXT_TMEM_OP vpmovsxwq, xmm(tmp_reg-IR_REG_FP_FIRST), dword, IR_MEM_ADD(mem, 4)
+								|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+							}
+						}
+					} else {
+						IR_ASSERT(dst_width == 16);
+						IR_ASSERT(src_width == 4);
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxwq, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+						} else {
+							|	ASM_TXT_TMEM_OP vpmovsxwq, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+						}
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width == 16);
+					IR_ASSERT(src_width == 4);
+					if (insn->op == IR_ZEXT) {
+						|	ASM_TXT_TMEM_OP pmovzxwq, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP pmovsxwq, xmm(def_reg-IR_REG_FP_FIRST), dword, mem
+					}
 				} else {
-					|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+					IR_ASSERT(dst_width <= 16);
+					ir_emit_load_mem_fp(ctx, src_type, def_reg, mem);
+					if (insn->op == IR_ZEXT) {
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					} else {
+						|	punpcklwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	pcmpgtd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	psrad xmm(def_reg-IR_REG_FP_FIRST), 16
+						|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
 				}
-				|	jmp aword [rax]
-|.endif
 			}
-		}
-		return 0;
-	}
-
-	if (op2_reg != IR_REG_NONE) {
-		if (IR_REG_SPILLED(op2_reg)) {
-			op2_reg = IR_REG_NUM(op2_reg);
-			ir_emit_load(ctx, type, op2_reg, insn->op2);
-		}
-		|	ASM_REG_REG_OP test, type, op2_reg, op2_reg
-	} else {
-		ir_mem mem;
-
-		if (ir_rule(ctx, insn->op2) & IR_FUSED) {
-			mem = ir_fuse_load(ctx, def, insn->op2);
 		} else {
-			mem = ir_ref_spill_slot(ctx, insn->op2);
+			IR_ASSERT(src_size == 4);
+			IR_ASSERT(dst_size == 8);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (dst_width == 32) {
+					IR_ASSERT(src_width == 16);
+					if (ctx->mflags & IR_X86_AVX2) {
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxdq, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+						} else {
+							|	ASM_TXT_TMEM_OP vpmovsxdq, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+						}
+					} else {
+						if (insn->op == IR_ZEXT) {
+							|	ASM_TXT_TMEM_OP vpmovzxdq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+							|	ASM_TXT_TMEM_OP vpmovzxdq, xmm(tmp_reg-IR_REG_FP_FIRST), qword, IR_MEM_ADD(mem, 8)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	ASM_TXT_TMEM_OP vpmovsxdq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+							|	ASM_TXT_TMEM_OP vpmovsxdq, xmm(tmp_reg-IR_REG_FP_FIRST), qword, IR_MEM_ADD(mem, 8)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 1
+						}
+					}
+				} else {
+					IR_ASSERT(dst_width == 16);
+					IR_ASSERT(src_width == 8);
+					if (insn->op == IR_ZEXT) {
+						|	ASM_TXT_TMEM_OP vpmovzxdq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					} else {
+						|	ASM_TXT_TMEM_OP vpmovsxdq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+					}
+				}
+			} else if (ctx->mflags & IR_X86_SSE41) {
+				IR_ASSERT(dst_width == 16);
+				IR_ASSERT(src_width == 8);
+				if (insn->op == IR_ZEXT) {
+					|	ASM_TXT_TMEM_OP pmovzxdq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				} else {
+					|	ASM_TXT_TMEM_OP pmovsxdq, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				}
+			} else {
+				IR_ASSERT(dst_width <= 16);
+				ir_emit_load_mem_fp(ctx, src_type, def_reg, mem);
+				if (insn->op == IR_ZEXT) {
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+				} else {
+					|	pxor xmm(tmp_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					|	pcmpgtd xmm(tmp_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	punpckldq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+				}
+			}
 		}
-		|	ASM_MEM_IMM_OP cmp, type, mem, 0
 	}

-	addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
-	if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
-		ir_op op;
-
-		if (insn->op == IR_GUARD) {
-			op = IR_EQ;
-		} else {
-			op = IR_NE;
-		}
-		return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 0);
-	} else {
-|.if X64
-		if (insn->op == IR_GUARD) {
-			|	je >1
-		} else {
-			|	jne >1
-		}
-		|.cold_code
-		|1:
-		if (IR_IS_SIGNED_32BIT(addr)) {
-			|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
-		} else {
-			|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
-		}
-		|	jmp aword [rax]
-		|.code
-|.endif
-		return 0;
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
 	}
 }

-static bool ir_emit_guard_cmp_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
+static void ir_emit_vector_trunc(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_insn *cmp_insn = &ctx->ir_base[insn->op2];
-	ir_op op = cmp_insn->op;
-	ir_type type = ctx->ir_base[cmp_insn->op1].type;
-	ir_ref op1 = cmp_insn->op1;
-	ir_ref op2 = cmp_insn->op2;
-	void *addr;
-	ir_reg op1_reg, op2_reg;
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
+	ir_type dst_element_type, src_element_type;
+	uint32_t src_width, src_size, dst_size;
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp_reg = ctx->regs[def][2];

-	if (UNEXPECTED(ctx->rules[insn->op2] & IR_FUSED_REG)) {
-		op1_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 1);
-		op2_reg = ir_get_fused_reg(ctx, def, insn->op2 * sizeof(ir_ref) + 2);
-	} else {
-		op1_reg = ctx->regs[insn->op2][1];
-		op2_reg = ctx->regs[insn->op2][2];
-	}
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type) && IR_IS_TYPE_VECTOR(dst_type));
+	src_element_type = IR_VECTOR_BASE_TYPE(src_type);
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_element_type = IR_VECTOR_BASE_TYPE(dst_type);

-	if (op1_reg != IR_REG_NONE && IR_REG_SPILLED(op1_reg)) {
+	IR_ASSERT(IR_IS_TYPE_INT(src_element_type));
+	IR_ASSERT(IR_IS_TYPE_INT(dst_element_type));
+	src_size = ir_type_size[src_element_type];
+	dst_size = ir_type_size[dst_element_type];
+	IR_ASSERT(src_size > dst_size);
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);
+
+	if (IR_REG_SPILLED(op1_reg)) {
 		op1_reg = IR_REG_NUM(op1_reg);
-		ir_emit_load(ctx, type, op1_reg, op1);
+		ir_emit_load(ctx, src_type, op1_reg, insn->op1);
 	}
-	if (op2_reg != IR_REG_NONE && IR_REG_SPILLED(op2_reg)) {
-		op2_reg = IR_REG_NUM(op2_reg);
-		if (op1 != op2) {
-			ir_emit_load(ctx, type, op2_reg, op2);
+	if (src_size == 8) {
+		if (dst_size == 4) {
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpshufd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 232
+						|	vpermq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 232
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 136
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 232
+				}
+			} else {
+				|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 232
+			}
+		} else if (dst_size == 2) {
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpshufd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 232
+						|	vpermq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 232
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 136
+					}
+					|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+					|	vpshufhw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+					|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 232
+					|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+				}
+			} else {
+				|	pshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 232
+				|	pshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+			}
+		} else {
+			IR_ASSERT(dst_size == 1);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpshufd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 232
+						|	vpermq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 232
+					} else {
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vshufps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST), 136
+					}
+					|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 24
+					|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 24
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	vpsllq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 56
+					|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 56
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				if (def_reg != op1_reg) {
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				|	psllq xmm(def_reg-IR_REG_FP_FIRST), 56
+				|	psrlq xmm(def_reg-IR_REG_FP_FIRST), 56
+				|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+			}
 		}
-	}
-
-	addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
-	if (IR_IS_CONST_REF(op2) && !IR_IS_SYM_CONST(ctx->ir_base[op2].op) && ctx->ir_base[op2].val.u64 == 0) {
-		if (op == IR_ULT) {
-			/* always false */
-			if (sizeof(void*) == 4 || IR_MAY_USE_32BIT_ADDR(ctx->code_buffer, addr)) {
-				|	jmp aword &addr
+	} else if (src_size == 4) {
+		if (dst_size == 2) {
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpslld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 16
+						|	vpsrld ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 16
+						|	vpackusdw ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+						|	vpermq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 232
+					} else {
+						if (!data->v_u32_to_u16) {
+							data->v_u32_to_u16 = 1;
+							ir_rodata(ctx);
+							|.align 32
+							|->v_u32_to_u16:
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|	.dword 0xffff
+							|.code
+						}
+						|	vandps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_u32_to_u16]
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+					|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 16
+					|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
 			} else {
-|.if X64
-				if (IR_IS_SIGNED_32BIT(addr)) {
-					|	mov rax, ((ptrdiff_t)addr)    // 0x48 0xc7 0xc0 <imm-32-bit>
+				if (op1_reg != def_reg) {
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				|	pslld xmm(def_reg-IR_REG_FP_FIRST), 16
+				|	psrld xmm(def_reg-IR_REG_FP_FIRST), 16
+				|	packusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+			}
+		} else {
+			IR_ASSERT(dst_size == 1);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					if (ctx->mflags & IR_X86_AVX2) {
+						|	vpslld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 24
+						|	vpsrld ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 24
+						|	vpackuswb ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+						|	vpermq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 232
+						|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						if (!data->v_u32_to_u8) {
+							data->v_u32_to_u8 = 1;
+							ir_rodata(ctx);
+							|.align 32
+							|->v_u32_to_u8:
+							|	.dword 0xff
+							|	.dword 0xff
+							|	.dword 0xff
+							|	.dword 0xff
+							|	.dword 0xff
+							|	.dword 0xff
+							|	.dword 0xff
+							|	.dword 0xff
+							|.code
+						}
+						|	vandps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_u32_to_u8]
+						|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+						|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
 				} else {
-					|	mov64 rax, ((ptrdiff_t)addr)  // 0x48 0xb8 <imm-64-bit>
+					IR_ASSERT(src_width <= 16);
+					|	vpslld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 24
+					|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 24
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
 				}
-				|	jmp aword [rax]
-|.endif
+			} else {
+				if (def_reg != op1_reg) {
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+				|	pslld xmm(def_reg-IR_REG_FP_FIRST), 24
+				|	psrld xmm(def_reg-IR_REG_FP_FIRST), 24
+				|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
 			}
-			return 0;
-		} else if (op == IR_UGE) {
-			/* always true */
-			return 0;
-		} else if (op == IR_ULE) {
-			op = IR_EQ;
-		} else if (op == IR_UGT) {
-			op = IR_NE;
 		}
-	}
-	ir_emit_cmp_int_common(ctx, type, def, cmp_insn, op1_reg, op1, op2_reg, op2);
-
-	if (insn->op == IR_GUARD) {
-		op ^= 1; // reverse
-	}
-
-	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 0);
-}
-
-static bool ir_emit_guard_cmp_fp(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
-{
-	ir_op op = ir_emit_cmp_fp_common(ctx, def, insn->op2, &ctx->ir_base[insn->op2]);
-	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
-
-	if (insn->op == IR_GUARD) {
-		if (op == IR_EQ || op == IR_NE || op == IR_ORDERED || op == IR_UNORDERED) {
-			op ^= 1; // reverse
+	} else if (src_size == 2) {
+		IR_ASSERT(dst_size == 1);
+		if (ctx->mflags & IR_X86_AVX) {
+			if (src_width == 32) {
+				if (ctx->mflags & IR_X86_AVX2) {
+					|	vpsllw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 8
+					|	vpsrlw ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 8
+					|	vpackuswb ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					|	vpermq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 232
+				} else {
+					if (!data->v_u16_to_u8) {
+						data->v_u16_to_u8 = 1;
+						ir_rodata(ctx);
+						|.align 32
+						|->v_u16_to_u8:
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|	.word 0xff
+						|.code
+					}
+					|	vandps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_u16_to_u8]
+					|	vextractf128 xmm(tmp_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+					|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(src_width <= 16);
+				|	vpsllw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 8
+				|	vpsrlw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 8
+				|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+			}
 		} else {
-			op ^= 5; // reverse
+			if (def_reg != op1_reg) {
+				|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+			}
+			|	psllw xmm(def_reg-IR_REG_FP_FIRST), 8
+			|	psrlw xmm(def_reg-IR_REG_FP_FIRST), 8
+			|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
 		}
 	}
-	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 0, 0);
-}
-
-static bool ir_emit_guard_test_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
-{
-	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
-	ir_op op = (insn->op == IR_GUARD) ? IR_EQ : IR_NE;

-	ir_emit_test_int_common(ctx, def, insn->op2, op);
-	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 0);
-}
-
-static bool ir_emit_guard_jcc_int(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn, uint32_t next_block)
-{
-	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
-	ir_op op = ctx->ir_base[insn->op2].op;
-
-	if (insn->op == IR_GUARD) {
-		op ^= 1; // reverse
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
 	}
-	return ir_emit_guard_jcc(ctx, b, def, next_block, op, addr, 1, 1);
 }

-static bool ir_emit_guard_overflow(ir_ctx *ctx, uint32_t b, ir_ref def, ir_insn *insn)
+static void ir_emit_vector_fp2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_type type;
-	void *addr = ir_jmp_addr(ctx, insn, &ctx->ir_base[insn->op3]);
+	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	uint32_t src_width;

-	type = ctx->ir_base[ctx->ir_base[insn->op2].op1].type;
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+	src_width = IR_VECTOR_SIZE(src_type);
+	src_type = IR_VECTOR_BASE_TYPE(src_type);
+	dst_type = IR_VECTOR_BASE_TYPE(dst_type);
+
+	IR_ASSERT(def_reg != IR_REG_NONE);
+	if (op1_reg != IR_REG_NONE) {
+		if (IR_REG_SPILLED(op1_reg)) {
+			op1_reg = IR_REG_NUM(op1_reg);
+			ir_emit_load(ctx, src_type, op1_reg, insn->op1);
+		}
+		if (src_type == dst_type) {
+			if (op1_reg != def_reg) {
+				ir_emit_fp_mov(ctx, dst_type, def_reg, op1_reg);
+			}
+		} else if (src_type == IR_DOUBLE) {
+			IR_ASSERT(dst_type == IR_FLOAT);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					|	vcvtpd2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	vcvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(src_width <= 16);
+				|	cvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+			}
+		} else {
+			IR_ASSERT(src_type == IR_FLOAT);
+			IR_ASSERT(dst_type == IR_DOUBLE);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 16) {
+					|	vcvtps2pd ymm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(src_width <= 8);
+					|	vcvtps2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+					IR_ASSERT(src_width <= 8);
+				|	cvtps2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+			}
+		}
+	} else if (IR_IS_CONST_REF(insn->op1)) {
+		IR_ASSERT(0);
+	} else {
+		ir_mem mem;

-	IR_ASSERT(IR_IS_TYPE_INT(type));
-	if (IR_IS_TYPE_SIGNED(type)) {
-		if (insn->op == IR_GUARD) {
-			|	jno &addr
+		if (ir_rule(ctx, insn->op1) & IR_FUSED) {
+			mem = ir_fuse_load(ctx, def, insn->op1);
 		} else {
-			|	jo &addr
+			mem = ir_ref_spill_slot(ctx, insn->op1);
 		}
-	} else {
-		if (insn->op == IR_GUARD) {
-			|	jnc &addr
+
+		if (src_type == IR_DOUBLE) {
+			IR_ASSERT(dst_type == IR_DOUBLE);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 32) {
+					|	ASM_TXT_TMEM_OP vcvtpd2ps, ymm(def_reg-IR_REG_FP_FIRST), yword, mem
+				} else {
+					IR_ASSERT(src_width == 16);
+					|	ASM_TXT_TMEM_OP vcvtpd2ps, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+				}
+			} else {
+				IR_ASSERT(src_width == 16);
+				|	ASM_TXT_TMEM_OP cvtpd2ps, xmm(def_reg-IR_REG_FP_FIRST), oword, mem
+			}
 		} else {
-			|	jc &addr
+			IR_ASSERT(src_type == IR_FLOAT);
+			IR_ASSERT(dst_type == IR_DOUBLE);
+			if (ctx->mflags & IR_X86_AVX) {
+				if (src_width == 16) {
+					|	ASM_TXT_TMEM_OP vcvtps2pd, ymm(def_reg-IR_REG_FP_FIRST), oword, mem
+				} else {
+					IR_ASSERT(src_width == 8);
+					|	ASM_TXT_TMEM_OP vcvtps2pd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+				}
+			} else {
+				IR_ASSERT(src_width == 8);
+				|	ASM_TXT_TMEM_OP cvtps2pd, xmm(def_reg-IR_REG_FP_FIRST), qword, mem
+			}
 		}
 	}
-	return 0;
+
+	if (IR_REG_SPILLED(ctx->regs[def][0])) {
+		ir_emit_store(ctx, dst_type, def, def_reg);
+	}
 }

-static void ir_emit_lea(ir_ctx *ctx, ir_ref def, ir_type type)
+static void ir_emit_vector_fp2int(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-	ir_mem mem = ir_fuse_addr(ctx, def, def);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp1_reg = ctx->regs[def][2];
+	ir_reg tmp2_reg = ctx->regs[def][3];
+	uint32_t src_width, dst_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_width = IR_VECTOR_SIZE(dst_type);
+	src_type = IR_VECTOR_BASE_TYPE(src_type);
+	dst_type = IR_VECTOR_BASE_TYPE(dst_type);
+
+	IR_ASSERT(IR_IS_TYPE_FP(src_type) && IR_IS_TYPE_INT(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);

-	IR_ASSERT(def_reg != IR_REG_NONE);
-	if (ir_type_size[type] == 4) {
-		if (IR_MEM_BASE(mem) == def_reg
-		 && IR_MEM_OFFSET(mem) == 0
-		 && IR_MEM_SCALE(mem) == 1
-		 && IR_MEM_INDEX(mem) != IR_REG_NONE) {
-			ir_reg reg = IR_MEM_INDEX(mem);
-			|	add Rd(def_reg), Rd(reg)
-		} else if (IR_MEM_INDEX(mem) == def_reg
-		 && IR_MEM_OFFSET(mem) == 0
-		 && IR_MEM_SCALE(mem) == 1
-		 && IR_MEM_BASE(mem) != IR_REG_NONE) {
-			ir_reg reg = IR_MEM_BASE(mem);
-			|	add Rd(def_reg), Rd(reg)
-		} else if (IR_MEM_INDEX(mem) == def_reg
-		 && IR_MEM_OFFSET(mem) == 0
-		 && IR_MEM_SCALE(mem) == 2
-		 && IR_MEM_BASE(mem) == IR_REG_NONE) {
-			|	add Rd(def_reg), Rd(def_reg)
+	if (src_type == IR_FLOAT) {
+		if (IR_IS_TYPE_SIGNED(dst_type)) {
+			if (dst_type == IR_I8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtps2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vpackssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vpacksswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpacksswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	packssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	packsswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_I16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtps2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vpackssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	packssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_I32) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtps2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_I64) {
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						ir_reg tmp3_reg = ctx->tmp_regs[def];
+
+						|	vshufps xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 255
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vshufpd xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vcvttss2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvttss2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vinserti128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vmovshdup xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vcvttss2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (ctx->mflags & IR_X86_SSE3) {
+						|	movshdup xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	shufps xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 85
+					}
+					|	cvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	movq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	cvttss2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif defined(IR_TARGET_X86)
+|.if not X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	sub esp, 32
+						|	vmovaps oword [esp+16], xmm(op1_reg-IR_REG_FP_FIRST)
+						|	fld dword [esp+16]
+						|	fisttp qword [esp]
+						|	fld dword [esp+20]
+						|	fisttp qword [esp+8]
+						|	fld dword [esp+24]
+						|	fisttp qword [esp+16]
+						|	fld dword [esp+28]
+						|	fisttp qword [esp+24]
+						|	vmovdqu ymm(def_reg-IR_REG_FP_FIRST), yword [esp]
+						|	add esp, 32
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	sub esp, 16
+						|	vmovlps qword [esp+8], xmm(op1_reg-IR_REG_FP_FIRST)
+						|	fld dword [esp+8]
+						|	fisttp qword [esp]
+						|	fld dword [esp+12]
+						|	fisttp qword [esp+8]
+						|	vmovdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+						|	add esp, 16
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	sub esp, 16
+					|	movlps qword [esp+8], xmm(op1_reg-IR_REG_FP_FIRST)
+					|	fld dword [esp+8]
+					|	fisttp qword [esp]
+					|	fld dword [esp+12]
+					|	fisttp qword [esp+8]
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+					|	add esp, 16
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
+			}
 		} else {
-			if (IR_MEM_SCALE(mem) == 2 && IR_MEM_BASE(mem) == IR_REG_NONE) {
-				mem = IR_MEM(IR_MEM_INDEX(mem), IR_MEM_OFFSET(mem), IR_MEM_INDEX(mem), 1);
+			if (dst_type == IR_U8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtps2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	packusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_U16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtps2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	packusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_U32) {
+				if (!data->v_f_to_u32) {
+					data->v_f_to_u32 = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+						|->v_f_to_u32:
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+					} else {
+						|.align 16
+						|->v_f_to_u32:
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+						|	.dword 0x4f000000
+					}
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvttps2dq ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						|	vsubps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_f_to_u32]
+						|	vcvttps2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpsrad ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 31
+							|	vpand ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+							|	vpor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vorps ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+							|	vblendvps ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST)
+						}
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvttps2dq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpsrad xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 31
+						|	vsubps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [->v_f_to_u32]
+						|	vcvttps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpand xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vpor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvttps2dq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	movapd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	psrad xmm(tmp2_reg-IR_REG_FP_FIRST), 31
+					if (def_reg != op1_reg) {
+						|	movapd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	subps xmm(def_reg-IR_REG_FP_FIRST), [->v_f_to_u32]
+					|	cvttps2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	pand xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+					|	por xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_U64) {
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (!data->ull2f_const) {
+					data->ull2f_const = 1;
+					ir_rodata(ctx);
+					|.align 4
+					|->ull2f_const:
+					|.dword 0x5f000000
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						ir_reg tmp3_reg = ctx->tmp_regs[def];
+
+						|	vshufps xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 255
+						|	vucomiss xmm(tmp1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	jnb >1
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubss xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vshufpd xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vucomiss xmm(tmp3_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	jnb >1
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubss xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vucomiss xmm(op1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	jnb >1
+						|	vcvttss2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubss xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vucomiss xmm(def_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	jnb >1
+						|	vcvttss2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	vcvttss2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vinserti128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vmovshdup xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vucomiss xmm(tmp1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	jnb >1
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubss xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	vcvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vucomiss xmm(op1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	jnb >1
+						|	vcvttss2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubss xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+						|	vcvttss2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (ctx->mflags & IR_X86_SSE3) {
+						|	movshdup xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	shufps xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 85
+					}
+					|	ucomiss xmm(tmp1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+					|	jnb >1
+					|	cvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	jmp >2
+					|1:
+					|	subss xmm(tmp1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+					|	cvttss2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	btc Rq(tmp2_reg), 63
+					|2:
+					|	movq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	ucomiss xmm(op1_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+					|	jnb >1
+					|	cvttss2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	jmp >2
+					|1:
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	subss xmm(def_reg-IR_REG_FP_FIRST), dword [->ull2f_const]
+					|	cvttss2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+					|	btc Rq(tmp2_reg), 63
+					|2:
+					|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif defined(IR_TARGET_X86)
+//				IR_ASSERT(0 && "v_f -> v_u64 (full unsigned range) ???");
+|.if not X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	sub esp, 32
+						|	vmovaps oword [esp+16], xmm(op1_reg-IR_REG_FP_FIRST)
+						|	fld dword [esp+16]
+						|	fisttp qword [esp]
+						|	fld dword [esp+20]
+						|	fisttp qword [esp+8]
+						|	fld dword [esp+24]
+						|	fisttp qword [esp+16]
+						|	fld dword [esp+28]
+						|	fisttp qword [esp+24]
+						|	vmovdqu ymm(def_reg-IR_REG_FP_FIRST), yword [esp]
+						|	add esp, 32
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	sub esp, 16
+						|	vmovlps qword [esp+8], xmm(op1_reg-IR_REG_FP_FIRST)
+						|	fld dword [esp+8]
+						|	fisttp qword [esp]
+						|	fld dword [esp+12]
+						|	fisttp qword [esp+8]
+						|	vmovdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+						|	add esp, 16
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	sub esp, 16
+					|	movlps qword [esp+8], xmm(op1_reg-IR_REG_FP_FIRST)
+					|	fld dword [esp+8]
+					|	fisttp qword [esp]
+					|	fld dword [esp+12]
+					|	fisttp qword [esp+8]
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+					|	add esp, 16
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
 			}
-			|	ASM_TXT_TMEM_OP lea, Rd(def_reg), dword, mem
 		}
 	} else {
-		if (IR_MEM_BASE(mem) == def_reg
-		 && IR_MEM_OFFSET(mem) == 0
-		 && IR_MEM_SCALE(mem) == 1
-		 && IR_MEM_INDEX(mem) != IR_REG_NONE) {
-			ir_reg reg = IR_MEM_INDEX(mem);
-			|	add Ra(def_reg), Ra(reg)
-		} else if (IR_MEM_INDEX(mem) == def_reg
-		 && IR_MEM_OFFSET(mem) == 0
-		 && IR_MEM_SCALE(mem) == 1
-		 && IR_MEM_BASE(mem) != IR_REG_NONE) {
-			ir_reg reg = IR_MEM_BASE(mem);
-			|	add Ra(def_reg), Ra(reg)
-		} else if (IR_MEM_INDEX(mem) == def_reg
-		 && IR_MEM_OFFSET(mem) == 0
-		 && IR_MEM_SCALE(mem) == 2
-		 && IR_MEM_BASE(mem) == IR_REG_NONE) {
-			|	add Ra(def_reg), Ra(def_reg)
+		IR_ASSERT(src_type == IR_DOUBLE);
+		if (IR_IS_TYPE_SIGNED(dst_type)) {
+			if (dst_type == IR_I8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtpd2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpacksswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						|	vcvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+						|	vpacksswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						IR_ASSERT(src_width <= 16);
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	pshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+					|	packsswb xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_I16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtpd2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackssdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	pshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+				}
+			} else if (dst_type == IR_I32) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtpd2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_I64) {
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						ir_reg tmp3_reg = ctx->tmp_regs[def];
+
+						|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vshufpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vcvttsd2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vshufpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vinserti128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vunpckhpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vcvttsd2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	movapd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	unpckhpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	cvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	movq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	cvttsd2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif defined(IR_TARGET_X86)
+|.if not X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	sub esp, 32
+						|	vmovupd yword [esp], ymm(op1_reg-IR_REG_FP_FIRST)
+						|	fld qword [esp]
+						|	fisttp qword [esp]
+						|	fld qword [esp+8]
+						|	fisttp qword [esp+8]
+						|	fld qword [esp+16]
+						|	fisttp qword [esp+16]
+						|	fld qword [esp+24]
+						|	fisttp qword [esp+24]
+						|	vmovdqu ymm(def_reg-IR_REG_FP_FIRST), yword [esp]
+						|	add esp, 32
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	sub esp, 16
+						|	vmovapd oword [esp], xmm(op1_reg-IR_REG_FP_FIRST)
+						|	fld qword [esp]
+						|	fisttp qword [esp]
+						|	fld qword [esp+8]
+						|	fisttp qword [esp+8]
+						|	vmovdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+						|	add esp, 16
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	sub esp, 16
+					|	movapd oword [esp], xmm(op1_reg-IR_REG_FP_FIRST)
+					|	fld qword [esp]
+					|	fisttp qword [esp]
+					|	fld qword [esp+8]
+					|	fisttp qword [esp+8]
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+					|	add esp, 16
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
+			}
 		} else {
-			if (IR_MEM_SCALE(mem) == 2 && IR_MEM_BASE(mem) == IR_REG_NONE) {
-				mem = IR_MEM(IR_MEM_INDEX(mem), IR_MEM_OFFSET(mem), IR_MEM_INDEX(mem), 1);
+			if (dst_type == IR_U8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtpd2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+						|	vpackuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	pshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+					|	packuswb xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_U16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvtpd2dq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						|	vpackusdw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vpshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvtpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	pshuflw xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 232
+				}
+			} else if (dst_type == IR_U32) {
+				if (!data->v_d_to_u32) {
+					data->v_d_to_u32 = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+						|->v_d_to_u32:
+						|	.qword 0xc1e0000000000000
+						|	.qword 0xc1e0000000000000
+						|	.qword 0xc1e0000000000000
+						|	.qword 0xc1e0000000000000
+					} else {
+						|.align 16
+						|->v_d_to_u32:
+						|	.qword 0xc1e0000000000000
+						|	.qword 0xc1e0000000000000
+					}
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	vcvttpd2dq xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						|	vaddpd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_d_to_u32]
+						|	vcvttpd2dq xmm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+						|	vpsrad xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 31
+						|	vandpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vcvttpd2dq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vaddpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [->v_d_to_u32]
+						|	vcvttpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						|	vpsrad xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 31
+						|	vandpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	cvttpd2dq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	movapd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					if (def_reg != op1_reg) {
+						|	movapd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	addpd xmm(def_reg-IR_REG_FP_FIRST), [->v_d_to_u32]
+					|	cvttpd2dq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					|	psrad xmm(tmp2_reg-IR_REG_FP_FIRST), 31
+					|	andpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+					|	orpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (dst_type == IR_U64) {
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (!data->ull2d_const) {
+					data->ull2d_const = 1;
+					ir_rodata(ctx);
+					|.align 8
+					|->ull2d_const:
+					|.dword 0, 0x43e00000
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						ir_reg tmp3_reg = ctx->tmp_regs[def];
+
+						|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vucomisd xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	jnb >1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubsd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vshufpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						|	vucomisd xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	jnb >1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubsd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vucomisd xmm(op1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	jnb >1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubsd xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vshufpd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vucomisd xmm(def_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	jnb >1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubsd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	vcvttsd2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vinserti128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						}
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vunpckhpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vucomisd xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	jnb >1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubsd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	vcvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vucomisd xmm(op1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	jnb >1
+						|	vcvttsd2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	jmp >2
+						|1:
+						|	vsubsd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+						|	vcvttsd2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+						|	btc Rq(tmp2_reg), 63
+						|2:
+						|	vmovq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vpunpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	movapd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	unpckhpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	ucomisd xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+					|	jnb >1
+					|	cvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	jmp >2
+					|1:
+					|	subsd xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+					|	cvttsd2si Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	btc Rq(tmp2_reg), 63
+					|2:
+					|	movq xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	ucomisd xmm(op1_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+					|	jnb >1
+					|	cvttsd2si Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	jmp >2
+					|1:
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	subsd xmm(def_reg-IR_REG_FP_FIRST), qword [->ull2d_const]
+					|	cvttsd2si Rq(tmp2_reg), xmm(def_reg-IR_REG_FP_FIRST)
+					|	btc Rq(tmp2_reg), 63
+					|2:
+					|	movq xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	punpcklqdq xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif defined(IR_TARGET_X86)
+//				IR_ASSERT(0 && "v_d -> v_u64 (full unsigned range) ???");
+|.if not X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						|	sub esp, 32
+						|	vmovupd yword [esp], ymm(op1_reg-IR_REG_FP_FIRST)
+						|	fld qword [esp]
+						|	fisttp qword [esp]
+						|	fld qword [esp+8]
+						|	fisttp qword [esp+8]
+						|	fld qword [esp+16]
+						|	fisttp qword [esp+16]
+						|	fld qword [esp+24]
+						|	fisttp qword [esp+24]
+						|	vmovdqu ymm(def_reg-IR_REG_FP_FIRST), yword [esp]
+						|	add esp, 32
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	sub esp, 16
+						|	vmovapd oword [esp], xmm(op1_reg-IR_REG_FP_FIRST)
+						|	fld qword [esp]
+						|	fisttp qword [esp]
+						|	fld qword [esp+8]
+						|	fisttp qword [esp+8]
+						|	vmovdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+						|	add esp, 16
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	sub esp, 16
+					|	movapd oword [esp], xmm(op1_reg-IR_REG_FP_FIRST)
+					|	fld qword [esp]
+					|	fisttp qword [esp]
+					|	fld qword [esp+8]
+					|	fisttp qword [esp+8]
+					|	movdqa xmm(def_reg-IR_REG_FP_FIRST), oword [esp]
+					|	add esp, 16
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
 			}
-			|	ASM_TXT_TMEM_OP lea, Ra(def_reg), aword, mem
 		}
 	}
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, type, def, def_reg);
-	}
-}
-
-static void ir_emit_tls(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_reg reg = IR_REG_NUM(ctx->regs[def][0]);
-
-	if (ctx->use_lists[def].count == 1) {
-		/* dead load */
-		return;
-	}
-
-|.if X64WIN
-|	gs
-|	mov Ra(reg), aword [0x58]
-|	mov Ra(reg), aword [Ra(reg)+insn->op2]
-|	mov Ra(reg), aword [Ra(reg)+insn->op3]
-|.elif WIN
-|	fs
-|	mov Ra(reg), aword [0x2c]
-|	mov Ra(reg), aword [Ra(reg)+insn->op2]
-|	mov Ra(reg), aword [Ra(reg)+insn->op3]
-|.elif X64APPLE
-|	gs
-||	if (insn->op3 == IR_NULL) {
-|		mov Ra(reg), aword [insn->op2]
-||	} else {
-|		mov Ra(reg), aword [insn->op2]
-|		mov Ra(reg), aword [Ra(reg)+insn->op3]
-||	}
-|.elif X64
-|	fs
-||	if (insn->op3 == IR_NULL) {
-|		mov Ra(reg), aword [insn->op2]
-||	} else {
-|		mov Ra(reg), [0x8]
-|		mov Ra(reg), aword [Ra(reg)+insn->op2]
-|		mov Ra(reg), aword [Ra(reg)+insn->op3]
-||	}
-|.else
-|	gs
-||	if (insn->op3 == IR_NULL) {
-|		mov Ra(reg), aword [insn->op2]
-||	} else {
-|		mov Ra(reg), [0x4]
-|		mov Ra(reg), aword [Ra(reg)+insn->op2]
-|		mov Ra(reg), aword [Ra(reg)+insn->op3]
-||	}
-|	.endif
-	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, IR_ADDR, def, reg);
-	}
-}
-
-static void ir_emit_sse_sqrt(ir_ctx *ctx, ir_ref def, ir_insn *insn)
-{
-	ir_backend_data *data = ctx->data;
-	dasm_State **Dst = &data->dasm_state;
-	ir_reg op3_reg = ctx->regs[def][3];
-	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
-
-	IR_ASSERT(IR_IS_TYPE_FP(insn->type));
-	IR_ASSERT(def_reg != IR_REG_NONE && op3_reg != IR_REG_NONE);
-
-	if (IR_REG_SPILLED(op3_reg)) {
-		op3_reg = IR_REG_NUM(op3_reg);
-		ir_emit_load(ctx, insn->type, op3_reg, insn->op3);
-	}
-
-	|	ASM_FP_REG_REG_OP sqrts, insn->type, def_reg, op3_reg

 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
+		ir_emit_store(ctx, dst_type, def, def_reg);
 	}
 }

-static void ir_emit_sse_round(ir_ctx *ctx, ir_ref def, ir_insn *insn, int round_op)
+static void ir_emit_vector_int2fp(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
+	ir_type dst_type = insn->type;
+	ir_type src_type = ctx->ir_base[insn->op1].type;
 	ir_backend_data *data = ctx->data;
 	dasm_State **Dst = &data->dasm_state;
-	ir_reg op3_reg = ctx->regs[def][3];
 	ir_reg def_reg = IR_REG_NUM(ctx->regs[def][0]);
+	ir_reg op1_reg = ctx->regs[def][1];
+	ir_reg tmp1_reg = ctx->regs[def][2];
+	ir_reg tmp2_reg = ctx->regs[def][3];
+	uint32_t src_width, dst_width;
+
+	IR_ASSERT(IR_IS_TYPE_VECTOR(src_type));
+	IR_ASSERT(IR_IS_TYPE_VECTOR(dst_type));
+	src_width = IR_VECTOR_SIZE(src_type);
+	dst_width = IR_VECTOR_SIZE(dst_type);
+	src_type = IR_VECTOR_BASE_TYPE(src_type);
+	dst_type = IR_VECTOR_BASE_TYPE(dst_type);
+
+	IR_ASSERT(IR_IS_TYPE_INT(src_type) && IR_IS_TYPE_FP(dst_type));
+	IR_ASSERT(def_reg != IR_REG_NONE && op1_reg != IR_REG_NONE);

-	IR_ASSERT(IR_IS_TYPE_FP(insn->type));
-	IR_ASSERT(def_reg != IR_REG_NONE && op3_reg != IR_REG_NONE);
-
-	if (IR_REG_SPILLED(op3_reg)) {
-		op3_reg = IR_REG_NUM(op3_reg);
-		ir_emit_load(ctx, insn->type, op3_reg, insn->op3);
-	}
+	if (dst_type == IR_FLOAT) {
+		if (IR_IS_TYPE_SIGNED(src_type)) {
+			if (src_type == IR_I8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpmovsxbd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxbd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vcvtdq2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_I16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpmovsxwd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovsxwd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpshufd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+							|	vpmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vcvtdq2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_I32) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	vcvtdq2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vcvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	cvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_I64) {
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						ir_reg tmp3_reg = ctx->tmp_regs[def];

-	if (ctx->mflags & IR_X86_AVX) {
-		|	ASM_SSE2_REG_REG_REG_TXT_OP vrounds, insn->type, def_reg, def_reg, op3_reg, round_op
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp3_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vextractf128 xmm(tmp3_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vpextrq  Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST), 1
+						|	vcvtsi2ss xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovq  Rq(tmp2_reg), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	vcvtsi2ss xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vunpcklps xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vpextrq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvtsi2ss xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vunpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovlhps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vpextrq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvtsi2ss xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vunpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(src_width <= 16);
+					|	pextrq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST), 1
+					|	cvtsi2ss xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	movq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	unpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	pshufd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+					|	movq Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	cvtsi2ss xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	movq Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtsi2ss xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	unpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif defined(IR_TARGET_X86)
+|.if not X64
+				if (!data->v_i64_to_d) {
+					data->v_i64_to_d = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+					} else {
+						|.align 16
+					}
+					|->v_i64_to_d_1:
+					|	.qword 0x4438000000000000
+					|	.qword 0x4438000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4438000000000000
+						|	.qword 0x4438000000000000
+					}
+					|->v_i64_to_d_2:
+					|	.qword 0x4330000000000000
+					|	.qword 0x4330000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4330000000000000
+						|	.qword 0x4330000000000000
+					}
+					|->v_i64_to_d_3:
+					|	.qword 0xc438001000000000
+					|	.qword 0xc438001000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0xc438001000000000
+						|	.qword 0xc438001000000000
+					}
+					if (!(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX))) {
+						|->v_i64_to_d_4:
+						|	.qword 0xffffffff
+						|	.qword 0xffffffff
+						|->v_i64_to_d_5:
+						|	.qword 0x0000ffff
+						|	.qword 0x0000ffff
+					} else if ((ctx->mflags & IR_X86_AVX) && !(ctx->mflags & IR_X86_AVX2)) {
+						|->v_i64_to_d_6:
+						|	.qword 0xffffffff00000000
+						|	.qword 0xffffffff00000000
+						|	.qword 0xffffffff00000000
+						|	.qword 0xffffffff00000000
+						|->v_i64_to_d_7:
+						|	.qword 0x0000ffffffffffff
+						|	.qword 0x0000ffffffffffff
+						|	.qword 0x0000ffffffffffff
+						|	.qword 0x0000ffffffffffff
+					}
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpsrad ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 16
+							|	vpxor ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+							|	vpblendw ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 204
+							|	vpaddq ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+							|	vpblendw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2], 136
+						} else {
+							|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	vpsrad xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 16
+							|	vpsrad xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+							|	vinsertf128 ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							|	vandps ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_6]
+							|	vpaddq ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+							|	vandps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_7]
+							|	vorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2]
+						}
+						|	vaddpd ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+						|	vaddpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vcvtpd2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vpsrad xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+						|	vpxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vpblendw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 204
+						|	vpaddq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+						|	vpblendw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2], 136
+						|	vaddpd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+						|	vaddpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vcvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(src_width <= 16);
+					|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	psrad xmm(tmp1_reg-IR_REG_FP_FIRST), 16
+					|	pxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+					|	pblendw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 204
+					|	paddq xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	pblendw xmm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2], 136
+					|	addpd xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+					|	addpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+					|	cvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(src_width <= 16);
+					|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	psrad xmm(tmp1_reg-IR_REG_FP_FIRST), 16
+					|	pand xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_4]
+					|	paddq xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	pand xmm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_5]
+					|	por xmm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2]
+					|	addpd xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+					|	addpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	cvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
+			}
+		} else {
+			if (src_type == IR_U8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpmovzxbd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpmovzxbd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vmovshdup xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vpmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vcvtdq2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_U16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpmovzxwd ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vpxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vpunpckhwd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vpmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vcvtdq2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_U32) {
+				if (!data->v_u32_to_f) {
+					data->v_u32_to_f = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+						|->v_u32_to_f_1:
+						|	.dword 0xffff, 0xffff, 0xffff, 0xffff
+						|	.dword 0xffff, 0xffff, 0xffff, 0xffff
+						|->v_u32_to_f_2:
+						|	.dword 0x4b000000, 0x4b000000, 0x4b000000, 0x4b000000
+						|	.dword 0x4b000000, 0x4b000000, 0x4b000000, 0x4b000000
+						|->v_u32_to_f_3:
+						|	.dword 0x53000000, 0x53000000, 0x53000000, 0x53000000
+						|	.dword 0x53000000, 0x53000000, 0x53000000, 0x53000000
+						|->v_u32_to_f_4:
+						|	.dword 0x53000080, 0x53000080, 0x53000080, 0x53000080
+						|	.dword 0x53000080, 0x53000080, 0x53000080, 0x53000080
+						|.code
+					} else {
+						|.align 16
+						|->v_u32_to_f_1:
+						|	.dword 0xffff, 0xffff, 0xffff, 0xffff
+						|->v_u32_to_f_2:
+						|	.dword 0x4b000000, 0x4b000000, 0x4b000000, 0x4b000000
+						|->v_u32_to_f_3:
+						|	.dword 0x53000000, 0x53000000, 0x53000000, 0x53000000
+						|->v_u32_to_f_4:
+						|	.dword 0x53000080, 0x53000080, 0x53000080, 0x53000080
+						|.code
+					}
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpand ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_1]
+							|	vpor ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_2]
+							|	vpsrld ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 16
+							|	vpor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_3]
+						} else {
+							|	vandps ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_1]
+							|	vorps ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_2]
+							|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	vpsrld xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 16
+							|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							|	vorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_3]
+						}
+						|	vsubps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_4]
+						|	vaddps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpand xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_1]
+						|	vpor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_2]
+						|	vpsrld xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+						|	vpor xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_3]
+						|	vsubps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_4]
+						|	vaddps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa  xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_1]
+					|	pand xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	por xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_u32_to_f_2]
+					|	psrld xmm(def_reg-IR_REG_FP_FIRST), 16
+					|	por xmm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_3]
+					|	subps xmm(def_reg-IR_REG_FP_FIRST), [->v_u32_to_f_4]
+					|	addps xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_U64) {
+				if (!data->v_u64_to_d) {
+					data->v_u64_to_d = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+					} else {
+						|.align 16
+					}
+					|->v_u64_to_d_1:
+					|	.qword 0x4330000000000000
+					|	.qword 0x4330000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4330000000000000
+						|	.qword 0x4330000000000000
+					}
+					|->v_u64_to_d_2:
+					|	.qword 0x4530000000000000
+					|	.qword 0x4530000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4530000000000000
+						|	.qword 0x4530000000000000
+					}
+					|->v_u64_to_d_3:
+					|	.qword 0x4530000000100000
+					|	.qword 0x4530000000100000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4530000000100000
+						|	.qword 0x4530000000100000
+					}
+					if (!(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX))) {
+						|->v_u64_to_d_4:
+						|	.qword 0xffffffff
+						|	.qword 0xffffffff
+					}
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (src_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vpblendd ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 170
+							|	vpor ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_1]
+							|	vpsrlq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 32
+							|	vpor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_2]
+						} else {
+							|	vxorps xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vblendps ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 170
+							|	vorps ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_1]
+							|	vshufps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 221
+							|	vshufps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 216
+							|	vorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_2]
+						}
+						|	vsubpd  ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_3]
+						|	vaddpd  ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vcvtpd2ps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(src_width <= 16);
+						|	vpxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vpblendw xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 204
+						|	vpor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_1]
+						|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 32
+						|	vpor    xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_2]
+						|	vsubpd  xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_3]
+						|	vaddpd  xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vcvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(src_width <= 16);
+					if (ctx->mflags & IR_X86_SSE41) {
+						|	pxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	pblendw xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 51
+					} else {
+						|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_4]
+						|	pand xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	por xmm(tmp1_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_1]
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	psrlq xmm(def_reg-IR_REG_FP_FIRST), 32
+					|	por    xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_2]
+					|	subpd  xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_3]
+					|	addpd  xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	cvtpd2ps xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(0);
+			}
+		}
 	} else {
-		|	ASM_SSE2_REG_REG_TXT_OP rounds, insn->type, def_reg, op3_reg, round_op
+		IR_ASSERT(dst_type == IR_DOUBLE);
+		if (IR_IS_TYPE_SIGNED(src_type)) {
+			if (src_type == IR_I8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	vpmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovsxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_I16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	vpmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovsxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_I32) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	vcvtdq2pd ymm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vcvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	cvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_I64) {
+#if defined(IR_TARGET_X64)
+|.if X64
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						ir_reg tmp3_reg = ctx->tmp_regs[def];
+
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vextracti128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						} else {
+							|	vextractf128 xmm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+						}
+						|	vpextrq  Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+						|	vcvtsi2sd xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovq  Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vcvtsi2sd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vunpcklpd xmm(tmp3_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST)
+						|	vpextrq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvtsi2sd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vunpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp3_reg-IR_REG_FP_FIRST), 1
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpextrq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST), 1
+						|	vcvtsi2sd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vmovq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+						|	vunpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					|	pextrq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST), 1
+					|	cvtsi2sd xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	movq  Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	unpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pshufd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 238
+					|	movq Rq(tmp2_reg), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	cvtsi2sd xmm(tmp1_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	movq Rq(tmp2_reg), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtsi2sd xmm(def_reg-IR_REG_FP_FIRST), Rq(tmp2_reg)
+					|	unpcklpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#elif defined(IR_TARGET_X86)
+|.if not X64
+				if (!data->v_i64_to_d) {
+					data->v_i64_to_d = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+					} else {
+						|.align 16
+					}
+					|->v_i64_to_d_1:
+					|	.qword 0x4438000000000000
+					|	.qword 0x4438000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4438000000000000
+						|	.qword 0x4438000000000000
+					}
+					|->v_i64_to_d_2:
+					|	.qword 0x4330000000000000
+					|	.qword 0x4330000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4330000000000000
+						|	.qword 0x4330000000000000
+					}
+					|->v_i64_to_d_3:
+					|	.qword 0xc438001000000000
+					|	.qword 0xc438001000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0xc438001000000000
+						|	.qword 0xc438001000000000
+					}
+					if (!(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX))) {
+						|->v_i64_to_d_4:
+						|	.qword 0xffffffff
+						|	.qword 0xffffffff
+						|->v_i64_to_d_5:
+						|	.qword 0x0000ffff
+						|	.qword 0x0000ffff
+					} else if ((ctx->mflags & IR_X86_AVX) && !(ctx->mflags & IR_X86_AVX2)) {
+						|->v_i64_to_d_6:
+						|	.qword 0xffffffff00000000
+						|	.qword 0xffffffff00000000
+						|	.qword 0xffffffff00000000
+						|	.qword 0xffffffff00000000
+						|->v_i64_to_d_7:
+						|	.qword 0x0000ffffffffffff
+						|	.qword 0x0000ffffffffffff
+						|	.qword 0x0000ffffffffffff
+						|	.qword 0x0000ffffffffffff
+					}
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpsrad ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 16
+							|	vpxor ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+							|	vpblendw ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 204
+							|	vpaddq ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+							|	vpblendw ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2], 136
+						} else {
+							|	vextractf128 xmm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 1
+							|	vpsrad xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 16
+							|	vpsrad xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+							|	vinsertf128 ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), 1
+							|	vandps ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_6]
+							|	vpaddq ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+							|	vandps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_7]
+							|	vorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2]
+						}
+						|	vaddpd ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+						|	vaddpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpsrad xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 16
+						|	vpxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+						|	vpblendw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 204
+						|	vpaddq xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+						|	vpblendw xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2], 136
+						|	vaddpd xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+						|	vaddpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+					}
+				} else if (ctx->mflags & IR_X86_SSE41) {
+					IR_ASSERT(dst_width <= 16);
+					|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	psrad xmm(tmp1_reg-IR_REG_FP_FIRST), 16
+					|	pxor xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+					|	pblendw xmm(tmp2_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 204
+					|	paddq xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	pblendw xmm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2], 136
+					|	addpd xmm(tmp2_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+					|	addpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp2_reg-IR_REG_FP_FIRST)
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	psrad xmm(tmp1_reg-IR_REG_FP_FIRST), 16
+					|	pand xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_4]
+					|	paddq xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_1]
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	pand xmm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_5]
+					|	por xmm(def_reg-IR_REG_FP_FIRST), [->v_i64_to_d_2]
+					|	addpd xmm(tmp1_reg-IR_REG_FP_FIRST), [->v_i64_to_d_3]
+					|	addpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+|.endif
+#endif
+			} else {
+				IR_ASSERT(0);
+			}
+		} else {
+			if (src_type == IR_U8) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	vpmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovzxbd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_U16) {
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						|	vpmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd ymm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+						|	vcvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					|	pmovzxwd xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					|	cvtdq2pd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_U32) {
+				if (!data->v_u32_to_d) {
+					data->v_u32_to_d = 1;
+					ir_rodata(ctx);
+					|.align 8
+					|->v_u32_to_d:
+					|	.qword 0x4330000000000000
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpmovzxdq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST)
+							|	vbroadcastsd ymm(tmp1_reg-IR_REG_FP_FIRST), qword [->v_u32_to_d]
+					        |	vorpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST)
+						} else {
+							|	vxorpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vpunpckhdq xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vpmovzxdq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+							|	vinsertf128 ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+							|	vmovddup xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->v_u32_to_d]
+							|	vinsertf128 ymm(tmp1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 1
+					        |	vorpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST)
+						}
+				        |	vsubpd ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vxorpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vunpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vmovddup xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->v_u32_to_d]
+				        |	vorpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				        |	vsubpd xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	xorpd xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	unpcklps xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					|	vmovddup xmm(tmp1_reg-IR_REG_FP_FIRST), qword [->v_u32_to_d]
+			        |	orpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+			        |	subpd xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+			} else if (src_type == IR_U64) {
+				if (!data->v_u64_to_d) {
+					data->v_u64_to_d = 1;
+					ir_rodata(ctx);
+					if (ctx->mflags & IR_X86_AVX) {
+						|.align 32
+					} else {
+						|.align 16
+					}
+					|->v_u64_to_d_1:
+					|	.qword 0x4330000000000000
+					|	.qword 0x4330000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4330000000000000
+						|	.qword 0x4330000000000000
+					}
+					|->v_u64_to_d_2:
+					|	.qword 0x4530000000000000
+					|	.qword 0x4530000000000000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4530000000000000
+						|	.qword 0x4530000000000000
+					}
+					|->v_u64_to_d_3:
+					|	.qword 0x4530000000100000
+					|	.qword 0x4530000000100000
+					if (ctx->mflags & IR_X86_AVX) {
+						|	.qword 0x4530000000100000
+						|	.qword 0x4530000000100000
+					}
+					if (!(ctx->mflags & (IR_X86_SSE41|IR_X86_AVX))) {
+						|->v_u64_to_d_4:
+						|	.qword 0xffffffff
+						|	.qword 0xffffffff
+					}
+					|.code
+				}
+				if (ctx->mflags & IR_X86_AVX) {
+					if (dst_width == 32) {
+						if (ctx->mflags & IR_X86_AVX2) {
+							|	vpxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vpblendd ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 170
+							|	vpor ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_1]
+							|	vpsrlq ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), 32
+							|	vpor ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_2]
+						} else {
+							|	vxorps xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+							|	vblendps ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 170
+							|	vorps ymm(tmp2_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_1]
+							|	vshufps ymm(def_reg-IR_REG_FP_FIRST), ymm(op1_reg-IR_REG_FP_FIRST), ymm(tmp1_reg-IR_REG_FP_FIRST), 221
+							|	vshufps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), 216
+							|	vorps ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_2]
+						}
+						|	vsubpd  ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), yword [->v_u64_to_d_3]
+						|	vaddpd  ymm(def_reg-IR_REG_FP_FIRST), ymm(def_reg-IR_REG_FP_FIRST), ymm(tmp2_reg-IR_REG_FP_FIRST)
+					} else {
+						IR_ASSERT(dst_width <= 16);
+						|	vpxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	vpblendw xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), 204
+						|	vpor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_1]
+						|	vpsrlq xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 32
+						|	vpor    xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_2]
+						|	vsubpd  xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_3]
+						|	vaddpd  xmm(def_reg-IR_REG_FP_FIRST), xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+					}
+				} else {
+					IR_ASSERT(dst_width <= 16);
+					if (ctx->mflags & IR_X86_SSE41) {
+						|	pxor xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+						|	pblendw xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST), 51
+					} else {
+						|	movdqa xmm(tmp1_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_4]
+						|	pand xmm(tmp1_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	por xmm(tmp1_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_1]
+					if (def_reg != op1_reg) {
+						|	movdqa xmm(def_reg-IR_REG_FP_FIRST), xmm(op1_reg-IR_REG_FP_FIRST)
+					}
+					|	psrlq xmm(def_reg-IR_REG_FP_FIRST), 32
+					|	por    xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_2]
+					|	subpd  xmm(def_reg-IR_REG_FP_FIRST), oword [->v_u64_to_d_3]
+					|	addpd  xmm(def_reg-IR_REG_FP_FIRST), xmm(tmp1_reg-IR_REG_FP_FIRST)
+				}
+			} else {
+				IR_ASSERT(0);
+			}
+		}
 	}

 	if (IR_REG_SPILLED(ctx->regs[def][0])) {
-		ir_emit_store(ctx, insn->type, def, def_reg);
+		ir_emit_store(ctx, dst_type, def, def_reg);
 	}
 }
+#endif

 static void ir_emit_exitcall(ir_ctx *ctx, ir_ref def, ir_insn *insn)
 {
@@ -11077,6 +25620,15 @@ static void ir_emit_param_move(ir_ctx *ctx, uint8_t type, ir_reg from_reg, ir_re
 			} else {
 				ir_emit_store(ctx, type, to, from_reg);
 			}
+#if IR_X86_I64
+		} else if (type == IR_I64 || type == IR_U64) {
+			ir_mem mem = IR_MEM_BO(fp, offset);
+			ir_mem mem_hi = IR_MEM_I64_HI(mem);
+			ir_reg to_reg_hi = IR_REG_I64_HI(to_reg);
+			to_reg = IR_REG_I64_LO(to_reg);
+			ir_emit_load_mem_int(ctx, IR_U32, to_reg, mem);
+			ir_emit_load_mem_int(ctx, IR_U32, to_reg_hi, mem_hi);
+#endif
 		} else {
 			ir_emit_load_mem_int(ctx, type, to_reg, IR_MEM_BO(fp, offset));
 		}
@@ -11100,6 +25652,9 @@ static void ir_emit_load_params(ir_ctx *ctx)
 	ir_ref i, n, *p, use;
 	int int_param_num = 0;
 	int fp_param_num = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param_num = 0;
+#endif
 	ir_reg src_reg;
 	ir_reg dst_reg;
 	ir_backend_data *data = ctx->data;
@@ -11109,10 +25664,10 @@ static void ir_emit_load_params(ir_ctx *ctx)

 	if (ctx->flags & IR_USE_FRAME_POINTER) {
 		/* skip old frame pointer and return address */
-		stack_start = sizeof(void*) * 2 + ctx->stack_frame_size;
+		stack_start = cc->shadow_store_size + sizeof(void*) * 2 + ctx->stack_frame_size;
 	} else {
 		 /* skip return address */
-		stack_start = sizeof(void*) + ctx->stack_frame_size;
+		stack_start = cc->shadow_store_size + sizeof(void*) + ctx->stack_frame_size;
 	}
 	n = use_list->count;
 	for (i = 0, p = &ctx->use_edges[use_list->refs]; i < n; i++, p++) {
@@ -11120,9 +25675,9 @@ static void ir_emit_load_params(ir_ctx *ctx)
 		insn = &ctx->ir_base[use];
 		if (insn->op == IR_PARAM) {
 			if (IR_IS_TYPE_INT(insn->type)) {
-				if (ctx->value_params && ctx->value_params[insn->op3 - 1].align) {
+				if (ctx->value_params && ctx->value_params[insn->op3 - 1].align && cc->pass_struct_by_val) {
 					/* struct passed by value on stack */
-					size_t align = ctx->value_params[insn->op3 - 1].align;
+					uint32_t align = ctx->value_params[insn->op3 - 1].align;

 					align = IR_MAX(sizeof(void*), align);
 					stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
@@ -11131,6 +25686,17 @@ static void ir_emit_load_params(ir_ctx *ctx)
 					continue;
 				} else if (int_param_num < cc->int_param_regs_count) {
 					src_reg = cc->int_param_regs[int_param_num];
+#if IR_X86_I64
+					if (src_reg != IR_REG_NONE && (insn->type == IR_I64 || insn->type == IR_U64)) {
+						if (int_param_num + 1 < cc->int_param_regs_count) {
+							int_param_num++;
+							if (cc->shadow_param_regs) {
+								fp_param_num++;
+							}
+						}
+						src_reg = IR_REG_NONE;
+					}
+#endif
 				} else {
 					src_reg = IR_REG_NONE;
 				}
@@ -11138,6 +25704,15 @@ static void ir_emit_load_params(ir_ctx *ctx)
 				if (cc->shadow_param_regs) {
 					fp_param_num++;
 				}
+#if IR_SIMD && defined(IR_TARGET_X86)
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (vector_param_num < cc->vector_param_regs_count) {
+					src_reg = cc->vector_param_regs[vector_param_num];
+				} else {
+					src_reg = IR_REG_NONE;
+				}
+				vector_param_num++;
+#endif
 			} else {
 				if (fp_param_num < cc->fp_param_regs_count) {
 					src_reg = cc->fp_param_regs[fp_param_num];
@@ -11164,25 +25739,13 @@ static void ir_emit_load_params(ir_ctx *ctx)
 				if (sizeof(void*) == 8) {
 					stack_offset += sizeof(void*);
 				} else {
-					stack_offset += IR_MAX(sizeof(void*), ir_type_size[insn->type]);
+					stack_offset += IR_MAX(sizeof(void*), ir_get_type_size(insn->type));
 				}
 			}
 		}
 	}
 }

-static ir_reg ir_get_free_reg(ir_type type, ir_regset available)
-{
-	if (IR_IS_TYPE_INT(type)) {
-		available = IR_REGSET_INTERSECTION(available, IR_REGSET_GP);
-	} else {
-		IR_ASSERT(IR_IS_TYPE_FP(type));
-		available = IR_REGSET_INTERSECTION(available, IR_REGSET_FP);
-	}
-	IR_ASSERT(!IR_REGSET_IS_EMPTY(available));
-	return IR_REGSET_FIRST(available);
-}
-
 static int ir_fix_dessa_tmps(ir_ctx *ctx, uint8_t type, ir_ref from, ir_ref to, void *dessa_from_block)
 {
 	ir_ref ref = ctx->cfg_blocks[(intptr_t)dessa_from_block].end;
@@ -11213,6 +25776,57 @@ static int ir_fix_dessa_tmps(ir_ctx *ctx, uint8_t type, ir_ref from, ir_ref to,
 	return 1;
 }

+/* This remaps spill slot of the VAR, where PARAM is stored, to the PARAMs slot.
+ * This eliminates copying, but more important, this fixes va_start() compilation
+ * on systems, where it's defined through address calculation macro (e.g. MSVC x86).
+ */
+static void ir_remap_param_spill_slot(ir_ctx *ctx, ir_ref param, int32_t spill_slot)
+{
+	ir_use_list *param_use_list = &ctx->use_lists[param];
+
+	if (param_use_list->count >= 1) {
+		ir_ref param_store = ctx->use_edges[param_use_list->refs];
+		ir_insn *param_store_insn = &ctx->ir_base[param_store];
+
+		if (param_store_insn->op == IR_VSTORE) {
+			ir_ref var = param_store_insn->op2;
+			ir_insn *var_insn = &ctx->ir_base[var];
+
+			IR_ASSERT(param_store_insn->op3 == param);
+			IR_ASSERT(ctx->ir_base[var].op == IR_VAR);
+
+			/* Remap all related VADDR and VLOAD */
+			ir_use_list *var_use_list = &ctx->use_lists[var];
+			ir_ref n = var_use_list->count;
+			ir_ref *p = &ctx->use_edges[var_use_list->refs];
+			for (; n > 0; p++, n--) {
+				ir_ref use = *p;
+				ir_insn *insn = &ctx->ir_base[use];
+
+				if (insn->op == IR_VADDR) {
+					if (ctx->rules[use] == IR_STATIC_ALLOCA) {
+						IR_ASSERT(insn->op3 == var_insn->op3);
+						insn->op3 = spill_slot;
+					}
+				} else if (insn->op == IR_VLOAD) {
+					if (ctx->vregs[use]) {
+						ir_live_interval *ival = ctx->live_intervals[ctx->vregs[use]];
+						if (ival->stack_spill_pos == var_insn->op3) {
+							ival->stack_spill_pos = spill_slot;
+						}
+					}
+				}
+			}
+
+			/* Remap VAR itself */
+			var_insn->op3 = spill_slot;
+
+			/* Avoid copying to itself */
+			ctx->rules[param_store] = IR_SKIPPED | IR_NOP;
+		}
+	}
+}
+
 static void ir_fix_param_spills(ir_ctx *ctx)
 {
 	ir_use_list *use_list = &ctx->use_lists[1];
@@ -11220,6 +25834,9 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 	ir_ref i, n, *p, use;
 	int int_param_num = 0;
 	int fp_param_num = 0;
+#if IR_SIMD && defined(IR_TARGET_X86)
+	int vector_param_num = 0;
+#endif
 	ir_reg src_reg;
 	ir_backend_data *data = ctx->data;
 	const ir_call_conv_dsc *cc = data->ra_data.cc;
@@ -11228,10 +25845,10 @@ static void ir_fix_param_spills(ir_ctx *ctx)

 	if (ctx->flags & IR_USE_FRAME_POINTER) {
 		/* skip old frame pointer and return address */
-		stack_start = sizeof(void*) * 2 + ctx->stack_frame_size;
+		stack_start = cc->shadow_store_size + sizeof(void*) * 2 + ctx->stack_frame_size;
 	} else {
 		 /* skip return address */
-		stack_start = sizeof(void*) + ctx->stack_frame_size;
+		stack_start = cc->shadow_store_size + sizeof(void*) + ctx->stack_frame_size;
 	}
 	n = use_list->count;
 	for (i = 0, p = &ctx->use_edges[use_list->refs]; i < n; i++, p++) {
@@ -11241,7 +25858,7 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 			if (IR_IS_TYPE_INT(insn->type)) {
 				if (ctx->value_params && ctx->value_params[insn->op3 - 1].align && cc->pass_struct_by_val) {
 					/* struct passed by value on stack */
-					size_t align = ctx->value_params[insn->op3 - 1].align;
+					uint32_t align = ctx->value_params[insn->op3 - 1].align;

 					align = IR_MAX(sizeof(void*), align);
 					stack_offset = IR_ALIGNED_SIZE(stack_offset, align);
@@ -11252,6 +25869,17 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 				}
 				if (int_param_num < cc->int_param_regs_count) {
 					src_reg = cc->int_param_regs[int_param_num];
+#if IR_X86_I64
+					if (src_reg != IR_REG_NONE && (insn->type == IR_I64 || insn->type == IR_U64)) {
+						if (int_param_num + 1 < cc->int_param_regs_count) {
+							int_param_num++;
+							if (cc->shadow_param_regs) {
+								fp_param_num++;
+							}
+						}
+						src_reg = IR_REG_NONE;
+					}
+#endif
 				} else {
 					src_reg = IR_REG_NONE;
 				}
@@ -11259,6 +25887,15 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 				if (cc->shadow_param_regs) {
 					fp_param_num++;
 				}
+#if IR_SIMD && defined(IR_TARGET_X86)
+			} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+				if (vector_param_num < cc->vector_param_regs_count) {
+					src_reg = cc->vector_param_regs[vector_param_num];
+				} else {
+					src_reg = IR_REG_NONE;
+				}
+				vector_param_num++;
+#endif
 			} else {
 				if (fp_param_num < cc->fp_param_regs_count) {
 					src_reg = cc->fp_param_regs[fp_param_num];
@@ -11274,15 +25911,16 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 				if (ctx->vregs[use]) {
 					ir_live_interval *ival = ctx->live_intervals[ctx->vregs[use]];
 					if ((ival->flags & IR_LIVE_INTERVAL_MEM_PARAM)
-					 && ival->stack_spill_pos == -1
-					 && (ival->next || ival->reg == IR_REG_NONE)) {
+					 && ival->stack_spill_pos == -1) {
 						ival->stack_spill_pos = stack_start + stack_offset;
+						/* Remap VAR to PARAM stack slot */
+						ir_remap_param_spill_slot(ctx, use, stack_start + stack_offset);
 					}
 				}
 				if (sizeof(void*) == 8) {
 					stack_offset += sizeof(void*);
 				} else {
-					stack_offset += IR_MAX(sizeof(void*), ir_type_size[insn->type]);
+					stack_offset += IR_MAX(sizeof(void*), ir_get_type_size(insn->type));
 				}
 			}
 		}
@@ -11298,228 +25936,6 @@ static void ir_fix_param_spills(ir_ctx *ctx)
 	ctx->param_stack_size = stack_offset;
 }

-static void ir_allocate_unique_spill_slots(ir_ctx *ctx)
-{
-	uint32_t b;
-	ir_block *bb;
-	ir_insn *insn;
-	ir_ref i, n, j, *p;
-	uint32_t *rule, insn_flags;
-	ir_regset available = 0;
-	ir_target_constraints constraints;
-	uint32_t def_flags;
-	ir_reg reg;
-	ir_backend_data *data = ctx->data;
-	const ir_call_conv_dsc *cc = data->ra_data.cc;
-	ir_regset scratch = ir_scratch_regset[cc->scratch_reg - IR_REG_NUM];
-
-#ifdef IR_TARGET_X86
-	if (ctx->flags2 & IR_HAS_FP_RET_SLOT) {
-		ctx->ret_slot = ir_allocate_spill_slot(ctx, IR_DOUBLE);
-	} else if ((ctx->ret_type == IR_FLOAT || ctx->ret_type == IR_DOUBLE)
-			&& cc->fp_ret_reg == IR_REG_NONE) {
-		ctx->ret_slot = ir_allocate_spill_slot(ctx, ctx->ret_type);
-	} else {
-		ctx->ret_slot = -1;
-	}
-#endif
-
-	ctx->regs = ir_mem_malloc(sizeof(ir_regs) * ctx->insns_count);
-	memset(ctx->regs, IR_REG_NONE, sizeof(ir_regs) * ctx->insns_count);
-
-	/* vregs + tmp + fixed + SRATCH + ALL */
-	ctx->live_intervals = ir_mem_calloc(ctx->vregs_count + 1 + IR_REG_NUM + 2, sizeof(ir_live_interval*));
-
-    if (!ctx->arena) {
-		ctx->arena = ir_arena_create(16 * 1024);
-	}
-
-	for (b = 1, bb = ctx->cfg_blocks + b; b <= ctx->cfg_blocks_count; b++, bb++) {
-		IR_ASSERT(!(bb->flags & IR_BB_UNREACHABLE));
-		for (i = bb->start, insn = ctx->ir_base + i, rule = ctx->rules + i; i <= bb->end;) {
-			switch (ctx->rules ? *rule : insn->op) {
-				case IR_START:
-				case IR_BEGIN:
-				case IR_END:
-				case IR_IF_TRUE:
-				case IR_IF_FALSE:
-				case IR_CASE_VAL:
-				case IR_CASE_RANGE:
-				case IR_CASE_DEFAULT:
-				case IR_MERGE:
-				case IR_LOOP_BEGIN:
-				case IR_LOOP_END:
-				case IR_IGOTO_DUP:
-					break;
-#ifdef IR_TARGET_X86
-				case IR_CALL:
-					if (ctx->ret_slot == -1
-					 && (insn->type == IR_FLOAT || insn->type == IR_DOUBLE)) {
-						const ir_proto_t *proto = ir_call_proto(ctx, insn);
-						const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);
-
-						if (cc->fp_ret_reg == IR_REG_NONE) {
-							ctx->ret_slot = ir_allocate_spill_slot(ctx, IR_DOUBLE);
-						}
-					}
-#endif
-					IR_FALLTHROUGH;
-				default:
-					def_flags = ir_get_target_constraints(ctx, i, &constraints);
-					if (ctx->rules
-					 && *rule != IR_CMP_AND_BRANCH_INT
-					 && *rule != IR_CMP_AND_BRANCH_FP
-					 && *rule != IR_TEST_AND_BRANCH_INT
-					 && *rule != IR_GUARD_CMP_INT
-					 && *rule != IR_GUARD_CMP_FP) {
-						available = scratch;
-					}
-					if (ctx->vregs[i]) {
-						reg = constraints.def_reg;
-						if (reg != IR_REG_NONE && IR_REGSET_IN(available, reg)) {
-							IR_REGSET_EXCL(available, reg);
-							ctx->regs[i][0] = reg | IR_REG_SPILL_STORE;
-						} else if (def_flags & IR_USE_MUST_BE_IN_REG) {
-							if ((insn->op == IR_VLOAD || insn->op == IR_VLOAD_v)
-							 && ctx->live_intervals[ctx->vregs[i]]
-							 && ctx->live_intervals[ctx->vregs[i]]->stack_spill_pos != -1
-							 && ir_is_same_mem_var(ctx, i, ctx->ir_base[insn->op2].op3)) {
-								/* pass */
-							} else if (insn->op != IR_PARAM) {
-								reg = ir_get_free_reg(insn->type, available);
-								IR_REGSET_EXCL(available, reg);
-								ctx->regs[i][0] = reg | IR_REG_SPILL_STORE;
-							}
-						}
-						if (!ctx->live_intervals[ctx->vregs[i]]) {
-							ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
-							memset(ival, 0, sizeof(ir_live_interval));
-							ctx->live_intervals[ctx->vregs[i]] = ival;
-							ival->type = insn->type;
-							ival->reg = IR_REG_NONE;
-							ival->vreg = ctx->vregs[i];
-							ival->stack_spill_pos = -1;
-							if (insn->op == IR_PARAM && reg == IR_REG_NONE) {
-								ival->flags |= IR_LIVE_INTERVAL_MEM_PARAM;
-							} else {
-								ival->stack_spill_pos = ir_allocate_spill_slot(ctx, ival->type);
-							}
-						} else if (insn->op == IR_PARAM) {
-							IR_ASSERT(0 && "unexpected PARAM");
-							return;
-						}
-					} else if (insn->op == IR_VAR) {
-						ir_use_list *use_list = &ctx->use_lists[i];
-						ir_ref n = use_list->count;
-
-						if (n > 0) {
-							int32_t stack_spill_pos = insn->op3 = ir_allocate_spill_slot(ctx, insn->type);
-							ir_ref i, *p, use;
-							ir_insn *use_insn;
-
-							for (i = 0, p = &ctx->use_edges[use_list->refs]; i < n; i++, p++) {
-								use = *p;
-								use_insn = &ctx->ir_base[use];
-								if (use_insn->op == IR_VLOAD || use_insn->op == IR_VLOAD_v) {
-									if (ctx->vregs[use]
-									 && !ctx->live_intervals[ctx->vregs[use]]) {
-										ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
-										memset(ival, 0, sizeof(ir_live_interval));
-										ctx->live_intervals[ctx->vregs[use]] = ival;
-										ival->type = insn->type;
-										ival->reg = IR_REG_NONE;
-										ival->vreg = ctx->vregs[use];
-										ival->stack_spill_pos = stack_spill_pos;
-									}
-								} else if (use_insn->op == IR_VSTORE || use_insn->op == IR_VSTORE_v) {
-									if (!IR_IS_CONST_REF(use_insn->op3)
-									 && ctx->vregs[use_insn->op3]
-									 && !ctx->live_intervals[ctx->vregs[use_insn->op3]]) {
-										ir_live_interval *ival = ir_arena_alloc(&ctx->arena, sizeof(ir_live_interval));
-										memset(ival, 0, sizeof(ir_live_interval));
-										ctx->live_intervals[ctx->vregs[use_insn->op3]] = ival;
-										ival->type = insn->type;
-										ival->reg = IR_REG_NONE;
-										ival->vreg = ctx->vregs[use_insn->op3];
-										ival->stack_spill_pos = stack_spill_pos;
-									}
-								}
-							}
-						}
-					}
-
-					insn_flags = ir_op_flags[insn->op];
-					n = constraints.tmps_count;
-					if (n) {
-						do {
-							n--;
-							if (constraints.tmp_regs[n].type) {
-								ir_reg reg = ir_get_free_reg(constraints.tmp_regs[n].type, available);
-								ir_ref *ops = insn->ops;
-								IR_REGSET_EXCL(available, reg);
-								if (constraints.tmp_regs[n].num > 0) {
-									if (IR_IS_CONST_REF(ops[constraints.tmp_regs[n].num])) {
-										/* rematerialization */
-										reg |= IR_REG_SPILL_LOAD;
-									} else if (ctx->ir_base[ops[constraints.tmp_regs[n].num]].op == IR_ALLOCA ||
-											ctx->ir_base[ops[constraints.tmp_regs[n].num]].op == IR_VADDR) {
-										/* local address rematerialization */
-										reg |= IR_REG_SPILL_LOAD;
-									}
-								}
-								ctx->regs[i][constraints.tmp_regs[n].num] = reg;
-							} else {
-								ir_reg reg = constraints.tmp_regs[n].reg;
-
-								if (reg > IR_REG_NUM) {
-									available = IR_REGSET_DIFFERENCE(available, ir_scratch_regset[reg - IR_REG_NUM]);
-								} else {
-									IR_REGSET_EXCL(available, reg);
-								}
-							}
-						} while (n);
-					}
-					n = insn->inputs_count;
-					for (j = 1, p = insn->ops + 1; j <= n; j++, p++) {
-						ir_ref input = *p;
-						if (IR_OPND_KIND(insn_flags, j) == IR_OPND_DATA && input > 0 && ctx->vregs[input]) {
-							if ((def_flags & IR_DEF_REUSES_OP1_REG) && j == 1) {
-								ir_reg reg = IR_REG_NUM(ctx->regs[i][0]);
-								ctx->regs[i][1] = reg | IR_REG_SPILL_LOAD;
-							} else {
-								uint8_t use_flags = IR_USE_FLAGS(def_flags, j);
-								ir_reg reg = (j < constraints.hints_count) ? constraints.hints[j] : IR_REG_NONE;
-
-								if (reg != IR_REG_NONE && IR_REGSET_IN(available, reg)) {
-									IR_REGSET_EXCL(available, reg);
-									ctx->regs[i][j] = reg | IR_REG_SPILL_LOAD;
-								} else if (IR_IS_FOLDABLE_OP(insn->op) && j > 1 && input == insn->op1 && ctx->regs[i][1] != IR_REG_NONE) {
-									ctx->regs[i][j] = ctx->regs[i][1];
-								} else if (use_flags & IR_USE_MUST_BE_IN_REG) {
-									reg = ir_get_free_reg(ctx->ir_base[input].type, available);
-									IR_REGSET_EXCL(available, reg);
-									ctx->regs[i][j] = reg | IR_REG_SPILL_LOAD;
-								}
-							}
-						}
-					}
-					break;
-			}
-			n = ir_insn_len(insn);
-			i += n;
-			insn += n;
-			rule += n;
-		}
-		if (bb->flags & IR_BB_DESSA_MOVES) {
-			ir_gen_dessa_moves(ctx, b, ir_fix_dessa_tmps, (void*)(intptr_t)b);
-		}
-	}
-
-	ctx->used_preserved_regs = ctx->fixed_save_regset;
-	ctx->flags |= IR_NO_STACK_COMBINE;
-	ir_fix_stack_frame(ctx);
-}
-
 static void ir_preallocate_call_stack(ir_ctx *ctx)
 {
 	int call_stack_size, copy_stack, peak_call_stack_size = 0;
@@ -11527,7 +25943,7 @@ static void ir_preallocate_call_stack(ir_ctx *ctx)
 	ir_insn *insn;

 	for (i = 1, insn = ctx->ir_base + 1; i < ctx->insns_count;) {
-		if (insn->op == IR_CALL) {
+		if (insn->op == IR_CALL && (ctx->rules[i] & IR_RULE_MASK) == IR_CALL) {
 			const ir_proto_t *proto = ir_call_proto(ctx, insn);
 			const ir_call_conv_dsc *cc = ir_get_call_conv_dsc(proto ? proto->flags : IR_CC_DEFAULT);

@@ -11570,6 +25986,13 @@ void ir_fix_stack_frame(ir_ctx *ctx)
 		}
 	}

+#ifdef IR_TARGET_X86
+	if (ctx->flags2 & IR_HAS_MEMCPY) {
+		IR_REGSET_INCL(ctx->used_preserved_regs, IR_REG_RSI);
+		IR_REGSET_INCL(ctx->used_preserved_regs, IR_REG_RDI);
+	}
+#endif
+
 	if (ctx->used_preserved_regs) {
 		ir_regset used_preserved_regs = (ir_regset)ctx->used_preserved_regs;
 		ir_reg reg;
@@ -11584,7 +26007,7 @@ void ir_fix_stack_frame(ir_ctx *ctx)
 	ctx->stack_frame_size += additional_size;
 	ctx->call_stack_size = 0;

-	if (ctx->flags2 & IR_16B_FRAME_ALIGNMENT) {
+	if (ctx->flags2 & (IR_16B_FRAME_ALIGNMENT|IR_HAS_CALLS)) {
 		/* Stack must be 16 byte aligned */
 		if (!(ctx->flags & IR_FUNCTION)) {
 			while (IR_ALIGNED_SIZE(ctx->stack_frame_size, 16) != ctx->stack_frame_size) {
@@ -11594,12 +26017,25 @@ void ir_fix_stack_frame(ir_ctx *ctx)
 			while (IR_ALIGNED_SIZE(ctx->stack_frame_size + sizeof(void*) * 2, 16) != ctx->stack_frame_size + sizeof(void*) * 2) {
 				ctx->stack_frame_size += sizeof(void*);
 			}
+		} else if (ctx->flags2 & IR_16B_FRAME_ALIGNMENT) {
+			while (IR_ALIGNED_SIZE(ctx->stack_frame_size + sizeof(void*), 16) != ctx->stack_frame_size + sizeof(void*)) {
+				ctx->stack_frame_size += sizeof(void*);
+			}
+			if (ctx->flags2 & IR_HAS_CALLS) {
+				if (!(ctx->flags & IR_NO_STACK_COMBINE)) {
+					ir_preallocate_call_stack(ctx);
+				}
+				while (IR_ALIGNED_SIZE(ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*), 16) !=
+					ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*)) {
+					ctx->call_stack_size += sizeof(void*);
+				}
+			}
 		} else {
 			if (!(ctx->flags & IR_NO_STACK_COMBINE)) {
 				ir_preallocate_call_stack(ctx);
 			}
 			while (IR_ALIGNED_SIZE(ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*), 16) !=
-					ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*)) {
+				ctx->stack_frame_size + ctx->call_stack_size + sizeof(void*)) {
 				ctx->stack_frame_size += sizeof(void*);
 			}
 		}
@@ -11635,30 +26071,13 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 	size_t size;
 	ir_ref igoto_dup_ref = IR_UNUSED;
 	uint32_t igoto_dup_block = 0;
+	size_t required_alignment = 16;

+	memset(&data, 0, sizeof(data));
 	data.ra_data.cc = ir_get_call_conv_dsc(ctx->flags);
-	data.ra_data.unused_slot_4 = 0;
-	data.ra_data.unused_slot_2 = 0;
-	data.ra_data.unused_slot_1 = 0;
-	data.ra_data.handled = NULL;
-	data.rodata_label = 0;
-	data.jmp_table_label = 0;
-	data.double_neg_const = 0;
-	data.float_neg_const = 0;
-	data.double_abs_const = 0;
-	data.float_abs_const = 0;
-	data.double_zero_const = 0;
-	data.u2d_const = 0;
-	data.u2f_const = 0;
-	data.resolved_label_syms = 0;
 	ctx->data = &data;

-	if (!ctx->live_intervals) {
-		ctx->stack_frame_size = 0;
-		ctx->call_stack_size = 0;
-		ctx->used_preserved_regs = 0;
-		ir_allocate_unique_spill_slots(ctx);
-	}
+	IR_ASSERT(ctx->live_intervals != NULL);

 	if (ctx->fixed_stack_frame_size != -1) {
 		if (ctx->fixed_stack_red_zone) {
@@ -11705,6 +26124,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)

 	if (!(ctx->flags & IR_SKIP_PROLOGUE)) {
 		ir_emit_prologue(ctx);
+		if (ctx->flags2 & IR_RECURSIVE_TAILCALL) {
+			|=>0:
+		}
 	}
 	if (ctx->flags & IR_FUNCTION) {
 		ir_emit_load_params(ctx);
@@ -11790,6 +26212,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				case IR_BIT_OP:
 					ir_emit_bit_op(ctx, i, insn);
 					break;
+				case IR_AND_ZEXT:
+					ir_emit_and_zext(ctx, i, insn);
+					break;
 				case IR_SDIV_PWR2:
 					ir_emit_sdiv_pwr2(ctx, i, insn);
 					break;
@@ -11848,6 +26273,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				case IR_TESTCC_INT:
 					ir_emit_testcc_int(ctx, i, insn);
 					break;
+				case IR_TESTCC_BIT:
+					ir_emit_testcc_bit(ctx, i, insn);
+					break;
 				case IR_SETCC_INT:
 					ir_emit_setcc_int(ctx, i, insn);
 					break;
@@ -11894,6 +26322,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				case IR_TEST_AND_BRANCH_INT:
 					ir_emit_test_and_branch_int(ctx, b, i, insn, _ir_next_block(ctx, _b));
 					break;
+				case IR_TEST_AND_BRANCH_BIT:
+					ir_emit_test_and_branch_bit(ctx, b, i, insn, _ir_next_block(ctx, _b));
+					break;
 				case IR_JCC_INT:
 					{
 						ir_op op = ctx->ir_base[insn->op2].op;
@@ -11926,6 +26357,11 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 						goto next_block;
 					}
 					break;
+				case IR_GUARD_TEST_BIT:
+					if (ir_emit_guard_test_bit(ctx, b, i, insn, _ir_next_block(ctx, _b))) {
+						goto next_block;
+					}
+					break;
 				case IR_GUARD_JCC_INT:
 					if (ir_emit_guard_jcc_int(ctx, b, i, insn, _ir_next_block(ctx, _b))) {
 						goto next_block;
@@ -11940,6 +26376,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				case IR_COND_TEST_INT:
 					ir_emit_cond_test_int(ctx, i, insn);
 					break;
+				case IR_COND_TEST_BIT:
+					ir_emit_cond_test_bit(ctx, i, insn);
+					break;
 				case IR_COND_CMP_INT:
 					ir_emit_cond_cmp_int(ctx, i, insn);
 					break;
@@ -12073,6 +26512,9 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 					ir_emit_rstore(ctx, i, insn);
 					break;
 				case IR_LOAD_INT:
+#if IR_X86_I64
+				case IR_LOAD_I64:
+#endif
 					ir_emit_load_int(ctx, i, insn);
 					break;
 				case IR_LOAD_FP:
@@ -12140,12 +26582,163 @@ void *ir_emit_code(ir_ctx *ctx, size_t *size_ptr)
 				case IR_SSE_NEARBYINT:
 					ir_emit_sse_round(ctx, i, insn, 12);
 					break;
-				case IR_TLS:
-					ir_emit_tls(ctx, i, insn);
+				case IR_TLS_ADDR:
+					ir_emit_tls_addr(ctx, i, insn);
+					break;
+				case IR_TLS_LOAD:
+					ir_emit_tls_load(ctx, i, insn);
+					break;
+				case IR_TLS_STORE:
+					ir_emit_tls_store(ctx, i, insn);
 					break;
 				case IR_TRAP:
 					|	int3
 					break;
+#if IR_SIMD
+				case IR_EXTRACT:
+					ir_emit_vector_extract(ctx, i, insn);
+					break;
+				case IR_REPLACE:
+					ir_emit_vector_replace(ctx, i, insn);
+					break;
+				case IR_SPLAT:
+					ir_emit_vector_splat(ctx, i, insn);
+					break;
+				case IR_SHUFPD_11:
+				case IR_SHUFPD_22:
+				case IR_SHUFPD_12:
+				case IR_SHUFPD_21:
+				case IR_MOVSD_12:
+				case IR_SHUFPS_11:
+				case IR_SHUFPS_22:
+				case IR_SHUFPS_12:
+				case IR_SHUFPS_21:
+				case IR_SHUFPS_12_0:
+				case IR_SHUFPS_12_1:
+				case IR_SHUFPS_12_2:
+				case IR_SHUFPS_1_21:
+				case IR_SHUFPS_2_12:
+				case IR_BLENDPS_12:
+					if (ctx->mflags & IR_X86_AVX) {
+						ir_emit_vector_shuffle_avx(ctx, i, insn, (*rule) & IR_RULE_MASK);
+					} else {
+						ir_emit_vector_shuffle_sse(ctx, i, insn, (*rule) & IR_RULE_MASK);
+					}
+					break;
+				case IR_SHUFFLE:
+					ir_emit_vector_shuffle(ctx, i, insn);
+					break;
+				case IR_VECTOR_OP:
+					ir_emit_vector_op(ctx, i, insn);
+					break;
+				case IR_VECTOR_BINOP_SSE2:
+					ir_emit_vector_binop_sse2(ctx, i, insn);
+					break;
+				case IR_VECTOR_BINOP_AVX:
+					ir_emit_vector_binop_avx(ctx, i, insn);
+					break;
+				case IR_VECTOR_BINOP_EXPAND:
+					ir_emit_vector_binop_expand(ctx, i, insn);
+					break;
+				case IR_VECTOR_EXT:
+					ir_emit_vector_ext(ctx, i, insn);
+					break;
+				case IR_VECTOR_TRUNC:
+					ir_emit_vector_trunc(ctx, i, insn);
+					break;
+				case IR_VECTOR_FP2FP:
+					ir_emit_vector_fp2fp(ctx, i, insn);
+					break;
+				case IR_VECTOR_FP2INT:
+					ir_emit_vector_fp2int(ctx, i, insn);
+					break;
+				case IR_VECTOR_INT2FP:
+					ir_emit_vector_int2fp(ctx, i, insn);
+					break;
+#endif
+#if IR_X86_I64
+				case IR_CMP_I64:
+					ir_emit_cmp_i64(ctx, i, insn);
+					break;
+				case IR_CMP_AND_BRANCH_I64:
+					ir_emit_cmp_and_branch_i64(ctx, b, i, insn, _ir_next_block(ctx, _b));
+					break;
+				case IR_BINOP_I64:
+					ir_emit_binop_i64(ctx, i, insn);
+					break;
+				case IR_MUL_I64:
+					ir_emit_mul_i64(ctx, i, insn);
+					break;
+				case IR_MUL_OV_I64:
+					ir_emit_mul_ov_i64(ctx, i, insn);
+					break;
+				case IR_BINOP_HELPER_I64:
+					ir_emit_binop_helper_i64(ctx, i, insn);
+					break;
+				case IR_BIT_COUNT_HELPER_I64:
+					ir_emit_bit_count_helper_i64(ctx, i, insn);
+					break;
+				case IR_OP_I64:
+					ir_emit_op_i64(ctx, i, insn);
+					break;
+				case IR_SEXT_I64:
+					ir_emit_sext_i64(ctx, i, insn);
+					break;
+				case IR_ZEXT_I64:
+					ir_emit_zext_i64(ctx, i, insn);
+					break;
+				case IR_SHIFT_I64:
+					ir_emit_shift_i64(ctx, i, insn);
+					break;
+				case IR_SHIFT_CONST_I64:
+					ir_emit_shift_const_i64(ctx, i, insn);
+					break;
+				case IR_BITCAST_I64:
+					ir_emit_bitcast_i64(ctx, i, insn);
+					break;
+				case IR_INT2FP_I64:
+					ir_emit_int2fp_i64(ctx, i, insn);
+					break;
+				case IR_FP2INT_I64:
+					ir_emit_fp2int_i64(ctx, i, insn);
+					break;
+				case IR_BIT_COUNT_I64:
+					ir_emit_bit_count_i64(ctx, i, insn);
+					break;
+				case IR_MIN_MAX_I64:
+					ir_emit_min_max_i64(ctx, i, insn);
+					break;
+				case IR_COND_I64:
+					ir_emit_cond_i64(ctx, i, insn);
+					break;
+				case IR_COND_I64_CMP_INT:
+					ir_emit_cond_i64_cmp_int(ctx, i, insn);
+					break;
+				case IR_COND_I64_CMP_FP:
+					ir_emit_cond_i64_cmp_fp(ctx, i, insn);
+					break;
+				case IR_COND_CMP_I64:
+					ir_emit_cond_cmp_i64(ctx, i, insn);
+					break;
+				case IR_IF_I64:
+					ir_emit_if_i64(ctx, b, i, insn, _ir_next_block(ctx, _b));
+					break;
+				case IR_GUARD_I64:
+					if (ir_emit_guard_i64(ctx, b, i, insn, _ir_next_block(ctx, _b))) {
+						goto next_block;
+					}
+					break;
+				case IR_GUARD_CMP_I64:
+					if (ir_emit_guard_cmp_i64(ctx, b, i, insn, _ir_next_block(ctx, _b))) {
+						goto next_block;
+					}
+					break;
+				case IR_PARAM_I64:
+					break;
+				case IR_RETURN_I64:
+					ir_emit_return_i64(ctx, i, insn);
+					break;
+#endif
 				default:
 					IR_ASSERT(0 && "NIY rule/instruction");
 					ir_mem_free(data.emit_constants);
@@ -12207,8 +26800,85 @@ next_block:;
 			}
 			|.byte 0

+		} else if (IR_IS_TYPE_VECTOR(insn->type)) {
+			int label = ctx->cfg_blocks_count + i;
+			uint32_t size = IR_VECTOR_SIZE(insn->type);
+			uint32_t n = IR_VECTOR_LENGTH(insn->type);
+			ir_type type = IR_VECTOR_BASE_TYPE(insn->type);
+			void *p = ir_long_const_ptr(ctx, -i);
+
+			if (!data.rodata_label) {
+				data.rodata_label = ctx->cfg_blocks_count + ctx->consts_count + 2;
+
+				|.rodata
+				|=>data.rodata_label:
+			}
+			if (size >= 32) {
+				|.align 32
+				required_alignment = 32;
+			} else if (size >= 16) {
+				|.align 16
+			} else if (size == 8) {
+				|.align 8
+			} else if (size == 4) {
+				|.align 4
+			} else if (size == 2) {
+				|.align 2
+			}
+			|=>label:
+			if (ir_type_size[type] == 8) {
+				while (n--) {
+					ir_val val;
+					val.u64 = *(uint64_t*)p;
+					|.dword val.u32, val.u32_hi
+					p = (char*)p + 8;
+				}
+			} else if (ir_type_size[type] == 4) {
+				while (n--) {
+					|.dword *(uint32_t*)p
+					p = (char*)p + 4;
+				}
+			} else if (ir_type_size[type] == 2) {
+				while (n--) {
+					|.word *(uint16_t*)p
+					p = (char*)p + 2;
+				}
+			} else if (ir_type_size[type] == 1) {
+				while (n--) {
+					|.byte *(uint8_t*)p
+					p = (char*)p + 1;
+				}
+			} else {
+				IR_ASSERT(0);
+			}
 		} else {
-			IR_ASSERT(0);
+			IR_ASSERT(IR_IS_TYPE_INT(insn->type));
+			int label = ctx->cfg_blocks_count + i;
+
+			if (!data.rodata_label) {
+				data.rodata_label = ctx->cfg_blocks_count + ctx->consts_count + 2;
+
+				|.rodata
+				|=>data.rodata_label:
+			}
+			if (ir_type_size[insn->type] == 8) {
+				|.align 8
+				|=>label:
+				|.dword insn->val.u32, insn->val.u32_hi
+			} else if (ir_type_size[insn->type] == 4) {
+				|.align 4
+				|=>label:
+				|.dword insn->val.u32
+			} else if (ir_type_size[insn->type] == 2) {
+				|.align 2
+				|=>label:
+				|.word insn->val.u16
+			} else if (ir_type_size[insn->type] == 1) {
+				|=>label:
+				|.byte insn->val.u8
+			} else {
+				IR_ASSERT(0);
+			}
 		}
 	} IR_BITSET_FOREACH_END();
 	if (data.rodata_label) {
@@ -12242,7 +26912,7 @@ next_block:;

 	if (ctx->code_buffer) {
 		entry = ctx->code_buffer->pos;
-		entry = (void*)IR_ALIGNED_SIZE(((size_t)(entry)), 16);
+		entry = (void*)IR_ALIGNED_SIZE(((size_t)(entry)), required_alignment);
 		if (size > (size_t)((char*)ctx->code_buffer->end - (char*)entry)) {
 			ctx->data = NULL;
 			ctx->status = IR_ERROR_CODE_MEM_OVERFLOW;
@@ -12441,6 +27111,7 @@ void *ir_emit_thunk(ir_code_buffer *code_buffer, void *addr, size_t *size_ptr)
 	}

 	if (size > (size_t)((char*)code_buffer->end - (char*)code_buffer->pos)) {
+		*size_ptr = size;
 		dasm_free(&dasm_state);
 		return NULL;
 	}
@@ -12466,8 +27137,10 @@ void ir_fix_thunk(void *thunk_entry, void *addr)
 	unsigned char *code = thunk_entry;

 	if (sizeof(void*) == 8 && !IR_IS_SIGNED_32BIT(((unsigned char*)addr - (code + 5)))) {
-		int32_t *offset_ptr;
-		void **addr_ptr;
+		typedef IR_SET_ALIGNED(1, int32_t unaligned_int32_t);
+		typedef IR_SET_ALIGNED(1, void* unaligned_ptr_t);
+		unaligned_int32_t *offset_ptr;
+		unaligned_ptr_t *addr_ptr;

 		IR_ASSERT(code[0] == 0xff && code[1] == 0x25);
 		offset_ptr = (int32_t*)(code + 2);
@@ -12482,3 +27155,23 @@ void ir_fix_thunk(void *thunk_entry, void *addr)
 		*addr_ptr = (int32_t)(intptr_t)(void*)((unsigned char*)addr - (code + 5));
 	}
 }
+
+#if defined(_MSC_VER) && defined(IR_TARGET_X86)
+/* MSVC doesn't enforce 16-byte stack alignment */
+__declspec(naked) int ir_call_with_aligned_stack(int (*func)(int, const char**), int argc, const char **argv) {
+	__asm {
+		push ebp
+		mov ebp, esp
+		and esp, -16
+		sub esp, 16
+		mov eax, [ebp+16]
+		mov [esp+4], eax
+		mov eax, [ebp+12]
+		mov [esp], eax
+		call [ebp+8]
+		mov esp, ebp
+		pop ebp
+		ret
+	}
+}
+#endif
diff --git a/ext/opcache/jit/ir/ir_x86.h b/ext/opcache/jit/ir/ir_x86.h
index 6399ca107fd..5b495b6983b 100644
--- a/ext/opcache/jit/ir/ir_x86.h
+++ b/ext/opcache/jit/ir/ir_x86.h
@@ -1,7 +1,7 @@
 /*
  * IR - Lightweight JIT Compilation Framework
  * (x86/x86_64 CPU specific definitions)
- * Copyright (C) 2022 Zend by Perforce.
+ * This file is part of the IR Project distributed under the MIT-style LICENSE.
  * Authors: Dmitry Stogov <dmitry@php.net>
  */

@@ -116,4 +116,10 @@ enum _ir_reg {
 #define IR_REG_RSI IR_REG_R6
 #define IR_REG_RDI IR_REG_R7

+#if IR_X86_I64
+# define IR_REG_I64_PAIR(lo, hi) (((lo) & 7) | (((hi) & 7) << 3))
+# define IR_REG_I64_LO(reg)      ((reg) & 7)
+# define IR_REG_I64_HI(reg)      (((reg) >> 3) & 7)
+#endif
+
 #endif /* IR_X86_H */