summaryrefslogtreecommitdiff
path: root/src/shaders
diff options
context:
space:
mode:
authorZhao Yakui <yakui.zhao@intel.com>2012-12-24 15:08:39 +0800
committerXiang, Haihao <haihao.xiang@intel.com>2013-01-17 13:08:39 +0800
commit788f649bc66b773e3af1e23870431c19c2c1136d (patch)
treeb48c3c76a2fad4ab543a894678d46135a206cefd /src/shaders
parentcfd70d7afcc22e28bfce8805d7de1e5f759fed8b (diff)
Add the MV prediction to optimize VME parameter on Haswell
Signed-off-by: Zhao Yakui <yakui.zhao@intel.com>
Diffstat (limited to 'src/shaders')
-rw-r--r--src/shaders/vme/inter_frame_haswell.asm338
-rw-r--r--src/shaders/vme/inter_frame_haswell.g75b150
-rw-r--r--src/shaders/vme/vme75.inc44
3 files changed, 532 insertions, 0 deletions
diff --git a/src/shaders/vme/inter_frame_haswell.asm b/src/shaders/vme/inter_frame_haswell.asm
index 5bb8ba7..36b394a 100644
--- a/src/shaders/vme/inter_frame_haswell.asm
+++ b/src/shaders/vme/inter_frame_haswell.asm
@@ -15,6 +15,9 @@
// Now, begin source code....
//
+#define SAVE_RET add (1) RETURN_REG<1>:ud ip:ud 32:ud
+#define RETURN mov (1) ip:ud RETURN_REG<0,1,0>:ud
+
/*
* __START
*/
@@ -73,6 +76,294 @@ mov (1) read1_header.8<1>:UD BLOCK_8X4 {align1};
mov (8) msg_reg0.0<1>:UD read1_header.0<8,8,1>:UD {align1};
send (8) msg_ind CHROMA_COL<1>:UB null read(BIND_IDX_CBCR, 0, 0, 4) mlen 1 rlen 1 {align1};
+mov (8) mb_mvp_ref.0<1>:ud 0:ud {align1};
+and.z.f0.0 (1) null:uw mb_hwdep<0,1,0>:uw 0x04:uw {align1};
+(f0.0) jmpi (1) __mb_hwdep_end;
+/* read back the data for MB A */
+/* the layout of MB result is: rx.0(Available). rx.4(MVa), rX.8(MVb), rX.16(Pred_L0 flag),
+* rX.18 (Pred_L1 flag), rX.20(Forward reference ID), rX.22(Backwared reference ID)
+*/
+mov (8) mba_result.0<1>:ud 0x0:ud {align1};
+mov (8) mbb_result.0<1>:ud 0x0:ud {align1};
+mov (8) mbc_result.0<1>:ud 0x0:ud {align1};
+mba_start:
+mov (8) mb_msg0.0<1>:ud 0:ud {align1};
+and.z.f0.0 (1) null:uw input_mb_intra_ub<0,1,0>:ub INTRA_PRED_AVAIL_FLAG_AE:uw {align1};
+/* MB A doesn't exist. Zero MV. mba_flag is zero and ref ID = -1 */
+(f0.0) mov (2) mba_result.20<1>:w -1:w {align1};
+(f0.0) jmpi (1) mbb_start;
+mov (1) mba_result.0<1>:d MB_AVAIL {align1};
+mov (2) tmp_reg0.0<1>:UW orig_xy_ub<2,2,1>:UB {align1};
+add (1) tmp_reg0.0<1>:w tmp_reg0.0<0,1,0>:w -1:w {align1};
+mul (1) mb_msg0.8<1>:UD w_in_mb_uw<0,1,0>:UW tmp_reg0.2<0,1,0>:UW {align1};
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD tmp_reg0.0<0,1,0>:uw {align1};
+mul (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD 24:UD {align1};
+mov (1) mb_msg0.20<1>:UB thread_id_ub {align1}; /* dispatch id */
+
+/* bind index 3, read 4 oword (64bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_wb.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_4,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 2
+ {align1};
+
+/* TODO: RefID is required after multi-references are added */
+cmp.l.f0.0 (1) null:w mb_intra_wb.16<0,1,0>:uw mb_inter_wb.8<0,1,0>:uw {align1};
+(f0.0) mov (2) mba_result.20<1>:w -1:w {align1};
+(f0.0) jmpi (1) mbb_start;
+
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:ud 3:ud {align1};
+/* Read MV for MB A */
+/* bind index 3, read 8 oword (128bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_mv0.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_8,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 4
+ {align1};
+/* TODO: RefID is required after multi-references are added */
+/* MV */
+mov (2) mba_result.4<1>:ud mb_mv1.8<2,2,1>:ud {align1};
+mov (1) mba_result.16<1>:w MB_PRED_FLAG {align1};
+
+mbb_start:
+mov (8) mb_msg0.0<1>:ud 0:ud {align1};
+and.z.f0.0 (1) null:uw input_mb_intra_ub<0,1,0>:ub INTRA_PRED_AVAIL_FLAG_B:uw {align1};
+/* MB B doesn't exist. Zero MV. mba_flag is zero */
+/* If MB B doesn't exist, neight of MB C nor D exists */
+(f0.0) mov (2) mbb_result.20<1>:w -1:w {align1};
+(f0.0) mov (2) mbc_result.20<1>:w -1:w {align1};
+(f0.0) jmpi (1) mb_mvp_start;
+mov (1) mbb_result.0<1>:d MB_AVAIL {align1};
+mov (2) tmp_reg0.0<1>:UW orig_xy_ub<2,2,1>:UB {align1};
+add (1) tmp_reg0.2<1>:w tmp_reg0.2<0,1,0>:w -1:w {align1};
+mul (1) mb_msg0.8<1>:UD w_in_mb_uw<0,1,0>:UW tmp_reg0.2<0,1,0>:UW {align1};
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD tmp_reg0.0<0,1,0>:uw {align1};
+mul (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD 24:UD {align1};
+mov (1) mb_msg0.20<1>:UB thread_id_ub {align1}; /* dispatch id */
+
+/* bind index 3, read 4 oword (64bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_wb.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_4,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 2
+ {align1};
+
+/* TODO: RefID is required after multi-references are added */
+cmp.l.f0.0 (1) null:w mb_intra_wb.16<0,1,0>:uw mb_inter_wb.8<0,1,0>:uw {align1};
+(f0.0) mov (2) mbb_result.20<1>:w -1:w {align1};
+(f0.0) jmpi (1) mbc_start;
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:ud 3:ud {align1};
+/* Read MV for MB B */
+/* bind index 3, read 8 oword (128bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_mv0.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_8,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 4
+ {align1};
+/* TODO: RefID is required after multi-references are added */
+mov (2) mbb_result.4<1>:ud mb_mv2.16<2,2,1>:ud {align1};
+mov (1) mbb_result.16<1>:w MB_PRED_FLAG {align1};
+
+mbc_start:
+mov (8) mb_msg0.0<1>:ud 0:ud {align1};
+and.z.f0.0 (1) null:uw input_mb_intra_ub<0,1,0>:ub INTRA_PRED_AVAIL_FLAG_C:uw {align1};
+/* MB C doesn't exist. Zero MV. mba_flag is zero */
+/* Based on h264 spec the MB D will be replaced if MB C doesn't exist */
+(f0.0) jmpi (1) mbd_start;
+mov (1) mbc_result.0<1>:d MB_AVAIL {align1};
+mov (2) tmp_reg0.0<1>:UW orig_xy_ub<2,2,1>:UB {align1};
+add (1) tmp_reg0.2<1>:w tmp_reg0.2<0,1,0>:w -1:w {align1};
+add (1) tmp_reg0.0<1>:w tmp_reg0.0<0,1,0>:w 1:w {align1};
+mul (1) mb_msg0.8<1>:UD w_in_mb_uw<0,1,0>:UW tmp_reg0.2<0,1,0>:UW {align1};
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD tmp_reg0.0<0,1,0>:uw {align1};
+mul (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD 24:UD {align1};
+mov (1) mb_msg0.20<1>:UB thread_id_ub {align1}; /* dispatch id */
+
+/* bind index 3, read 4 oword (64bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_wb.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_4,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 2
+ {align1};
+
+/* TODO: RefID is required after multi-references are added */
+cmp.l.f0.0 (1) null:w mb_intra_wb.16<0,1,0>:uw mb_inter_wb.8<0,1,0>:uw {align1};
+(f0.0) mov (2) mbc_result.20<1>:w -1:w {align1};
+(f0.0) jmpi (1) mb_mvp_start;
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:ud 3:ud {align1};
+/* Read MV for MB C */
+/* bind index 3, read 8 oword (128bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_mv0.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_8,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 4
+ {align1};
+/* TODO: RefID is required after multi-references are added */
+/* Forward MV */
+mov (2) mbc_result.4<1>:ud mb_mv2.16<2,2,1>:ud {align1};
+mov (1) mbc_result.16<1>:w MB_PRED_FLAG {align1};
+
+jmpi (1) mb_mvp_start;
+mbd_start:
+mov (1) mbc_result.0<1>:d MB_AVAIL {align1};
+mov (2) tmp_reg0.0<1>:UW orig_xy_ub<2,2,1>:UB {align1};
+add (2) tmp_reg0.0<1>:w tmp_reg0.0<2,2,1>:w -1:w {align1};
+mul (1) mb_msg0.8<1>:UD w_in_mb_uw<0,1,0>:UW tmp_reg0.2<0,1,0>:UW {align1};
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD tmp_reg0.0<0,1,0>:uw {align1};
+mul (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:UD 24:UD {align1};
+mov (1) mb_msg0.20<1>:UB thread_id_ub {align1}; /* dispatch id */
+
+/* bind index 3, read 4 oword (64bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_wb.0<1>:ud
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_4,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 2
+ {align1};
+
+cmp.l.f0.0 (1) null:w mb_intra_wb.16<0,1,0>:uw mb_inter_wb.8<0,1,0>:uw {align1};
+(f0.0) mov (2) mbc_result.20<1>:w -1:w {align1};
+(f0.0) jmpi (1) mb_mvp_start;
+
+add (1) mb_msg0.8<1>:UD mb_msg0.8<0,1,0>:ud 3:ud {align1};
+/* Read MV for MB D */
+/* bind index 3, read 8 oword (128bytes), msg type: 0(OWord Block Read) */
+send (16)
+ mb_ind
+ mb_mv0.0<1>:ub
+ NULL
+ data_port(
+ OBR_CACHE_TYPE,
+ OBR_MESSAGE_TYPE,
+ OBR_CONTROL_8,
+ OBR_BIND_IDX,
+ OBR_WRITE_COMMIT_CATEGORY,
+ OBR_HEADER_PRESENT
+ )
+ mlen 1
+ rlen 4
+ {align1};
+
+/* TODO: RefID is required after multi-references are added */
+
+/* Forward MV */
+mov (2) mbc_result.4<1>:ud mb_mv3.24<2,2,1>:ud {align1};
+mov (1) mbc_result.18<1>:w MB_PRED_FLAG {align1};
+
+mb_mvp_start:
+/*TODO: Add the skip prediction */
+/* Check whether both MB and C are invailable */
+add (1) tmp_reg0.0<1>:d mbb_result.0<0,1,0>:d mbc_result.0<0,1,0>:d {align1};
+cmp.z.f0.0 (1) null:d tmp_reg0.0<0,1,0>:d 0:d {align1};
+(-f0.0) jmpi (1) mb_median_start;
+cmp.nz.f0.0 (1) null:d mba_result.0<0,1,0>:d 1:d {align1};
+(f0.0) mov (1) mbb_result.4<1>:ud mba_result.4<0,1,0>:ud {align1};
+(f0.0) mov (1) mbc_result.4<1>:ud mba_result.4<0,1,0>:ud {align1};
+(f0.0) mov (1) mbb_result.20<1>:uw mba_result.20<0,1,0>:uw {align1};
+(f0.0) mov (1) mbc_result.20<1>:uw mba_result.20<0,1,0>:uw {align1};
+(f0.0) mov (1) mb_mvp_ref.0<1>:ud mba_result.4<0,1,0>:ud {align1};
+(-f0.0) mov (1) mb_mvp_ref.0<1>:ud 0:ud {align1};
+jmpi (1) __mb_hwdep_end;
+
+mb_median_start:
+/* check whether only one neighbour MB has the same ref ID with the current MB */
+mov (8) tmp_reg0.0<1>:ud 0:ud {align1};
+cmp.z.f0.0 (1) null:d mba_result.20<1>:w 0:w {align1};
+(f0.0) add (1) tmp_reg0.0<1>:w tmp_reg0.0<1>:w 1:w {align1};
+(f0.0) mov (1) tmp_reg0.4<1>:ud mba_result.4<0,1,0>:ud {align1};
+cmp.z.f0.0 (1) null:d mbb_result.20<1>:w 0:w {align1};
+(f0.0) add (1) tmp_reg0.0<1>:w tmp_reg0.0<1>:w 1:w {align1};
+(f0.0) mov (1) tmp_reg0.4<1>:ud mbb_result.4<0,1,0>:ud {align1};
+cmp.z.f0.0 (1) null:d mbc_result.20<1>:w 0:w {align1};
+(f0.0) add (1) tmp_reg0.0<1>:w tmp_reg0.0<1>:w 1:w {align1};
+(f0.0) mov (1) tmp_reg0.4<1>:ud mbc_result.4<0,1,0>:ud {align1};
+cmp.e.f0.0 (1) null:d tmp_reg0.0<1>:w 1:w {align1};
+(f0.0) mov (1) mb_mvp_ref.0<1>:ud tmp_reg0.4<0,1,0>:ud {align1};
+(f0.0) jmpi (1) __mb_hwdep_end;
+
+mov (1) INPUT_ARG0.0<1>:w mba_result.4<0,1,0>:w {align1};
+mov (1) INPUT_ARG0.4<1>:w mbb_result.4<0,1,0>:w {align1};
+mov (1) INPUT_ARG0.8<1>:w mbc_result.4<0,1,0>:w {align1};
+SAVE_RET {align1};
+ jmpi (1) word_imedian;
+mov (1) mb_mvp_ref.0<1>:w RET_ARG<0,1,0>:w {align1};
+mov (1) INPUT_ARG0.0<1>:w mba_result.6<0,1,0>:w {align1};
+mov (1) INPUT_ARG0.4<1>:w mbb_result.6<0,1,0>:w {align1};
+mov (1) INPUT_ARG0.8<1>:w mbc_result.6<0,1,0>:w {align1};
+SAVE_RET {align1};
+jmpi (1) word_imedian;
+mov (1) mb_mvp_ref.2<1>:w RET_ARG<0,1,0>:w {align1};
+
+__mb_hwdep_end:
/* m2, get the MV/Mb cost passed from constant buffer when
spawning thread by MEDIA_OBJECT */
mov (8) vme_m2<1>:UD r1.0<8,8,1>:UD {align1};
@@ -196,6 +487,9 @@ mov (1) vme_m1.0<1>:UD ADAPTIVE_SEARCH_ENABLE:ud {align1} ;
/* the Max MV number is passed by constant buffer */
mov (1) vme_m1.4<1>:UB r4.28<0,1,0>:UB {align1};
mov (1) vme_m1.8<1>:UD START_CENTER + SEARCH_PATH_LEN:UD {align1};
+/* Set the MV cost center */
+mov (1) vme_m1.16<1>:ud mb_mvp_ref.0<0,1,0>:ud {align1};
+mov (1) vme_m1.20<1>:ud mb_mvp_ref.0<0,1,0>:ud {align1};
mov (8) vme_msg_1.0<1>:UD vme_m1.0<8,8,1>:UD {align1};
mov (8) vme_msg_2<1>:UD vme_m2.0<8,8,1>:UD {align1};
@@ -348,3 +642,47 @@ __EXIT:
*/
mov (8) ts_msg_reg0<1>:UD r0<8,8,1>:UD {align1};
send (16) ts_msg_ind acc0<1>UW null thread_spawner(0, 0, 1) mlen 1 rlen 0 {align1 EOT};
+
+
+ nop ;
+ nop ;
+/* Compare three word data to get the min value */
+word_imin:
+ cmp.le.f0.0 (1) null:w INPUT_ARG0.0<0,1,0>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ (f0.0) mov (1) TEMP_VAR0.0<1>:w INPUT_ARG0.0<0,1,0>:w {align1};
+ (-f0.0) mov (1) TEMP_VAR0.0<1>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ cmp.le.f0.0 (1) null:w TEMP_VAR0.0<0,1,0>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (f0.0) mov (1) RET_ARG<1>:w TEMP_VAR0.0<0,1,0>:w {align1};
+ (-f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ RETURN {align1};
+
+/* Compare three word data to get the max value */
+word_imax:
+ cmp.ge.f0.0 (1) null:w INPUT_ARG0.0<0,1,0>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ (f0.0) mov (1) TEMP_VAR0.0<1>:w INPUT_ARG0.0<0,1,0>:w {align1};
+ (-f0.0) mov (1) TEMP_VAR0.0<1>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ cmp.ge.f0.0 (1) null:w TEMP_VAR0.0<0,1,0>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (f0.0) mov (1) RET_ARG<1>:w TEMP_VAR0.0<0,1,0>:w {align1};
+ (-f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ RETURN {align1};
+
+word_imedian:
+ cmp.ge.f0.0 (1) null:w INPUT_ARG0.0<0,1,0>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ (f0.0) jmpi (1) cmp_a_ge_b;
+ cmp.ge.f0.0 (1) null:w INPUT_ARG0.0<0,1,0>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.0<0,1,0>:w {align1};
+ (f0.0) jmpi (1) cmp_end;
+ cmp.ge.f0.0 (1) null:w INPUT_ARG0.4<0,1,0>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (-f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ jmpi (1) cmp_end;
+cmp_a_ge_b:
+ cmp.ge.f0.0 (1) null:w INPUT_ARG0.4<0,1,0>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.4<0,1,0>:w {align1};
+ (f0.0) jmpi (1) cmp_end;
+ cmp.ge.f0.0 (1) null:w INPUT_ARG0.0<0,1,0>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.8<0,1,0>:w {align1};
+ (-f0.0) mov (1) RET_ARG<1>:w INPUT_ARG0.0<0,1,0>:w {align1};
+cmp_end:
+ RETURN {align1};
+
diff --git a/src/shaders/vme/inter_frame_haswell.g75b b/src/shaders/vme/inter_frame_haswell.g75b
index 5ae0aad..1ef526c 100644
--- a/src/shaders/vme/inter_frame_haswell.g75b
+++ b/src/shaders/vme/inter_frame_haswell.g75b
@@ -33,6 +33,122 @@
{ 0x00000001, 0x242800e1, 0x00000000, 0x00070003 },
{ 0x00600001, 0x28000021, 0x008d0420, 0x00000000 },
{ 0x04600031, 0x26201cb1, 0x00000800, 0x02190006 },
+ { 0x00600001, 0x2ac00061, 0x00000000, 0x00000000 },
+ { 0x01000005, 0x20002d28, 0x000000a6, 0x00040004 },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000710 },
+ { 0x00600001, 0x2ae00061, 0x00000000, 0x00000000 },
+ { 0x00600001, 0x2b000061, 0x00000000, 0x00000000 },
+ { 0x00600001, 0x2b200061, 0x00000000, 0x00000000 },
+ { 0x00600001, 0x2b400061, 0x00000000, 0x00000000 },
+ { 0x01000005, 0x20002e28, 0x000000a5, 0x00600060 },
+ { 0x00210001, 0x2af401ed, 0x00000000, 0xffffffff },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x000000f0 },
+ { 0x00000001, 0x2ae000e5, 0x00000000, 0x00000001 },
+ { 0x00200001, 0x24000229, 0x004500a0, 0x00000000 },
+ { 0x00000040, 0x24003dad, 0x00000400, 0xffffffff },
+ { 0x00000041, 0x2b482521, 0x000000a2, 0x00000402 },
+ { 0x00000040, 0x2b482421, 0x00000b48, 0x00000400 },
+ { 0x00000041, 0x2b480c21, 0x00000b48, 0x00000018 },
+ { 0x00000001, 0x2b540231, 0x00000014, 0x00000000 },
+ { 0x0a800031, 0x2b601ca1, 0x00000b40, 0x02280303 },
+ { 0x05000010, 0x2000252c, 0x00000b70, 0x00000b88 },
+ { 0x00210001, 0x2af401ed, 0x00000000, 0xffffffff },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000040 },
+ { 0x00000040, 0x2b480c21, 0x00000b48, 0x00000003 },
+ { 0x0a800031, 0x2ba01ca1, 0x00000b40, 0x02480403 },
+ { 0x00200001, 0x2ae40021, 0x00450bc8, 0x00000000 },
+ { 0x00000001, 0x2af001ed, 0x00000000, 0x00010001 },
+ { 0x00600001, 0x2b400061, 0x00000000, 0x00000000 },
+ { 0x01000005, 0x20002e28, 0x000000a5, 0x00100010 },
+ { 0x00210001, 0x2b1401ed, 0x00000000, 0xffffffff },
+ { 0x00210001, 0x2b3401ed, 0x00000000, 0xffffffff },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000320 },
+ { 0x00000001, 0x2b0000e5, 0x00000000, 0x00000001 },
+ { 0x00200001, 0x24000229, 0x004500a0, 0x00000000 },
+ { 0x00000040, 0x24023dad, 0x00000402, 0xffffffff },
+ { 0x00000041, 0x2b482521, 0x000000a2, 0x00000402 },
+ { 0x00000040, 0x2b482421, 0x00000b48, 0x00000400 },
+ { 0x00000041, 0x2b480c21, 0x00000b48, 0x00000018 },
+ { 0x00000001, 0x2b540231, 0x00000014, 0x00000000 },
+ { 0x0a800031, 0x2b601ca1, 0x00000b40, 0x02280303 },
+ { 0x05000010, 0x2000252c, 0x00000b70, 0x00000b88 },
+ { 0x00210001, 0x2b1401ed, 0x00000000, 0xffffffff },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000040 },
+ { 0x00000040, 0x2b480c21, 0x00000b48, 0x00000003 },
+ { 0x0a800031, 0x2ba01ca1, 0x00000b40, 0x02480403 },
+ { 0x00200001, 0x2b040021, 0x00450bf0, 0x00000000 },
+ { 0x00000001, 0x2b1001ed, 0x00000000, 0x00010001 },
+ { 0x00600001, 0x2b400061, 0x00000000, 0x00000000 },
+ { 0x01000005, 0x20002e28, 0x000000a5, 0x00080008 },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000110 },
+ { 0x00000001, 0x2b2000e5, 0x00000000, 0x00000001 },
+ { 0x00200001, 0x24000229, 0x004500a0, 0x00000000 },
+ { 0x00000040, 0x24023dad, 0x00000402, 0xffffffff },
+ { 0x00000040, 0x24003dad, 0x00000400, 0x00010001 },
+ { 0x00000041, 0x2b482521, 0x000000a2, 0x00000402 },
+ { 0x00000040, 0x2b482421, 0x00000b48, 0x00000400 },
+ { 0x00000041, 0x2b480c21, 0x00000b48, 0x00000018 },
+ { 0x00000001, 0x2b540231, 0x00000014, 0x00000000 },
+ { 0x0a800031, 0x2b601ca1, 0x00000b40, 0x02280303 },
+ { 0x05000010, 0x2000252c, 0x00000b70, 0x00000b88 },
+ { 0x00210001, 0x2b3401ed, 0x00000000, 0xffffffff },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000140 },
+ { 0x00000040, 0x2b480c21, 0x00000b48, 0x00000003 },
+ { 0x0a800031, 0x2ba01ca1, 0x00000b40, 0x02480403 },
+ { 0x00200001, 0x2b240021, 0x00450bf0, 0x00000000 },
+ { 0x00000001, 0x2b3001ed, 0x00000000, 0x00010001 },
+ { 0x00000020, 0x34001c00, 0x00001400, 0x000000f0 },
+ { 0x00000001, 0x2b2000e5, 0x00000000, 0x00000001 },
+ { 0x00200001, 0x24000229, 0x004500a0, 0x00000000 },
+ { 0x00200040, 0x24003dad, 0x00450400, 0xffffffff },
+ { 0x00000041, 0x2b482521, 0x000000a2, 0x00000402 },
+ { 0x00000040, 0x2b482421, 0x00000b48, 0x00000400 },
+ { 0x00000041, 0x2b480c21, 0x00000b48, 0x00000018 },
+ { 0x00000001, 0x2b540231, 0x00000014, 0x00000000 },
+ { 0x0a800031, 0x2b601ca1, 0x00000b40, 0x02280303 },
+ { 0x05000010, 0x2000252c, 0x00000b70, 0x00000b88 },
+ { 0x00210001, 0x2b3401ed, 0x00000000, 0xffffffff },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000040 },
+ { 0x00000040, 0x2b480c21, 0x00000b48, 0x00000003 },
+ { 0x0a800031, 0x2ba01cb1, 0x00000b40, 0x02480403 },
+ { 0x00200001, 0x2b240021, 0x00450c18, 0x00000000 },
+ { 0x00000001, 0x2b3201ed, 0x00000000, 0x00010001 },
+ { 0x00000040, 0x240014a5, 0x00000b00, 0x00000b20 },
+ { 0x01000010, 0x20001ca4, 0x00000400, 0x00000000 },
+ { 0x00110020, 0x34001c00, 0x00001400, 0x00000080 },
+ { 0x02000010, 0x20001ca4, 0x00000ae0, 0x00000001 },
+ { 0x00010001, 0x2b040021, 0x00000ae4, 0x00000000 },
+ { 0x00010001, 0x2b240021, 0x00000ae4, 0x00000000 },
+ { 0x00010001, 0x2b140129, 0x00000af4, 0x00000000 },
+ { 0x00010001, 0x2b340129, 0x00000af4, 0x00000000 },
+ { 0x00010001, 0x2ac00021, 0x00000ae4, 0x00000000 },
+ { 0x00110001, 0x2ac00061, 0x00000000, 0x00000000 },
+ { 0x00000020, 0x34001c00, 0x00001400, 0x00000190 },
+ { 0x00600001, 0x24000061, 0x00000000, 0x00000000 },
+ { 0x01000010, 0x20003da4, 0x00200af4, 0x00000000 },
+ { 0x00010040, 0x24003dad, 0x00200400, 0x00010001 },
+ { 0x00010001, 0x24040021, 0x00000ae4, 0x00000000 },
+ { 0x01000010, 0x20003da4, 0x00200b14, 0x00000000 },
+ { 0x00010040, 0x24003dad, 0x00200400, 0x00010001 },
+ { 0x00010001, 0x24040021, 0x00000b04, 0x00000000 },
+ { 0x01000010, 0x20003da4, 0x00200b34, 0x00000000 },
+ { 0x00010040, 0x24003dad, 0x00200400, 0x00010001 },
+ { 0x00010001, 0x24040021, 0x00000b24, 0x00000000 },
+ { 0x01000010, 0x20003da4, 0x00200400, 0x00010001 },
+ { 0x00010001, 0x2ac00021, 0x00000404, 0x00000000 },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x000000c0 },
+ { 0x00000001, 0x2fa001ad, 0x00000ae4, 0x00000000 },
+ { 0x00000001, 0x2fa401ad, 0x00000b04, 0x00000000 },
+ { 0x00000001, 0x2fa801ad, 0x00000b24, 0x00000000 },
+ { 0x00000040, 0x2fe00c01, 0x00001400, 0x00000020 },
+ { 0x00000020, 0x34001c00, 0x00001400, 0x000007b0 },
+ { 0x00000001, 0x2ac001ad, 0x00000fe4, 0x00000000 },
+ { 0x00000001, 0x2fa001ad, 0x00000ae6, 0x00000000 },
+ { 0x00000001, 0x2fa401ad, 0x00000b06, 0x00000000 },
+ { 0x00000001, 0x2fa801ad, 0x00000b26, 0x00000000 },
+ { 0x00000040, 0x2fe00c01, 0x00001400, 0x00000020 },
+ { 0x00000020, 0x34001c00, 0x00001400, 0x00000750 },
+ { 0x00000001, 0x2ac201ad, 0x00000fe4, 0x00000000 },
{ 0x00600001, 0x25600021, 0x008d0020, 0x00000000 },
{ 0x00600001, 0x28400021, 0x008d0560, 0x00000000 },
{ 0x00600001, 0x28600061, 0x00000000, 0x00000000 },
@@ -81,6 +197,8 @@
{ 0x00000001, 0x24600061, 0x00000000, 0x00000002 },
{ 0x00000001, 0x24640231, 0x0000009c, 0x00000000 },
{ 0x00000001, 0x24680061, 0x00000000, 0x30003030 },
+ { 0x00000001, 0x24700021, 0x00000ac0, 0x00000000 },
+ { 0x00000001, 0x24740021, 0x00000ac0, 0x00000000 },
{ 0x00600001, 0x28200021, 0x008d0460, 0x00000000 },
{ 0x00600001, 0x28400021, 0x008d0560, 0x00000000 },
{ 0x00000001, 0x28600061, 0x00000000, 0x01010101 },
@@ -131,3 +249,35 @@
{ 0x0a800031, 0x20001cac, 0x00000800, 0x040a0203 },
{ 0x00600001, 0x2e000021, 0x008d0000, 0x00000000 },
{ 0x07800031, 0x24001ca8, 0x00000e00, 0x82000010 },
+ { 0x0000007e, 0x00000000, 0x00000000, 0x00000000 },
+ { 0x0000007e, 0x00000000, 0x00000000, 0x00000000 },
+ { 0x06000010, 0x200035ac, 0x00000fa0, 0x00000fa4 },
+ { 0x00010001, 0x2f6001ad, 0x00000fa0, 0x00000000 },
+ { 0x00110001, 0x2f6001ad, 0x00000fa4, 0x00000000 },
+ { 0x06000010, 0x200035ac, 0x00000f60, 0x00000fa8 },
+ { 0x00010001, 0x2fe401ad, 0x00000f60, 0x00000000 },
+ { 0x00110001, 0x2fe401ad, 0x00000fa8, 0x00000000 },
+ { 0x00000001, 0x34000020, 0x00000fe0, 0x00000000 },
+ { 0x04000010, 0x200035ac, 0x00000fa0, 0x00000fa4 },
+ { 0x00010001, 0x2f6001ad, 0x00000fa0, 0x00000000 },
+ { 0x00110001, 0x2f6001ad, 0x00000fa4, 0x00000000 },
+ { 0x04000010, 0x200035ac, 0x00000f60, 0x00000fa8 },
+ { 0x00010001, 0x2fe401ad, 0x00000f60, 0x00000000 },
+ { 0x00110001, 0x2fe401ad, 0x00000fa8, 0x00000000 },
+ { 0x00000001, 0x34000020, 0x00000fe0, 0x00000000 },
+ { 0x04000010, 0x200035ac, 0x00000fa0, 0x00000fa4 },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000070 },
+ { 0x04000010, 0x200035ac, 0x00000fa0, 0x00000fa8 },
+ { 0x00010001, 0x2fe401ad, 0x00000fa0, 0x00000000 },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x000000a0 },
+ { 0x04000010, 0x200035ac, 0x00000fa4, 0x00000fa8 },
+ { 0x00010001, 0x2fe401ad, 0x00000fa8, 0x00000000 },
+ { 0x00110001, 0x2fe401ad, 0x00000fa4, 0x00000000 },
+ { 0x00000020, 0x34001c00, 0x00001400, 0x00000060 },
+ { 0x04000010, 0x200035ac, 0x00000fa4, 0x00000fa8 },
+ { 0x00010001, 0x2fe401ad, 0x00000fa4, 0x00000000 },
+ { 0x00010020, 0x34001c00, 0x00001400, 0x00000030 },
+ { 0x04000010, 0x200035ac, 0x00000fa0, 0x00000fa8 },
+ { 0x00010001, 0x2fe401ad, 0x00000fa8, 0x00000000 },
+ { 0x00110001, 0x2fe401ad, 0x00000fa0, 0x00000000 },
+ { 0x00000001, 0x34000020, 0x00000fe0, 0x00000000 },
diff --git a/src/shaders/vme/vme75.inc b/src/shaders/vme/vme75.inc
index 35033c0..a65a4ee 100644
--- a/src/shaders/vme/vme75.inc
+++ b/src/shaders/vme/vme75.inc
@@ -265,3 +265,47 @@ define(`BIND_IDX_CBCR', `6')
define(`LUMA_CHROMA_MODE', `0x0')
define(`LUMA_INTRA_MODE', `0x1')
define(`LUMA_INTRA_DISABLE', `0x2')
+
+define(`RETURN_REG', `r127.0')
+define(`RET_ARG', `r127.4')
+
+/* Now at most two registers are used for input parameter */
+define(`INPUT_ARG0', `r125')
+define(`INPUT_ARG1', `r126')
+
+/* Two temporal registers are used in the function */
+define(`TEMP_VAR0', `r123')
+define(`TEMP_VAR1', `r124')
+
+
+define(`OBR_MESSAGE_TYPE', `0')
+define(`OBR_CACHE_TYPE', `10')
+define(`OBR_BIND_IDX', `BIND_IDX_OUTPUT')
+
+define(`OBR_CONTROL_0', `0') /* 1 OWord, low 128 bits */
+define(`OBR_CONTROL_1', `1') /* 1 OWord, high 128 bits */
+define(`OBR_CONTROL_2', `2') /* 2 OWords */
+define(`OBR_CONTROL_4', `3') /* 4 OWords */
+define(`OBR_CONTROL_8', `4') /* 8 OWords */
+define(`OBR_WRITE_COMMIT_CATEGORY', `0') /* category on SNB+ for Data port */
+define(`OBR_HEADER_PRESENT', `1')
+
+define(`mb_hwdep', `r5.6')
+define(`MB_AVAIL', `1:d')
+define(`MB_PRED_FLAG', `1:w')
+
+define(`mb_mvp_ref', `r86')
+define(`mba_result', `r87')
+define(`mbb_result', `r88')
+define(`mbc_result', `r89')
+define(`mb_ind', `90')
+define(`mb_msg0', `r90')
+define(`mb_wb', `r91')
+define(`mb_intra_wb', `r91')
+define(`mb_inter_wb', `r92')
+define(`mb_mv0', `r93')
+define(`mb_mv1', `r94')
+define(`mb_mv2', `r95')
+define(`mb_mv3', `r96')
+define(`mb_ref', `r97')
+