(inputs, memory, bias, mem_bias, params, state=None,
dtype=None, scope=None)
| 88 | |
| 89 | |
| 90 | def transformer_decoder(inputs, memory, bias, mem_bias, params, state=None, |
| 91 | dtype=None, scope=None): |
| 92 | with tf.variable_scope(scope, default_name="decoder", dtype=dtype, |
| 93 | values=[inputs, memory, bias, mem_bias]): |
| 94 | x = inputs |
| 95 | next_state = {} |
| 96 | for layer in range(params.num_decoder_layers): |
| 97 | layer_name = "layer_%d" % layer |
| 98 | with tf.variable_scope(layer_name): |
| 99 | layer_state = state[layer_name] if state is not None else None |
| 100 | max_relative_dis = params.max_relative_dis \ |
| 101 | if params.position_info_type == 'relative' else None |
| 102 | |
| 103 | with tf.variable_scope("self_attention"): |
| 104 | y = layers.attention.multihead_attention( |
| 105 | _layer_process(x, params.layer_preprocess), |
| 106 | None, |
| 107 | bias, |
| 108 | params.num_heads, |
| 109 | params.attention_key_channels or params.hidden_size, |
| 110 | params.attention_value_channels or params.hidden_size, |
| 111 | params.hidden_size, |
| 112 | 1.0 - params.attention_dropout, |
| 113 | state=layer_state, |
| 114 | max_relative_dis=max_relative_dis, |
| 115 | ) |
| 116 | |
| 117 | if layer_state is not None: |
| 118 | next_state[layer_name] = y["state"] |
| 119 | |
| 120 | y = y["outputs"] |
| 121 | x = _residual_fn(x, y, 1.0 - params.residual_dropout) |
| 122 | x = _layer_process(x, params.layer_postprocess) |
| 123 | |
| 124 | with tf.variable_scope("encdec_attention"): |
| 125 | y = layers.attention.multihead_attention( |
| 126 | _layer_process(x, params.layer_preprocess), |
| 127 | memory, |
| 128 | mem_bias, |
| 129 | params.num_heads, |
| 130 | params.attention_key_channels or params.hidden_size, |
| 131 | params.attention_value_channels or params.hidden_size, |
| 132 | params.hidden_size, |
| 133 | 1.0 - params.attention_dropout, |
| 134 | max_relative_dis=max_relative_dis, |
| 135 | ) |
| 136 | y = y["outputs"] |
| 137 | x = _residual_fn(x, y, 1.0 - params.residual_dropout) |
| 138 | x = _layer_process(x, params.layer_postprocess) |
| 139 | |
| 140 | with tf.variable_scope("feed_forward"): |
| 141 | y = _ffn_layer( |
| 142 | _layer_process(x, params.layer_preprocess), |
| 143 | params.filter_size, |
| 144 | params.hidden_size, |
| 145 | 1.0 - params.relu_dropout, |
| 146 | ) |
| 147 | x = _residual_fn(x, y, 1.0 - params.residual_dropout) |
no test coverage detected