| 404 | } |
| 405 | |
| 406 | vector<float> |
| 407 | PCfrSolver::actionUtility(int player, shared_ptr<ActionNode> node, const vector<float> &reach_probs, int iter, |
| 408 | uint64_t current_board,int deal) { |
| 409 | int oppo = 1 - player; |
| 410 | const vector<PrivateCards>& node_player_private_cards = this->ranges[node->getPlayer()]; |
| 411 | |
| 412 | vector<float> payoffs = vector<float>(this->ranges[player].size()); |
| 413 | fill(payoffs.begin(),payoffs.end(),0); |
| 414 | vector<shared_ptr<GameTreeNode>>& children = node->getChildrens(); |
| 415 | vector<GameActions>& actions = node->getActions(); |
| 416 | |
| 417 | shared_ptr<Trainable> trainable; |
| 418 | |
| 419 | /* |
| 420 | if(iter <= this->warmup){ |
| 421 | vector<int> deals = this->getAllAbstractionDeal(deal); |
| 422 | trainable = node->getTrainable(deals[0]); |
| 423 | }else{ |
| 424 | trainable = node->getTrainable(deal); |
| 425 | } |
| 426 | */ |
| 427 | trainable = node->getTrainable(deal); |
| 428 | |
| 429 | #ifdef DEBUG |
| 430 | if(trainable == nullptr){ |
| 431 | throw runtime_error("null trainable"); |
| 432 | } |
| 433 | #endif |
| 434 | |
| 435 | const vector<float> current_strategy = trainable->getcurrentStrategy(); |
| 436 | #ifdef DEBUG |
| 437 | if (current_strategy.size() != actions.size() * node_player_private_cards.size()) { |
| 438 | node->printHistory(); |
| 439 | throw runtime_error(tfm::format( |
| 440 | "length not match %s - %s \n action size %s private_card size %s" |
| 441 | ,current_strategy.size() |
| 442 | ,actions.size() * node_player_private_cards.size() |
| 443 | ,actions.size() |
| 444 | ,node_player_private_cards.size() |
| 445 | )); |
| 446 | } |
| 447 | #endif |
| 448 | |
| 449 | //为了节省计算成本将action regret 存在一位数组而不是二维数组中,两个纬度分别是(该infoset有多少动作,该palyer有多少holecard) |
| 450 | vector<float> regrets(actions.size() * node_player_private_cards.size()); |
| 451 | |
| 452 | vector<vector<float>> all_action_utility(actions.size()); |
| 453 | int node_player = node->getPlayer(); |
| 454 | |
| 455 | vector<vector<float>> results(actions.size()); |
| 456 | for (int action_id = 0; action_id < actions.size(); action_id++) { |
| 457 | |
| 458 | if (node_player != player) { |
| 459 | vector<float> new_reach_prob = vector<float>(reach_probs.size()); |
| 460 | for (int hand_id = 0; hand_id < new_reach_prob.size(); hand_id++) { |
| 461 | float strategy_prob = current_strategy[hand_id + action_id * node_player_private_cards.size()]; |
| 462 | new_reach_prob[hand_id] = reach_probs[hand_id] * strategy_prob; |
| 463 | } |
nothing calls this directly
no test coverage detected