getThreadIdToNodeMapping() build once-per-process vector that maps every logical processor ID (0 … onlineCPUCount-1, enumerated in [group, number] order) to the NUMA-node the CPU belongs to. 1. Query each online logical processor with GetNumaProcessorNodeEx() to learn its NUMA node. Optionally split nodes that span several processor groups. 2. Re-order the CPUs *round-robin* by node, so the firs
| 83 | /// NUMA nodes evenly instead of exhausting node-0 first. |
| 84 | /// 3. Return a vector<int> threads → node-id. |
| 85 | std::vector<int> getThreadIdToNodeMapping() |
| 86 | { |
| 87 | HMODULE k32 = GetModuleHandle("Kernel32.dll"); |
| 88 | if (!k32) |
| 89 | return {}; |
| 90 | |
| 91 | auto gnpne = reinterpret_cast<fun1_t>(GetProcAddress(k32, "GetNumaProcessorNodeEx")); |
| 92 | if (!gnpne) |
| 93 | return {}; |
| 94 | |
| 95 | // enumerate CPUs |
| 96 | const WORD groupCnt = GetActiveProcessorGroupCount(); |
| 97 | size_t totalLps = 0; |
| 98 | for (WORD g = 0; g < groupCnt; ++g) |
| 99 | totalLps += GetActiveProcessorCount(g); |
| 100 | |
| 101 | // buckets[node-id] -> list of CPUs that belong to that (possibly split) node |
| 102 | std::map<int, std::vector<int>> buckets; // ordered by node-id |
| 103 | std::map<std::pair<USHORT, WORD>, int> splitId; // (node,group) -> split-id |
| 104 | int nextSplitId = 0; |
| 105 | |
| 106 | int cpuIndex = 0; // global, monotonically increasing |
| 107 | for (WORD g = 0; g < groupCnt; ++g) { |
| 108 | const DWORD lpInGroup = GetActiveProcessorCount(g); |
| 109 | for (DWORD p = 0; p < lpInGroup; ++p, ++cpuIndex) { |
| 110 | PROCESSOR_NUMBER pn {g, static_cast<BYTE>(p), 0}; |
| 111 | USHORT node = USHRT_MAX; |
| 112 | if (!gnpne(&pn, &node) || node == USHRT_MAX) |
| 113 | continue; // skip offline / unknown |
| 114 | |
| 115 | // split physical node by processor-group to avoid scheduler bias |
| 116 | auto key = std::make_pair(node, g); |
| 117 | auto it = splitId.find(key); |
| 118 | if (it == splitId.end()) |
| 119 | it = splitId.emplace(key, nextSplitId++).first; |
| 120 | |
| 121 | const int splitNodeId = it->second; |
| 122 | buckets[splitNodeId].push_back(cpuIndex); |
| 123 | } |
| 124 | } |
| 125 | |
| 126 | if (buckets.empty()) |
| 127 | return {}; // nothing usable |
| 128 | |
| 129 | // build round-robin order |
| 130 | std::vector<int> mapping; |
| 131 | mapping.reserve(totalLps); |
| 132 | |
| 133 | for (bool still = true; still;) { |
| 134 | still = false; |
| 135 | for (auto &[nodeId, cpus] : buckets) |
| 136 | if (!cpus.empty()) { |
| 137 | mapping.push_back(nodeId); // take one CPU of this node |
| 138 | cpus.pop_back(); // remove it |
| 139 | still = true; // at least one bucket not empty |
| 140 | } |
| 141 | } |
| 142 |