Created
September 24, 2026 18:50
-
-
Save dlsniper/ca8d35fcd2a5d14a2258eaf767197632 to your computer and use it in GitHub Desktop.
Detect CPU core types on Apple M-series laptops
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| // main.go - list Apple Silicon core IDs by tier, then verify that | |
| // QoS classes steer threads onto the expected cores. | |
| // Works on two-tier chips regardless of naming: M4 Pro / M4 Max (P + E) | |
| // and M5 Pro / M5 Max (super + performance). | |
| // | |
| // Run: go run main.go (macOS 11+, Apple Silicon, cgo enabled) | |
| package main | |
| /* | |
| #include <pthread.h> | |
| #include <pthread/qos.h> | |
| static int set_qos(int background) { | |
| return pthread_set_qos_class_self_np( | |
| background ? QOS_CLASS_BACKGROUND : QOS_CLASS_USER_INTERACTIVE, 0); | |
| } | |
| // Returns the logical CPU the calling thread is currently running on. | |
| static long cpu_now(void) { | |
| size_t n; | |
| if (pthread_cpu_number_np(&n) != 0) return -1; | |
| return (long)n; | |
| } | |
| */ | |
| import "C" | |
| import ( | |
| "bufio" | |
| "bytes" | |
| "fmt" | |
| "os" | |
| "os/exec" | |
| "regexp" | |
| "runtime" | |
| "sort" | |
| "strconv" | |
| "strings" | |
| "sync" | |
| "sync/atomic" | |
| "time" | |
| ) | |
| var ( | |
| reNode = regexp.MustCompile(`\+-o (\S+)`) | |
| reCPU = regexp.MustCompile(`^cpu\d+@`) | |
| // Accepts both <"E"> (data) and "E" (string) forms. | |
| reType = regexp.MustCompile(`"cluster-type" = <?"([A-Za-z]+)"`) | |
| // Older macOS prints the ID as little-endian hex data (<04000000>), | |
| // newer releases as a plain decimal (4). | |
| reID = regexp.MustCompile(`"logical-cpu-id" = (?:<([0-9a-f]+)>|(\d+))`) | |
| sink uint64 | |
| ) | |
| // coreTypes returns logical CPU ID -> cluster type ("E", "P", ...), | |
| // read from the cpu nodes in the IODeviceTree. | |
| func coreTypes() (map[int]string, error) { | |
| out, err := exec.Command("ioreg", "-l", "-p", "IODeviceTree").Output() | |
| if err != nil { | |
| return nil, err | |
| } | |
| types := map[int]string{} | |
| inCPU, t, id := false, "", -1 | |
| flush := func() { | |
| if inCPU && t != "" && id >= 0 { | |
| types[id] = t | |
| } | |
| t, id = "", -1 | |
| } | |
| sc := bufio.NewScanner(bytes.NewReader(out)) | |
| sc.Buffer(make([]byte, 1<<20), 1<<24) | |
| for sc.Scan() { | |
| line := sc.Text() | |
| if m := reNode.FindStringSubmatch(line); m != nil { | |
| flush() | |
| inCPU = reCPU.MatchString(m[1]) | |
| continue | |
| } | |
| if !inCPU { | |
| continue | |
| } | |
| if m := reType.FindStringSubmatch(line); m != nil { | |
| t = m[1] | |
| } | |
| if m := reID.FindStringSubmatch(line); m != nil { | |
| if m[1] != "" { | |
| id = leHex(m[1]) | |
| } else { | |
| id, _ = strconv.Atoi(m[2]) | |
| } | |
| } | |
| } | |
| flush() | |
| if err := sc.Err(); err != nil { | |
| return nil, err | |
| } | |
| if len(types) == 0 { | |
| return nil, fmt.Errorf("no cpu nodes with cluster-type/logical-cpu-id found") | |
| } | |
| return types, nil | |
| } | |
| // leHex decodes ioreg's little-endian hex data, e.g. "0c000000" -> 12. | |
| func leHex(s string) int { | |
| v := 0 | |
| for i := len(s) - 2; i >= 0; i -= 2 { | |
| b, _ := strconv.ParseUint(s[i:i+2], 16, 8) | |
| v = v<<8 | int(b) | |
| } | |
| return v | |
| } | |
| func sysctlInt(key string) (int, error) { | |
| out, err := exec.Command("sysctl", "-n", key).Output() | |
| if err != nil { | |
| return 0, err | |
| } | |
| return strconv.Atoi(strings.TrimSpace(string(out))) | |
| } | |
| // perfLevels returns the kernel's tier names and core counts, fastest first | |
| // (hw.perflevel0 is always the fastest tier). | |
| func perfLevels() (names []string, counts []int) { | |
| n, _ := sysctlInt("hw.nperflevels") | |
| for i := 0; i < n; i++ { | |
| name, _ := exec.Command("sysctl", "-n", fmt.Sprintf("hw.perflevel%d.name", i)).Output() | |
| c, _ := sysctlInt(fmt.Sprintf("hw.perflevel%d.logicalcpu", i)) | |
| names = append(names, strings.TrimSpace(string(name))) | |
| counts = append(counts, c) | |
| } | |
| return | |
| } | |
| type tier struct { | |
| letter string // device-tree cluster-type | |
| name string // kernel perflevel name | |
| level int // 0 = fastest | |
| ids []int // logical CPU IDs | |
| } | |
| // rankTiers matches device-tree clusters to kernel perf levels by core | |
| // count, so ranking never depends on what the letters or names mean. | |
| func rankTiers(types map[int]string) []tier { | |
| groups := map[string][]int{} | |
| for id, t := range types { | |
| groups[t] = append(groups[t], id) | |
| } | |
| names, counts := perfLevels() | |
| used := make([]bool, len(counts)) | |
| var tiers []tier | |
| for _, t := range sortedKeys(groups) { | |
| sort.Ints(groups[t]) | |
| tr := tier{letter: t, name: "unmatched", level: len(counts), ids: groups[t]} | |
| for i, c := range counts { | |
| if !used[i] && c == len(groups[t]) { | |
| used[i], tr.name, tr.level = true, names[i], i | |
| break | |
| } | |
| } | |
| tiers = append(tiers, tr) | |
| } | |
| sort.SliceStable(tiers, func(a, b int) bool { return tiers[a].level < tiers[b].level }) | |
| return tiers | |
| } | |
| // sample runs `workers` spinning threads at the given QoS for d and | |
| // returns a histogram of CPU IDs they were observed running on. | |
| func sample(background bool, workers int, d time.Duration) map[int]int { | |
| bg := C.int(0) | |
| if background { | |
| bg = 1 | |
| } | |
| var mu sync.Mutex | |
| var wg sync.WaitGroup | |
| hist := map[int]int{} | |
| deadline := time.Now().Add(d) | |
| for i := 0; i < workers; i++ { | |
| wg.Add(1) | |
| go func(seed uint64) { | |
| defer wg.Done() | |
| // Never unlocked on purpose: when the goroutine exits, Go discards | |
| // the thread, so the changed QoS can't leak to other goroutines. | |
| runtime.LockOSThread() | |
| if rc := C.set_qos(bg); rc != 0 { | |
| fmt.Fprintf(os.Stderr, "set_qos failed: %d\n", rc) | |
| return | |
| } | |
| local := map[int]int{} | |
| x := seed | |
| for time.Now().Before(deadline) { | |
| for j := 0; j < 200000; j++ { | |
| x = x*6364136223846793005 + 1442695040888963407 | |
| } | |
| local[int(C.cpu_now())]++ | |
| } | |
| atomic.AddUint64(&sink, x) | |
| mu.Lock() | |
| for k, v := range local { | |
| hist[k] += v | |
| } | |
| mu.Unlock() | |
| }(uint64(i + 1)) | |
| } | |
| wg.Wait() | |
| return hist | |
| } | |
| // report prints the share of samples per tier, fastest tier first. | |
| func report(name string, workers int, hist map[int]int, types map[int]string, tiers []tier) { | |
| total, perType, seen := 0, map[string]int{}, []int{} | |
| for cpu, n := range hist { | |
| t := types[cpu] | |
| if t == "" { | |
| t = "?" | |
| } | |
| total += n | |
| perType[t] += n | |
| seen = append(seen, cpu) | |
| } | |
| if total == 0 { | |
| fmt.Printf(" %-26s no samples\n", name) | |
| return | |
| } | |
| sort.Ints(seen) | |
| fmt.Printf(" %-26s (%2d threads)", name, workers) | |
| for _, tr := range tiers { | |
| fmt.Printf(" L%d=%5.1f%%", tr.level, 100*float64(perType[tr.letter])/float64(total)) | |
| } | |
| if n := perType["?"]; n > 0 { | |
| fmt.Printf(" ?=%5.1f%%", 100*float64(n)/float64(total)) | |
| } | |
| fmt.Printf(" cpus seen: %s\n", join(seen)) | |
| } | |
| func sortedKeys[V any](m map[string]V) []string { | |
| keys := make([]string, 0, len(m)) | |
| for k := range m { | |
| keys = append(keys, k) | |
| } | |
| sort.Strings(keys) | |
| return keys | |
| } | |
| func join(ids []int) string { | |
| s := make([]string, len(ids)) | |
| for i, id := range ids { | |
| s[i] = strconv.Itoa(id) | |
| } | |
| return strings.Join(s, ",") | |
| } | |
| func main() { | |
| if runtime.GOOS != "darwin" || runtime.GOARCH != "arm64" { | |
| fmt.Fprintln(os.Stderr, "this program needs macOS on Apple Silicon") | |
| os.Exit(1) | |
| } | |
| types, err := coreTypes() | |
| if err != nil { | |
| fmt.Fprintln(os.Stderr, "reading core types:", err) | |
| os.Exit(1) | |
| } | |
| if len(types) != runtime.NumCPU() { | |
| fmt.Fprintf(os.Stderr, "warning: device tree lists %d cpus, OS reports %d\n", | |
| len(types), runtime.NumCPU()) | |
| } | |
| tiers := rankTiers(types) | |
| brand, _ := exec.Command("sysctl", "-n", "machdep.cpu.brand_string").Output() | |
| fmt.Printf("%s, %d cores\n", strings.TrimSpace(string(brand)), len(types)) | |
| for _, tr := range tiers { | |
| fmt.Printf(" L%d perflevel%d %-12s (cluster-type %s, %2d cores): %s\n", | |
| tr.level, tr.level, tr.name, tr.letter, len(tr.ids), join(tr.ids)) | |
| } | |
| fmt.Println(" (L0 = fastest tier)") | |
| // Background: one thread per core, so staying on the slowest tier proves | |
| // confinement (12 slow cores on M5 Max, 4 on M4 Pro/Max). | |
| // Interactive: one thread per fastest-tier core, to show preference for it. | |
| fmt.Println("\nVerifying QoS steering (2s per test):") | |
| d := 2 * time.Second | |
| nBG, nFG := runtime.NumCPU(), len(tiers[0].ids) | |
| report("QOS_CLASS_BACKGROUND", nBG, sample(true, nBG, d), types, tiers) | |
| report("QOS_CLASS_USER_INTERACTIVE", nFG, sample(false, nFG, d), types, tiers) | |
| } |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment