Skip to content

Instantly share code, notes, and snippets.

@dlsniper
Created September 24, 2026 18:50
Show Gist options
  • Select an option

  • Save dlsniper/ca8d35fcd2a5d14a2258eaf767197632 to your computer and use it in GitHub Desktop.

Select an option

Save dlsniper/ca8d35fcd2a5d14a2258eaf767197632 to your computer and use it in GitHub Desktop.
Detect CPU core types on Apple M-series laptops
// main.go - list Apple Silicon core IDs by tier, then verify that
// QoS classes steer threads onto the expected cores.
// Works on two-tier chips regardless of naming: M4 Pro / M4 Max (P + E)
// and M5 Pro / M5 Max (super + performance).
//
// Run: go run main.go (macOS 11+, Apple Silicon, cgo enabled)
package main
/*
#include <pthread.h>
#include <pthread/qos.h>
static int set_qos(int background) {
return pthread_set_qos_class_self_np(
background ? QOS_CLASS_BACKGROUND : QOS_CLASS_USER_INTERACTIVE, 0);
}
// Returns the logical CPU the calling thread is currently running on.
static long cpu_now(void) {
size_t n;
if (pthread_cpu_number_np(&n) != 0) return -1;
return (long)n;
}
*/
import "C"
import (
"bufio"
"bytes"
"fmt"
"os"
"os/exec"
"regexp"
"runtime"
"sort"
"strconv"
"strings"
"sync"
"sync/atomic"
"time"
)
var (
reNode = regexp.MustCompile(`\+-o (\S+)`)
reCPU = regexp.MustCompile(`^cpu\d+@`)
// Accepts both <"E"> (data) and "E" (string) forms.
reType = regexp.MustCompile(`"cluster-type" = <?"([A-Za-z]+)"`)
// Older macOS prints the ID as little-endian hex data (<04000000>),
// newer releases as a plain decimal (4).
reID = regexp.MustCompile(`"logical-cpu-id" = (?:<([0-9a-f]+)>|(\d+))`)
sink uint64
)
// coreTypes returns logical CPU ID -> cluster type ("E", "P", ...),
// read from the cpu nodes in the IODeviceTree.
func coreTypes() (map[int]string, error) {
out, err := exec.Command("ioreg", "-l", "-p", "IODeviceTree").Output()
if err != nil {
return nil, err
}
types := map[int]string{}
inCPU, t, id := false, "", -1
flush := func() {
if inCPU && t != "" && id >= 0 {
types[id] = t
}
t, id = "", -1
}
sc := bufio.NewScanner(bytes.NewReader(out))
sc.Buffer(make([]byte, 1<<20), 1<<24)
for sc.Scan() {
line := sc.Text()
if m := reNode.FindStringSubmatch(line); m != nil {
flush()
inCPU = reCPU.MatchString(m[1])
continue
}
if !inCPU {
continue
}
if m := reType.FindStringSubmatch(line); m != nil {
t = m[1]
}
if m := reID.FindStringSubmatch(line); m != nil {
if m[1] != "" {
id = leHex(m[1])
} else {
id, _ = strconv.Atoi(m[2])
}
}
}
flush()
if err := sc.Err(); err != nil {
return nil, err
}
if len(types) == 0 {
return nil, fmt.Errorf("no cpu nodes with cluster-type/logical-cpu-id found")
}
return types, nil
}
// leHex decodes ioreg's little-endian hex data, e.g. "0c000000" -> 12.
func leHex(s string) int {
v := 0
for i := len(s) - 2; i >= 0; i -= 2 {
b, _ := strconv.ParseUint(s[i:i+2], 16, 8)
v = v<<8 | int(b)
}
return v
}
func sysctlInt(key string) (int, error) {
out, err := exec.Command("sysctl", "-n", key).Output()
if err != nil {
return 0, err
}
return strconv.Atoi(strings.TrimSpace(string(out)))
}
// perfLevels returns the kernel's tier names and core counts, fastest first
// (hw.perflevel0 is always the fastest tier).
func perfLevels() (names []string, counts []int) {
n, _ := sysctlInt("hw.nperflevels")
for i := 0; i < n; i++ {
name, _ := exec.Command("sysctl", "-n", fmt.Sprintf("hw.perflevel%d.name", i)).Output()
c, _ := sysctlInt(fmt.Sprintf("hw.perflevel%d.logicalcpu", i))
names = append(names, strings.TrimSpace(string(name)))
counts = append(counts, c)
}
return
}
type tier struct {
letter string // device-tree cluster-type
name string // kernel perflevel name
level int // 0 = fastest
ids []int // logical CPU IDs
}
// rankTiers matches device-tree clusters to kernel perf levels by core
// count, so ranking never depends on what the letters or names mean.
func rankTiers(types map[int]string) []tier {
groups := map[string][]int{}
for id, t := range types {
groups[t] = append(groups[t], id)
}
names, counts := perfLevels()
used := make([]bool, len(counts))
var tiers []tier
for _, t := range sortedKeys(groups) {
sort.Ints(groups[t])
tr := tier{letter: t, name: "unmatched", level: len(counts), ids: groups[t]}
for i, c := range counts {
if !used[i] && c == len(groups[t]) {
used[i], tr.name, tr.level = true, names[i], i
break
}
}
tiers = append(tiers, tr)
}
sort.SliceStable(tiers, func(a, b int) bool { return tiers[a].level < tiers[b].level })
return tiers
}
// sample runs `workers` spinning threads at the given QoS for d and
// returns a histogram of CPU IDs they were observed running on.
func sample(background bool, workers int, d time.Duration) map[int]int {
bg := C.int(0)
if background {
bg = 1
}
var mu sync.Mutex
var wg sync.WaitGroup
hist := map[int]int{}
deadline := time.Now().Add(d)
for i := 0; i < workers; i++ {
wg.Add(1)
go func(seed uint64) {
defer wg.Done()
// Never unlocked on purpose: when the goroutine exits, Go discards
// the thread, so the changed QoS can't leak to other goroutines.
runtime.LockOSThread()
if rc := C.set_qos(bg); rc != 0 {
fmt.Fprintf(os.Stderr, "set_qos failed: %d\n", rc)
return
}
local := map[int]int{}
x := seed
for time.Now().Before(deadline) {
for j := 0; j < 200000; j++ {
x = x*6364136223846793005 + 1442695040888963407
}
local[int(C.cpu_now())]++
}
atomic.AddUint64(&sink, x)
mu.Lock()
for k, v := range local {
hist[k] += v
}
mu.Unlock()
}(uint64(i + 1))
}
wg.Wait()
return hist
}
// report prints the share of samples per tier, fastest tier first.
func report(name string, workers int, hist map[int]int, types map[int]string, tiers []tier) {
total, perType, seen := 0, map[string]int{}, []int{}
for cpu, n := range hist {
t := types[cpu]
if t == "" {
t = "?"
}
total += n
perType[t] += n
seen = append(seen, cpu)
}
if total == 0 {
fmt.Printf(" %-26s no samples\n", name)
return
}
sort.Ints(seen)
fmt.Printf(" %-26s (%2d threads)", name, workers)
for _, tr := range tiers {
fmt.Printf(" L%d=%5.1f%%", tr.level, 100*float64(perType[tr.letter])/float64(total))
}
if n := perType["?"]; n > 0 {
fmt.Printf(" ?=%5.1f%%", 100*float64(n)/float64(total))
}
fmt.Printf(" cpus seen: %s\n", join(seen))
}
func sortedKeys[V any](m map[string]V) []string {
keys := make([]string, 0, len(m))
for k := range m {
keys = append(keys, k)
}
sort.Strings(keys)
return keys
}
func join(ids []int) string {
s := make([]string, len(ids))
for i, id := range ids {
s[i] = strconv.Itoa(id)
}
return strings.Join(s, ",")
}
func main() {
if runtime.GOOS != "darwin" || runtime.GOARCH != "arm64" {
fmt.Fprintln(os.Stderr, "this program needs macOS on Apple Silicon")
os.Exit(1)
}
types, err := coreTypes()
if err != nil {
fmt.Fprintln(os.Stderr, "reading core types:", err)
os.Exit(1)
}
if len(types) != runtime.NumCPU() {
fmt.Fprintf(os.Stderr, "warning: device tree lists %d cpus, OS reports %d\n",
len(types), runtime.NumCPU())
}
tiers := rankTiers(types)
brand, _ := exec.Command("sysctl", "-n", "machdep.cpu.brand_string").Output()
fmt.Printf("%s, %d cores\n", strings.TrimSpace(string(brand)), len(types))
for _, tr := range tiers {
fmt.Printf(" L%d perflevel%d %-12s (cluster-type %s, %2d cores): %s\n",
tr.level, tr.level, tr.name, tr.letter, len(tr.ids), join(tr.ids))
}
fmt.Println(" (L0 = fastest tier)")
// Background: one thread per core, so staying on the slowest tier proves
// confinement (12 slow cores on M5 Max, 4 on M4 Pro/Max).
// Interactive: one thread per fastest-tier core, to show preference for it.
fmt.Println("\nVerifying QoS steering (2s per test):")
d := 2 * time.Second
nBG, nFG := runtime.NumCPU(), len(tiers[0].ids)
report("QOS_CLASS_BACKGROUND", nBG, sample(true, nBG, d), types, tiers)
report("QOS_CLASS_USER_INTERACTIVE", nFG, sample(false, nFG, d), types, tiers)
}
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment