Last active
September 15, 2026 14:57
-
-
Save hartsock/23ace45ba11e84c67d8902ef446dae70 to your computer and use it in GitHub Desktop.
codex-dgx1 — route Codex CLI inference to a local llama.cpp DGX endpoint
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env bash | |
| # | |
| # codex-dgx1 — run Codex CLI but route inference to our local NVIDIA DGX Spark | |
| # (dgx1.home.lan) using the model currently loaded there, instead | |
| # of OpenAI. I use lama.cpp on my NVIDIA DGX Spark. | |
| # | |
| # This is a thin wrapper around the `codex` binary. At startup it asks the | |
| # llama.cpp model manager on dgx1 which model is currently `loaded`, then | |
| # re-points Codex's "model provider" at that endpoint and forces the model | |
| # name — without touching your real ~/.codex/config.toml. | |
| # | |
| # Shared/sanitized copy: personal paths and host-specific defaults have been | |
| # replaced with portable, env-overridable values. See "Configuration" below. | |
| set -euo pipefail | |
| # --- Configuration (override via environment) --------------------------------- | |
| # Location of the codex binary. We try, in order: | |
| # 1. $CODEX_BIN if set | |
| # 2. any `codex` found on $PATH (so Homebrew, pipx, cargo, etc. all work) | |
| # 3. common install locations as a fallback | |
| detect_codex_bin() { | |
| if [[ -n "${CODEX_BIN:-}" && -x "${CODEX_BIN}" ]]; then | |
| printf '%s' "$CODEX_BIN" | |
| return 0 | |
| fi | |
| # Prefer an executable actually on PATH. | |
| if command -v codex >/dev/null 2>&1; then | |
| # Resolve to an absolute path when possible so exec is robust. | |
| local resolved | |
| if resolved="$(command -v codex 2>/dev/null)" && [[ -x "$resolved" ]]; then | |
| printf '%s' "$resolved" | |
| return 0 | |
| fi | |
| printf '%s' "codex" | |
| return 0 | |
| fi | |
| # Fallback to well-known install locations. | |
| local cand | |
| for cand in \ | |
| /opt/homebrew/bin/codex \ | |
| /usr/local/bin/codex \ | |
| "$HOME/.local/bin/codex" \ | |
| "$HOME/.cargo/bin/codex" \ | |
| "$HOME/bin/codex"; do | |
| if [[ -x "$cand" ]]; then | |
| printf '%s' "$cand" | |
| return 0 | |
| fi | |
| done | |
| return 1 | |
| } | |
| CODEX_BIN="$(detect_codex_bin)" || CODEX_BIN="" | |
| if [[ -z "$CODEX_BIN" ]]; then | |
| echo "codex-dgx1: could not find a codex binary. Set CODEX_BIN to its path, or put codex on PATH." >&2 | |
| exit 127 | |
| fi | |
| # DGX inference endpoint. Codex appends /models and the responses path to this, | |
| # so it must include the API prefix. Probe found the llama-server model manager | |
| # answering /v1/* on port 8080. Override host/port/base URL as needed. | |
| DGX_HOST="${DGX_HOST:-dgx1.home.lan}" | |
| DGX_PORT="${DGX_PORT:-8080}" | |
| DGX_BASE_URL="${DGX_BASE_URL:-http://${DGX_HOST}:${DGX_PORT}/v1}" | |
| # Network timeout (seconds) for the model probe. | |
| DGX_TIMEOUT="${DGX_TIMEOUT:-10}" | |
| # Auth env var for the provider. Local servers accept anything; this just needs | |
| # to be non-empty. Override DGX_API_KEY if yours needs a specific value. | |
| DGX_API_KEY_ENV="${DGX_API_KEY_ENV:-CODEX_DGX1_API_KEY}" | |
| if [[ -z "${!DGX_API_KEY_ENV:-}" ]]; then | |
| export "$DGX_API_KEY_ENV"="${DGX_API_KEY:-dgx1-local-noauth}" | |
| fi | |
| # --- Detect which model is currently loaded on dgx1 --------------------------- | |
| # | |
| # Honour an explicit MODEL= override as-is. Otherwise, ask the server which | |
| # model is loaded and use that, so this keeps working as models get swapped. | |
| if [[ -n "${MODEL:-}" ]]; then | |
| DETECTED_MODEL="" | |
| echo "codex-dgx1: MODEL override set -> ${MODEL}" >&2 | |
| else | |
| echo "codex-dgx1: probing ${DGX_BASE_URL}/models for the loaded model..." >&2 | |
| MODEL_JSON="$(curl -sS -m "$DGX_TIMEOUT" "${DGX_BASE_URL}/models" 2>/dev/null)" || MODEL_JSON="" | |
| if [[ -z "$MODEL_JSON" ]]; then | |
| echo "codex-dgx1: could not reach ${DGX_HOST} at ${DGX_BASE_URL} (check network/VPN/host). Set MODEL= to bypass." >&2 | |
| exit 1 | |
| fi | |
| # Parse the model manager's JSON and print the id of the first model whose | |
| # status is "loaded" (llama.cpp normally holds one model in memory). Status | |
| # may be either {"value":"loaded"} or a plain string. JSON is passed via an | |
| # env var (not stdin) so the heredoc can supply the script without a clash. | |
| DETECTED_MODEL="$(DGX_MODELS="$MODEL_JSON" python3 - <<'PY' | |
| import json, os | |
| try: | |
| data = json.loads(os.environ.get("DGX_MODELS", "{}")) | |
| except Exception: | |
| raise SystemExit(2) | |
| loaded = [] | |
| for m in data.get("data", []): | |
| st = m.get("status") or {} | |
| val = st.get("value") if isinstance(st, dict) else st | |
| if val == "loaded": | |
| loaded.append(m.get("id")) | |
| if not loaded: | |
| raise SystemExit(3) # nothing loaded | |
| print(loaded[0]) # use the first loaded model | |
| PY | |
| )" || { | |
| case $? in | |
| 2) echo "codex-dgx1: could not parse model list from ${DGX_HOST}. Set MODEL= to bypass." >&2; exit 1 ;; | |
| 3) echo "codex-dgx1: no model is currently loaded on ${DGX_HOST}. Load one (or set MODEL= to bypass)." >&2; exit 1 ;; | |
| *) echo "codex-dgx1: failed to detect loaded model from ${DGX_HOST}. Set MODEL= to bypass." >&2; exit 1 ;; | |
| esac | |
| } | |
| if [[ -z "$DETECTED_MODEL" ]]; then | |
| echo "codex-dgx1: no loaded model found on ${DGX_HOST}. Set MODEL= to bypass." >&2 | |
| exit 1 | |
| fi | |
| echo "codex-dgx1: using model loaded on ${DGX_HOST} -> ${DETECTED_MODEL}" >&2 | |
| fi | |
| MODEL="${DETECTED_MODEL:-$MODEL}" | |
| # --- Sanity checks ------------------------------------------------------------ | |
| if [[ "$CODEX_BIN" != "codex" ]] && ! [[ -x "$CODEX_BIN" ]]; then | |
| echo "codex-dgx1: codex binary not found at: $CODEX_BIN" >&2 | |
| echo "Set CODEX_BIN to the correct path, or add codex to PATH." >&2 | |
| exit 127 | |
| fi | |
| echo "codex-dgx1: codex at ${CODEX_BIN}; routing inference to ${DGX_BASE_URL} (model: ${MODEL})" >&2 | |
| # --- Launch ------------------------------------------------------------------- | |
| exec "$CODEX_BIN" \ | |
| --strict-config \ | |
| -c "model=\"${MODEL}\"" \ | |
| -c "model_provider=\"dgx1\"" \ | |
| -c "model_providers.dgx1 = {name = \"DGX1\", base_url = \"${DGX_BASE_URL}\", env_key = \"${DGX_API_KEY_ENV}\", requires_openai_auth = false, supports_websockets = false}" \ | |
| "$@" |
Author
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
This will allow you to finish (albeit with a VERY VERY small model) a session you started with an officially hosted OpenAI model. I've found that hosting an
gpt-oss-120bis possible on an NVIDIA DGX Spark but I usually fall back toOrnith-1.5-27B-A3B-Coderinstead of the larger more capable model just for speed's sake. I use this to "land" in-flight work. NOT to generate NEW work.