{
  "count": 17,
  "articles": [
    {
      "slug": "flash-tier-shootout-benchmark",
      "title": "Four Flash Models on Practical Prompt Tests",
      "description": "GLM-5.3-Flash led this four-model benchmark at 4.83 and had the lowest completed-response latency, while Qwen3.8-Flash had the lowest recorded cost.",
      "date": "2026-09-12",
      "tags": [
        "Model evaluation",
        "Benchmarks",
        "LLMs",
        "GLM-5.3-Flash",
        "Qwen3.8-Flash",
        "Gemini 3.8-Flash",
        "DeepSeek V4.1-Flash"
      ],
      "url": "https://bshp.io/articles/flash-tier-shootout-benchmark/",
      "apiUrl": "https://bshp.io/api/v1/articles/flash-tier-shootout-benchmark"
    },
    {
      "slug": "how-much-can-you-build-on-an-11-hour-flight",
      "title": "How much can you build on an 11-hour flight?",
      "description": "I reconstructed one LHR to SFO flight from Mission Control, git, and Linear: 5 hours 46 minutes of active work, 27 merged PRs, and several limits the logs cannot explain.",
      "date": "2026-08-27",
      "tags": [
        "AI tooling",
        "Mission Control",
        "Grok",
        "Claude Code",
        "OpenCode",
        "Local inference",
        "Agent workflows"
      ],
      "url": "https://bshp.io/articles/how-much-can-you-build-on-an-11-hour-flight/",
      "apiUrl": "https://bshp.io/api/v1/articles/how-much-can-you-build-on-an-11-hour-flight"
    },
    {
      "slug": "i-finally-hit-the-supergrok-limit",
      "title": "I finally hit the SuperGrok limit",
      "description": "What it took to use a full $30 SuperGrok weekly allowance: 31 coding-agent sessions, 9.66 million input tokens, and a lot of deliberate work.",
      "date": "2026-08-17",
      "tags": [
        "Grok",
        "Mission Control",
        "AI tooling",
        "Observability"
      ],
      "url": "https://bshp.io/articles/i-finally-hit-the-supergrok-limit/",
      "apiUrl": "https://bshp.io/api/v1/articles/i-finally-hit-the-supergrok-limit"
    },
    {
      "slug": "deepseek-v4-flash-vs-pro-benchmark",
      "title": "DeepSeek V4 Flash vs Pro on Practical Prompt Tests",
      "description": "DeepSeek V4 Pro scored 4.50 to Flash's 4.38 in one practical prompt-test batch, but cost about 21 times as much for candidate answers.",
      "date": "2026-08-15",
      "tags": [
        "Model evaluation",
        "Benchmarks",
        "LLMs",
        "DeepSeek",
        "DeepSeek V4 Flash",
        "DeepSeek V4 Pro"
      ],
      "url": "https://bshp.io/articles/deepseek-v4-flash-vs-pro-benchmark/",
      "apiUrl": "https://bshp.io/api/v1/articles/deepseek-v4-flash-vs-pro-benchmark"
    },
    {
      "slug": "grok-imagine",
      "title": "Grok Imagine: a small hands-on image review",
      "description": "Four create and edit calls through Grok Build image tools: what held up for decorative site assets, what this sample does not prove, and when I still reach for code or real screenshots.",
      "date": "2026-08-13",
      "tags": [
        "Grok",
        "Image generation",
        "AI tooling",
        "xAI",
        "Site craft"
      ],
      "url": "https://bshp.io/articles/grok-imagine/",
      "apiUrl": "https://bshp.io/api/v1/articles/grok-imagine"
    },
    {
      "slug": "july-2026-agent-usage",
      "title": "July 2026 agent usage, measured",
      "description": "A July extract from the Fedora Mission Control hub: 402 sessions across hosted agents and local OpenCode with Qwen 3.6 plus Hermes, alongside $14.46 of separate Direct API Spend.",
      "date": "2026-08-13",
      "tags": [
        "Mission Control",
        "Observability",
        "Grok",
        "Claude Code",
        "OpenCode",
        "Codex",
        "Local inference",
        "AI tooling"
      ],
      "url": "https://bshp.io/articles/july-2026-agent-usage/",
      "apiUrl": "https://bshp.io/api/v1/articles/july-2026-agent-usage"
    },
    {
      "slug": "local-gemma-vs-frontier-benchmark",
      "title": "Local Gemma vs Grok 4.5 and Sonnet 5 on Practical Prompt Tests",
      "description": "A model-prompt-tests run that put a local Gemma candidate next to Grok 4.5, Sonnet 5, and GPT-5.5: peer scores, latency reality, and what a high local score does not prove.",
      "date": "2026-08-13",
      "tags": [
        "Model evaluation",
        "Benchmarks",
        "LLMs",
        "Local inference",
        "Grok",
        "Claude"
      ],
      "url": "https://bshp.io/articles/local-gemma-vs-frontier-benchmark/",
      "apiUrl": "https://bshp.io/api/v1/articles/local-gemma-vs-frontier-benchmark"
    },
    {
      "slug": "astro-llms",
      "title": "astro-llms: llms.txt from Astro content collections",
      "description": "A content-collection-first Astro integration for curated llms.txt agent surfaces. Why HTML scraping is the wrong default, how setup works, and what the build emits.",
      "date": "2026-08-04",
      "tags": [
        "Astro",
        "llms.txt",
        "Content Layer",
        "AI tooling",
        "Open source"
      ],
      "url": "https://bshp.io/articles/astro-llms/",
      "apiUrl": "https://bshp.io/api/v1/articles/astro-llms"
    },
    {
      "slug": "mission-control-spend-budgets",
      "title": "Mission Control: provider spend, budgets, and burn rate",
      "description": "Mission Control now pulls account-level billing from provider APIs, surfaces Direct API Spend on the homepage, and adds budgets, burn-rate forecasts, and spend alerts without mixing them into agent session costs.",
      "date": "2026-07-30",
      "tags": [
        "Mission Control",
        "Observability",
        "Cost monitoring",
        "OpenRouter",
        "Anthropic",
        "OpenAI",
        "xAI",
        "AI tooling"
      ],
      "url": "https://bshp.io/articles/mission-control-spend-budgets/",
      "apiUrl": "https://bshp.io/api/v1/articles/mission-control-spend-budgets"
    },
    {
      "slug": "kimi-capacity-constrained",
      "title": "Kimi sold out: capacity-constrained AI demand",
      "description": "Moonshot AI's Kimi subscriptions sold out. That is a capacity signal, not a demand problem: when labs turn away paying users, compute cost is the binding constraint.",
      "date": "2026-07-19",
      "tags": [
        "AI economics",
        "Inference",
        "Frontier models",
        "Local models"
      ],
      "url": "https://bshp.io/articles/kimi-capacity-constrained/",
      "apiUrl": "https://bshp.io/api/v1/articles/kimi-capacity-constrained"
    },
    {
      "slug": "openai-model-naming",
      "title": "OpenAI model naming and independent release lines",
      "description": "Why version-coupled model names break down as product lines multiply, what Anthropic's shift away from shared version labels fixed, and a guess at independent OpenAI lines with their own ladders.",
      "date": "2026-07-19",
      "tags": [
        "OpenAI",
        "Frontier models",
        "Product",
        "LLMs"
      ],
      "url": "https://bshp.io/articles/openai-model-naming/",
      "apiUrl": "https://bshp.io/api/v1/articles/openai-model-naming"
    },
    {
      "slug": "adding-grok-support-to-mission-control",
      "title": "Adding Grok support to Mission Control",
      "description": "How Grok Build CLI became a first-class source in Mission Control: desktop collector, source filter, sessions, activities, and honest status lights.",
      "date": "2026-07-16",
      "tags": [
        "Mission Control",
        "Grok",
        "Observability",
        "Claude Code",
        "Codex",
        "AI tooling"
      ],
      "url": "https://bshp.io/articles/adding-grok-support-to-mission-control/",
      "apiUrl": "https://bshp.io/api/v1/articles/adding-grok-support-to-mission-control"
    },
    {
      "slug": "grok-plugin-cc",
      "title": "Grok Build for Code Review in Claude Code",
      "description": "A look at grok-plugin-cc, a Claude Code plugin that routes review and rescue workflows through xAI's Grok Build CLI, with setup, safety gates, and test results.",
      "date": "2026-07-16",
      "tags": [
        "Claude Code",
        "Grok",
        "xAI",
        "Code review"
      ],
      "url": "https://bshp.io/articles/grok-plugin-cc/",
      "apiUrl": "https://bshp.io/api/v1/articles/grok-plugin-cc"
    },
    {
      "slug": "mission-control",
      "title": "Mission Control: unified AI usage after the OpenClaw pivot",
      "description": "What Mission Control is today - a multi-source usage and runtime dashboard - and how it was rebuilt from an OpenClaw-only activity feed into collectors for Claude Code, Codex, Hermes, ComfyUI, and more.",
      "date": "2026-07-15",
      "tags": [
        "Mission Control",
        "Observability",
        "AI tooling",
        "Claude Code",
        "Codex",
        "Local inference",
        "Systems"
      ],
      "url": "https://bshp.io/articles/mission-control/",
      "apiUrl": "https://bshp.io/api/v1/articles/mission-control"
    },
    {
      "slug": "grok-45-vs-sonnet-5-benchmark",
      "title": "Grok 4.5 vs Sonnet 5 on Practical Prompt Tests",
      "description": "A benchmark writeup comparing Grok 4.5 and Sonnet 5 across 13 practical prompt tests, with methodology, caveats, and raw artifacts preserved in model-prompt-tests.",
      "date": "2026-07-14",
      "tags": [
        "Model evaluation",
        "Benchmarks",
        "LLMs",
        "Grok",
        "Claude"
      ],
      "url": "https://bshp.io/articles/grok-45-vs-sonnet-5-benchmark/",
      "apiUrl": "https://bshp.io/api/v1/articles/grok-45-vs-sonnet-5-benchmark"
    },
    {
      "slug": "local-model-plugin-cc",
      "title": "Local Models for Code Review in Claude Code",
      "description": "A look at local-model-plugin-cc, a Claude Code plugin that routes review and rescue workflows through the Codex CLI and local OpenAI-compatible model servers.",
      "date": "2026-07-13",
      "tags": [
        "Claude Code",
        "Codex",
        "Local models",
        "Code review"
      ],
      "url": "https://bshp.io/articles/local-model-plugin-cc/",
      "apiUrl": "https://bshp.io/api/v1/articles/local-model-plugin-cc"
    },
    {
      "slug": "model-prompt-tests-benchmarking",
      "title": "Benchmarking Model Prompt Tests",
      "description": "Introducing the first benchmark runner for model-prompt-tests: a small harness for running prompt suites across model providers, scoring outputs against rubrics, and publishing comparable reports.",
      "date": "2026-07-13",
      "tags": [
        "Model evaluation",
        "Benchmarks",
        "LLMs",
        "Tooling"
      ],
      "url": "https://bshp.io/articles/model-prompt-tests-benchmarking/",
      "apiUrl": "https://bshp.io/api/v1/articles/model-prompt-tests-benchmarking"
    }
  ]
}