{
  "version": "https://jsonfeed.org/version/1.1",
  "title": "The Pack · VerdictPal",
  "home_page_url": "https://verdictpal.com/pack",
  "feed_url": "https://verdictpal.com/feed.json",
  "description": "Editorial drops when a tool ships, verdicts move, or methodology changes.",
  "language": "en-US",
  "authors": [
    {
      "name": "VerdictPal editorial desk",
      "url": "https://verdictpal.com/about"
    }
  ],
  "items": [
    {
      "id": "2026-09-14-pricing-privacy-refresh",
      "url": "https://verdictpal.com/tools",
      "title": "Live pricing pass: Consensus Pro $20, Gumloop credits, Canva yearly, Mistral Student $5.99",
      "content_text": "Vendor pages re-checked 14 Sep. Consensus Pro is $20/mo or $144/yr. Gumloop dropped the Free 5k and $97 Pro SKUs — Pro starts at $37/20k credits. Canva Pro is US$144/year, Business US$250/year/person. Mistral Education is $5.99. Scholarcy Library is $9.99/mo or $90/year. /tools",
      "date_published": "2026-09-14T00:00:00.000Z",
      "tags": [
        "tools"
      ]
    },
    {
      "id": "2026-09-12-vals-v2-deepseek-flash",
      "url": "https://verdictpal.com/benchmarks/vals-index",
      "title": "Vals Index v2 is the live board — DeepSeek V4.1 Flash is on the atlas",
      "content_text": "Pre-13 Aug Vals rows (Kimi K3 74.7, Sol 73.1) no longer outrank the live v2 board. Fable 5.1 still leads at 68.83, then Opus 5 and GPT-6 Astra. DeepSeek's 10 Sep V4.1 Flash is recorded from the official launch: 552B MoE, deepseek-flash, off-peak $0.15 / $0.60. /benchmarks/vals-index",
      "date_published": "2026-09-12T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-09-10-aa-index-v43",
      "url": "https://verdictpal.com/benchmarks/artificial-analysis-index",
      "title": "AA Intelligence Index v4.3 is on the ledger — Fable 5.1 53.4, Astra 52.8",
      "content_text": "The 10 Sep Artificial Analysis API snapshot now publishes Intelligence Index v4.3. τ³-Banking left the composite for AutomationBench-AA; Terminal-Bench v4.0 is in the index but not yet a separate API key, so /benchmarks/terminal-bench still tracks 2.1. Claude Fable 5.1 leads at 53.4, GPT-6 Astra 52.8. Vals is unchanged (Fable 5.1 68.83%). /benchmarks/artificial-analysis-index",
      "date_published": "2026-09-10T00:00:00.000Z",
      "tags": [
        "benchmarks"
      ]
    },
    {
      "id": "2026-09-05-frontier-september",
      "url": "https://verdictpal.com/models",
      "title": "September 2026 frontier: GPT-6 Astra, Gemini 3.8 Flash, Muse Spark 1.3, Hy4 preview",
      "content_text": "OpenAI's 3 Sep GPT-6 Astra, Google's 2 Sep Gemini 3.8 Flash, and Meta's 2 Sep Muse Spark 1.3 are now on the atlas with live Artificial Analysis Intelligence Index v4.2 scores. Tencent's Hy4 preview is recorded from the official launch. Claude Fable 5.1 still leads the live Vals Index at 68.83%. /models",
      "date_published": "2026-09-05T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-09-01-claude-fable-5-1",
      "url": "https://verdictpal.com/models/claude-fable-5-1",
      "title": "Claude Fable 5.1 is on the model atlas — Vals Index 67.87%",
      "content_text": "Anthropic's 1 Sep Mythos-class refresh of Fable 5. Same $10 / $50 list, cache reads cut to $0.25 per 1M tokens. Vals puts it at 67.87% on the Index, ahead of Opus 5 and Fable 5. AA Intelligence Index is not in the 1 Sep API snapshot yet. Compare it with GPT-5.6 Sol or Fable 5. /models/claude-fable-5-1",
      "date_published": "2026-09-01T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-09-01-omniscience-critpt",
      "url": "https://verdictpal.com/benchmarks",
      "title": "AA-Omniscience and CritPt join the benchmark atlas",
      "content_text": "Two missing slices of AA Intelligence Index v4.1 now have explainers and sourced rows: AA-Omniscience (factual recall vs hallucination — the knowledge primary) and CritPt (research-level physics, still under 40% at the frontier). /benchmarks/aa-omniscience · /benchmarks/critpt",
      "date_published": "2026-09-01T00:00:00.000Z",
      "tags": [
        "benchmarks"
      ]
    },
    {
      "id": "2026-08-27-terminal-bench-science",
      "url": "https://verdictpal.com/benchmarks/terminal-bench-science",
      "title": "Terminal-Bench-Science 0.1 is on the atlas — best row 30%",
      "content_text": "Stanford and Laude's 70-task science-agent suite: real lab workflows, still far from ceiling. Claude Opus 5 + Claude Code leads at 30.0%; GPT-5.6 Sol + Codex is 22.4%; GLM-5.3 is the strongest open-weight row at 8.1%. Not Terminal-Bench 2.1. /benchmarks/terminal-bench-science",
      "date_published": "2026-08-27T00:00:00.000Z",
      "tags": [
        "benchmarks"
      ]
    },
    {
      "id": "2026-08-26-glm-5-3-flash",
      "url": "https://verdictpal.com/models/glm-5-3-flash",
      "title": "GLM-5.3-Flash is on the model atlas — AA index 57.5",
      "content_text": "Z.ai's Flash sibling of GLM-5.3, from the 1 Sep Artificial Analysis snapshot: Intelligence Index 57.5 at $0.15 / $0.50 per 1M tokens. Two points behind GLM-5.3 max at a tenth of the list price. /models/glm-5-3-flash · /compare/models/glm-5-3-vs-glm-5-3-flash",
      "date_published": "2026-08-26T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-08-27-academic-guides",
      "url": "https://verdictpal.com/guides",
      "title": "Three academic guide trios, rebuilt",
      "content_text": "Student researcher, literature reviewer, and peer reviewer — each with a matching stack and playbook. Peer review starts with the venue AI policy. Retired grant, coding, and deck paths redirect to /guides.",
      "date_published": "2026-08-27T00:00:00.000Z",
      "tags": [
        "guide"
      ]
    },
    {
      "id": "2026-08-24-glm-5-3",
      "url": "https://verdictpal.com/models/glm-5-3",
      "title": "GLM-5.3 is on the model atlas — AA index 59.5",
      "content_text": "Z.ai's August coding refresh is on the atlas from the 2026-08-24 Artificial Analysis snapshot: max-effort Intelligence Index 59.5, API list $1.40 / $4.40 per 1M tokens. Same base as GLM-5.2; weights were still unpublished on this check. Ledger rows sync from AA; editorial depth still needs a desk pass. Compare it with Qwen3.8 27B. /models/glm-5-3 · /compare/models/glm-5-3-vs-qwen3-8-27b",
      "date_published": "2026-08-24T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-08-24-qwen3-8-27b",
      "url": "https://verdictpal.com/models/qwen3-8-27b",
      "title": "Qwen3.8 27B is on the model atlas — AA index 52",
      "content_text": "Alibaba's dense open-weight Qwen3.8 sibling is on the atlas from the 2026-08-24 Artificial Analysis snapshot: xhigh Intelligence Index 52, hosted list $0.50 / $3 per 1M tokens, Apache 2.0 on Hugging Face. This is not hosted Qwen3.8 Max. Ledger rows sync from AA; editorial depth still needs a desk pass. Compare it with Qwen3.8 Max. /models/qwen3-8-27b · /compare/models/qwen3-8-27b-vs-qwen3-8-max",
      "date_published": "2026-08-24T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-08-15-qwen3-8-2-4t",
      "url": "https://verdictpal.com/models/qwen3-8-2-4t-a95b",
      "title": "Qwen3.8 2.4T is on the model atlas — AA index 57.7",
      "content_text": "Alibaba's open-weight Qwen-Max-class MoE (2.4T total / 95B active) is on the atlas from the 2026-08-15 Artificial Analysis snapshot: Intelligence Index 57.7, hosted list $2 / $6 per 1M tokens. This is the published checkpoint, not hosted Qwen3.8 Max. Ledger rows sync from AA; editorial depth still needs a desk pass. Compare it with Qwen3.8 Max. /models/qwen3-8-2-4t-a95b · /compare/models/qwen3-8-2-4t-a95b-vs-qwen3-8-max",
      "date_published": "2026-08-15T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-08-13-gemini-3-7-flash",
      "url": "https://verdictpal.com/models/gemini-3-7-flash",
      "title": "Gemini 3.7 Flash is on the model atlas — AA index 56",
      "content_text": "Google's August Flash workhorse is on the atlas from the Artificial Analysis snapshot: high-effort Intelligence Index 56, intro price $0.75 / $3.75 per 1M tokens through 31 Dec 2026. Ledger rows sync from AA; editorial depth still needs a desk pass. Compare it with 3.6 Flash or Claude Sonnet 4.6. /models/gemini-3-7-flash · /compare/models/gemini-3-6-flash-vs-gemini-3-7-flash",
      "date_published": "2026-08-13T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-08-12-grok-4-6",
      "url": "https://verdictpal.com/models/grok-4-6",
      "title": "Grok 4.6 is on the model atlas — AA index 60.9",
      "content_text": "xAI's August flagship is on the atlas from the Artificial Analysis snapshot: high-effort Intelligence Index 60.9, $2 / $6 per 1M tokens. Ledger rows sync from AA; editorial depth still needs a desk pass. Compare it with GPT-5.6 Sol. /models/grok-4-6 · /compare/models/gpt-5-6-sol-vs-grok-4-6",
      "date_published": "2026-08-12T00:00:00.000Z",
      "tags": [
        "models"
      ]
    },
    {
      "id": "2026-08-03-launch-and-atlas-report",
      "url": "https://verdictpal.com/report",
      "title": "VerdictPal is live, and the Atlas Report puts numbers on the whole atlas",
      "content_text": "The launch note is published, plus a new original-research surface: the Atlas Report computes aggregate pricing, privacy, benchmark-saturation and freshness figures from every published tool, and it is free to cite with a link. Highlights this edition — most dossiers do not record any stance on training with your content, paid entry plans cluster around the median, and frontier model input prices spread by more than three orders of magnitude. /report · /launch",
      "date_published": "2026-08-03T00:00:00.000Z",
      "tags": [
        "desk"
      ]
    },
    {
      "id": "2026-07-12-ai-builder-design-test",
      "url": "https://verdictpal.com/benchmarks/ai-builder-design-test",
      "title": "AI Builder Design Test on the Lab bench — four builders, same prompts, very different results",
      "content_text": "VerdictPal Lab's second first-party protocol: four identical multi-turn design prompts through Lovable, v0, Replit, and Bolt. Run 01 recorded a Turn 1 ranking (Lovable → v0 → Replit → Bolt) and the cost-of-arrival in each vendor's native unit — no padded leaderboard, formal scored rows pending a repeatable rubric. /benchmarks/ai-builder-design-test · /benchmarks/lab",
      "date_published": "2026-07-12T00:00:00.000Z",
      "tags": [
        "bench"
      ]
    },
    {
      "id": "2026-07-06-citation-fidelity-protocol-freeze",
      "url": "https://verdictpal.com/benchmarks/citation-fidelity",
      "title": "Citation Fidelity v0.1 protocol frozen on the Lab bench",
      "content_text": "VerdictPal Lab's first first-party benchmark: twelve research questions, dual-reviewer citation grading, frozen runbook on the atlas page — protocol published, desk run still pending, no padded leaderboard. /benchmarks/citation-fidelity · /benchmarks/lab",
      "date_published": "2026-07-06T00:00:00.000Z",
      "tags": [
        "bench"
      ]
    },
    {
      "id": "2026-07-02-research-guides-batch",
      "url": "https://verdictpal.com/guides",
      "title": "Five research guides now on the atlas",
      "content_text": "Citation audit, systematic review protocol, data cleaning pipeline, grant writer path, and peer reviewer path — solid guides with step blocks and verification checklists.",
      "date_published": "2026-07-02T00:00:00.000Z",
      "tags": [
        "guide"
      ]
    },
    {
      "id": "2026-06-25-student-research-workflow",
      "url": "https://verdictpal.com/guides/trusted-source-brief",
      "title": "Student research workflow on the guides pillar",
      "content_text": "Path, stack, playbook, and showcase for turning broad prompts into source-backed briefs — the workflow we point new readers at first.",
      "date_published": "2026-06-25T00:00:00.000Z",
      "tags": [
        "guide"
      ]
    },
    {
      "id": "2026-06-24-factory-flagship",
      "url": "https://verdictpal.com/tools/factory",
      "title": "Factory flagship dossier shipped",
      "content_text": "Agent-native delivery with Droid, Spec Mode, missions, and enterprise controls — fit 74 with the autonomy and pricing caveats on the record.",
      "date_published": "2026-06-24T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-23-cursor-flagship",
      "url": "https://verdictpal.com/tools/cursor",
      "title": "Cursor flagship dossier shipped",
      "content_text": "VS Code-compatible AI editor at fit 81 — Tab, Chat, Agent, Privacy Mode, team governance, and the diff-review failure modes we publish before you merge.",
      "date_published": "2026-06-23T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-22-gemini-flagship",
      "url": "https://verdictpal.com/tools/gemini",
      "title": "Gemini flagship dossier shipped",
      "content_text": "Google's multimodal assistant ecosystem at fit 75 — Search grounding, Deep Research, storage bundles, and where it is not a cited search engine.",
      "date_published": "2026-06-22T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-21-notebooklm-flagship",
      "url": "https://verdictpal.com/tools/notebooklm",
      "title": "NotebookLM flagship dossier shipped",
      "content_text": "Source-grounded reading workspace at fit 78 — corpus chat, Audio Overviews, Flashcards, and the empty-corpus trap documented on the dossier.",
      "date_published": "2026-06-21T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-20-perplexity-flagship",
      "url": "https://verdictpal.com/tools/perplexity",
      "title": "Perplexity flagship dossier shipped",
      "content_text": "Cited search front door at fit 72 — consumer Pro/Max, Deep Research, Sonar APIs, and failure modes for bibliographies you have not opened.",
      "date_published": "2026-06-20T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-15-opencode-flagship",
      "url": "https://verdictpal.com/tools/opencode",
      "title": "OpenCode flagship dossier shipped",
      "content_text": "The open-source coding agent at fit 95 — Plan/Build modes, Go and Zen pricing, 75+ providers, and the free-model privacy caveats on the record.",
      "date_published": "2026-06-15T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-15-command-code-flagship",
      "url": "https://verdictpal.com/tools/command-code",
      "title": "Command Code flagship dossier shipped",
      "content_text": "Terminal coding agent with taste-1 learning, open-model deals from $1/mo, and the full pricing and privacy picture — including what lives in .commandcode/.",
      "date_published": "2026-06-15T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-15-raycast-flagship",
      "url": "https://verdictpal.com/tools/raycast",
      "title": "Raycast flagship dossier shipped",
      "content_text": "The macOS launcher we install before debating another AI tab — extensions, Quick AI, clipboard, and honest failure modes at fit 97.",
      "date_published": "2026-06-15T00:00:00.000Z",
      "tags": [
        "tool"
      ]
    },
    {
      "id": "2026-06-08-grant-deadline-set",
      "url": "https://verdictpal.com/guides",
      "title": "Grant deadline set (archived)",
      "content_text": "The grant-deadline path has been retired. Academic guides now live as three trios: student researcher, literature reviewer, and peer reviewer.",
      "date_published": "2026-06-08T00:00:00.000Z",
      "tags": [
        "guide"
      ]
    },
    {
      "id": "2026-06-02-trusted-source-brief",
      "url": "https://verdictpal.com/guides/trusted-source-brief",
      "title": "Trusted-source brief workflow shipped",
      "content_text": "The Guides pillar now has a real student research path, tool stack, playbook, and showcase for turning broad prompts into source-backed briefs.",
      "date_published": "2026-06-02T00:00:00.000Z",
      "tags": [
        "guide"
      ]
    },
    {
      "id": "2026-06-02-dark-mode-polish",
      "url": "https://verdictpal.com/brand",
      "title": "Dark mode rebuilt for the evidence instrument",
      "content_text": "Dark surfaces now use muted amber, readable borders, and component-specific overrides instead of neon panels and inverted cream panels.",
      "date_published": "2026-06-02T00:00:00.000Z",
      "tags": [
        "design"
      ]
    }
  ]
}
