[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"categories":3,"post-how-we-test-ai-assistants":85,"post-how-we-test-ai-assistants-related":150},[4,15,25,35,45,55,65,75],{"id":5,"documentId":6,"name":7,"slug":8,"shortName":9,"description":10,"icon":11,"createdAt":12,"updatedAt":12,"publishedAt":13,"seoTitle":14,"seoDescription":14},43,"d2n69w6u5jw2n9uxksqsks74","AI Chatbots & Assistants","ai-chatbots","AI Chatbots","Conversational AI assistants for reasoning, writing and everyday tasks.","Bot","2026-09-07T09:42:12.650Z","2026-09-07T09:42:12.677Z",null,{"id":16,"documentId":17,"name":18,"slug":19,"shortName":20,"description":21,"icon":22,"createdAt":23,"updatedAt":23,"publishedAt":24,"seoTitle":14,"seoDescription":14},45,"q3aub1u192qczyq8cyurhbau","AI Coding Tools","ai-coding","AI Coding","AI pair programmers, code completion and agentic coding tools.","Code","2026-09-07T09:42:12.712Z","2026-09-07T09:42:12.726Z",{"id":26,"documentId":27,"name":28,"slug":29,"shortName":30,"description":31,"icon":32,"createdAt":33,"updatedAt":33,"publishedAt":34,"seoTitle":14,"seoDescription":14},49,"kyxgrju9qcbsahxr6c3gy1rk","AI Dev Frameworks","ai-frameworks","AI Frameworks","Frameworks and tooling for building LLM-powered applications.","Blocks","2026-09-07T09:42:12.793Z","2026-09-07T09:42:12.817Z",{"id":36,"documentId":37,"name":38,"slug":39,"shortName":40,"description":41,"icon":42,"createdAt":43,"updatedAt":43,"publishedAt":44,"seoTitle":14,"seoDescription":14},47,"et5dhnhik9xxbypugwx4pfh5","AI Image & Video","ai-image-video","AI Image\u002FVideo","Generative models for images, art and video production.","Image","2026-09-07T09:42:12.751Z","2026-09-07T09:42:12.767Z",{"id":46,"documentId":47,"name":48,"slug":49,"shortName":50,"description":51,"icon":52,"createdAt":53,"updatedAt":53,"publishedAt":54,"seoTitle":14,"seoDescription":14},37,"cnjfk8r7abaq5mkotekzo8fz","Backend & Databases","backend-databases","Backend","Databases, APIs and data-layer architecture for server-side systems.","Database","2026-09-07T09:42:12.415Z","2026-09-07T09:42:12.474Z",{"id":56,"documentId":57,"name":58,"slug":59,"shortName":60,"description":61,"icon":62,"createdAt":63,"updatedAt":63,"publishedAt":64,"seoTitle":14,"seoDescription":14},39,"s4ujx5e37x7cla49kfbz9xnq","DevOps & Cloud","devops-cloud","DevOps","Containers, infrastructure and cloud platforms for shipping and running software.","Cloud","2026-09-07T09:42:12.508Z","2026-09-07T09:42:12.541Z",{"id":66,"documentId":67,"name":68,"slug":69,"shortName":70,"description":71,"icon":72,"createdAt":73,"updatedAt":73,"publishedAt":74,"seoTitle":14,"seoDescription":14},35,"zy2v9u5ac9aeyfl4q9mnsm1g","Frontend Frameworks","frontend","Frontend","UI frameworks, meta-frameworks, styling and state management for the browser.","LayoutTemplate","2026-09-07T09:42:12.327Z","2026-09-07T09:42:12.375Z",{"id":76,"documentId":77,"name":78,"slug":79,"shortName":80,"description":81,"icon":82,"createdAt":83,"updatedAt":83,"publishedAt":84,"seoTitle":14,"seoDescription":14},41,"kqqemwew4iwvgu0896gur8la","Programming Languages","languages","Languages","Systems and general-purpose languages compared for performance and ergonomics.","Terminal","2026-09-07T09:42:12.577Z","2026-09-07T09:42:12.595Z",{"id":56,"documentId":86,"title":87,"slug":88,"dek":89,"author":90,"date":91,"readingTime":92,"featured":93,"blocks":94,"createdAt":140,"updatedAt":140,"publishedAt":141,"seoTitle":14,"seoDescription":14,"category":142,"tags":143},"f3sy3t85izr49cmll6o0wcs1","How We Actually Test AI Assistants","how-we-test-ai-assistants","Benchmarks tell you one story. Here’s how we form an opinion beyond the leaderboard.","ComparedStack Team","2026-09-01","5 min read",true,[95,101,107,128,132,136],{"type":96,"children":97},"paragraph",[98],{"text":99,"type":100},"Public benchmarks are useful, but they’re also gameable, quickly saturated, and often disconnected from what a task actually feels like day to day. So while we track them, we don’t let them write our verdicts.","text",{"type":102,"level":103,"children":104},"heading",2,[105],{"text":106,"type":100},"Our actual test suite",{"type":108,"format":109,"children":110},"list","unordered",[111,116,120,124],{"type":112,"children":113},"list-item",[114],{"text":115,"type":100},"Long-document comprehension: feed it a 40-page spec and ask questions that require connecting details from page 3 and page 35",{"type":112,"children":117},[118],{"text":119,"type":100},"Real coding tasks: multi-file refactors in an existing, messy codebase — not a fresh scaffold",{"type":112,"children":121},[122],{"text":123,"type":100},"Tone under pressure: does it push back on a bad idea, or just agree with whatever you said",{"type":112,"children":125},[126],{"text":127,"type":100},"Recovery: how gracefully it handles being told it made a mistake",{"type":96,"children":129},[130],{"text":131,"type":100},"None of these produce a clean numeric score on their own — they inform the qualitative pros and cons you see in each comparison, which we think matters more than a leaderboard rank that can shift with the next model update.",{"type":102,"level":103,"children":133},[134],{"text":135,"type":100},"Why we keep revisiting",{"type":96,"children":137},[138],{"text":139,"type":100},"Model updates land fast enough that a comparison written six months ago can quietly go stale. That’s why AI comparisons carry an “updated” date front and center, and why we treat this category as a living document rather than a one-time verdict.","2026-09-07T09:42:17.106Z","2026-09-07T09:42:17.131Z",{"id":5,"documentId":6,"name":7,"slug":8,"shortName":9,"description":10,"icon":11,"createdAt":12,"updatedAt":12,"publishedAt":13,"seoTitle":14,"seoDescription":14},[144,147],{"id":145,"value":146},974,"ai",{"id":148,"value":149},975,"methodology",[]]