[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-microsoft-releases-opensource-framework-to-test-ai-behavior-from-text":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},236,"microsoft-releases-opensource-framework-to-test-ai-behavior-from-text","Microsoft Open-Sources a Text-Based AI Evaluation Framework","Microsoft's new open-source framework generates AI behavior tests from plain-text descriptions, aiming to lower the bar for model evaluation.","Microsoft shipped an open-source framework this week that converts plain-text behavior descriptions into AI evaluation tests.\n\nThe tool is called Adaptive Spec-driven Scoring for Evaluation and Regression Testing, a name that covers more syllables than most machine learning papers. Released Tuesday, it gives developers a way to describe how a model should behave in natural language, then generates evaluations from those descriptions. The framework includes a regression testing component, designed to catch when a model update breaks behavior that previously worked.\n\nBuilding reliable AI evaluations has historically required dedicated engineering work: writing test harnesses, defining scoring rubrics, and running them at scale. If text-based spec generation actually reduces that friction, it could make eval coverage accessible to teams without a dedicated ML research function. Regression testing in particular is underserved - models are updated constantly, and most teams have no systematic way to detect when a capability quietly degrades.\n\nWhether a natural-language spec can be precise enough to catch the failure modes that actually matter is the harder question. Vague descriptions produce vague tests, and vague tests produce false confidence.","[\"ai\",\"open-source\",\"testing\",\"microsoft\"]","2026-06-02T19:02:21.000Z","2026-06-02T19:50:43.968Z","2026-06-18T03:26:32.604Z","published",null,[],"dev-tools",[26,27,28,29],"ai","open-source","testing","microsoft",[31],{"name":32,"url":33},"TechCrunch","https:\u002F\u002Ftechcrunch.com\u002F2026\u002F06\u002F02\u002Fnew-microsoft-tool-lets-devs-spin-up-ai-behavior-tests-using-text-descriptions\u002F",0,{"sections":36},[37,41,46,51,56,61,66,71,76,80,85,90,95,100],{"name":38,"slug":26,"count":39,"latest_published_at":40},"AI",2602,"2026-07-18T18:30:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",315,"2026-07-17T19:30:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Policy","policy",169,"2026-07-17T19:49:53.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",126,"2026-07-16T20:09:48.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",94,"2026-07-16T16:29:46.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Software","software",72,"2026-07-17T09:42:05.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Science","science",66,"2026-07-10T10:29:37.000Z",{"name":77,"slug":24,"count":78,"latest_published_at":79},"Dev Tools",60,"2026-07-16T16:59:13.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",42,"2026-07-16T16:30:35.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",29,"2026-07-10T22:28:58.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]