[
  {
    "id": "G01",
    "title": "Software engineering trends for 2025 and beyond",
    "publisher": "Gartner",
    "date": "2025-07-01",
    "kind": "Forecast",
    "url": "https://www.gartner.com/en/newsroom/press-releases/2025-07-01-gartner-identifies-the-top-strategic-trends-in-software-engineering-for-2025-and-beyond",
    "takeaway": "Forecasts 90% of enterprise software engineers using AI code assistants by 2028.",
    "caveat": "A forecast about tool use, not a measured QE outcome or a savings estimate.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "G02",
    "title": "How to Select an AI-Augmented Software Testing Platform",
    "publisher": "Gartner",
    "date": "2026-02-03",
    "kind": "Analyst perspective",
    "url": "https://www.gartner.com/en/documents/7395930",
    "takeaway": "The public abstract frames platform selection around quality and business value beyond test automation.",
    "caveat": "Only the abstract was reviewed; detailed criteria, findings and vendor assessments require licensed access.",
    "scope": "Industry",
    "access": "Public abstract; full report gated",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "G03",
    "title": "Journey Guide to Improving Software Testing in the AI Age",
    "publisher": "Gartner",
    "date": "2026-04-02",
    "kind": "Analyst perspective",
    "url": "https://www.gartner.com/en/documents/7678461",
    "takeaway": "The public summary positions continuous quality as a response to more complex AI-era delivery.",
    "caveat": "The full guide was not available for review. No proprietary maturity model is reproduced.",
    "scope": "Industry",
    "access": "Public abstract; full report gated",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "G04",
    "title": "Smaller software engineering teams by 2029",
    "publisher": "Gartner",
    "date": "2026-07-07",
    "kind": "Forecast",
    "url": "https://www.gartner.com/en/newsroom/press-releases/2026-07-07-gartner-predicts-60-percent-of-organizations-will-adopt-smaller-software-engineering-teams-by-2029",
    "takeaway": "Forecasts 60% of organizations adopting smaller engineering teams at scale by 2029; emphasizes team redesign and platform enablement.",
    "caveat": "Not evidence for reducing QE headcount; the release distinguishes restructuring from cost optimization.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "M01",
    "title": "Unlocking the value of AI in software development",
    "publisher": "McKinsey",
    "date": "2025-11-03",
    "kind": "Survey",
    "url": "https://www.mckinsey.com/industries/technology-media-and-telecommunications/our-insights/unlocking-the-value-of-ai-in-software-development",
    "takeaway": "Surveyed nearly 300 leaders; 100 assessed impact. The top outcome quintile reported stronger productivity, quality and speed.",
    "caveat": "Self-reported outcomes and selected top performers; no causal estimate and no representative bank savings rate.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "M02",
    "title": "The AI revolution in software development",
    "publisher": "McKinsey",
    "date": "2026-04-01",
    "kind": "Case study",
    "url": "https://www.mckinsey.com/capabilities/tech-and-ai/our-insights/the-ai-revolution-in-software-development",
    "takeaway": "Describes agent-based delivery and an unnamed global bank implementation as a strategic direction.",
    "caveat": "Consultancy-reported case; its speed and cost claims lack a public independent counterfactual. Do not transfer them into a business case.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": "https://www.mckinsey.com/~/media/mckinsey/business%20functions/tech%20and%20ai/our%20insights/the%20ai%20revolution%20in%20software%20development/the-ai-revolution-in-software-development_final.pdf",
    "reviewed": "2026-09-05"
  },
  {
    "id": "W01",
    "title": "World Quality Report 2025\u201326: public findings",
    "publisher": "Capgemini / Sogeti / OpenText",
    "date": "2025-11-13",
    "kind": "Survey",
    "url": "https://www.capgemini.com/ch-en/insights/research-library/world-quality-report-2025-26/",
    "takeaway": "Reports widespread experimentation, limited enterprise scale and material privacy, integration, reliability and skills barriers.",
    "caveat": "Vendor-sponsored survey of over 2,000 executives in 22 countries and 10 sectors. Question-specific sample sizes are not in the public release.",
    "scope": "Industry",
    "access": "Public release; full report registration",
    "document_url": "https://www.capgemini.com/fi-en/wp-content/uploads/sites/26/2025/11/2025_11_13_World_Quality_Report_2025_.pdf",
    "reviewed": "2026-09-05"
  },
  {
    "id": "D01",
    "title": "State of AI-assisted Software Development 2025",
    "publisher": "DORA / Google Cloud",
    "date": "2025-09-23",
    "kind": "Survey",
    "url": "https://dora.dev/research/2025/dora-report/",
    "takeaway": "AI adoption interacts with the surrounding engineering system; organizational foundations matter for delivery outcomes.",
    "caveat": "Observational research and reported associations do not isolate the effect of a tool. Partner-sponsored research.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": "https://www.thoughtworks.com/content/dam/thoughtworks/documents/report/tw_report_state_of_ai_assisted_software_development_2025.pdf",
    "reviewed": "2026-09-05"
  },
  {
    "id": "D02",
    "title": "Balancing AI tensions: moving from AI adoption to effective software development",
    "publisher": "DORA",
    "date": "2026-03-10",
    "kind": "Qualitative study",
    "url": "https://dora.dev/insights/balancing-ai-tensions/",
    "takeaway": "Analysis of 1,110 open-ended Google engineer responses highlights verification work, context limitations and review friction.",
    "caveat": "One-company qualitative sample from Q3 2025; prompts may have focused respondents on code generation.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "E01",
    "title": "Early-2025 AI on experienced open-source developer productivity",
    "publisher": "METR",
    "date": "2025-07-10",
    "kind": "Randomized study",
    "url": "https://metr.org/blog/2025-07-10-early-2025-ai-experienced-os-dev-study/",
    "takeaway": "16 developers and 246 tasks: access to early-2025 AI increased completion time by 19% (95% CI: +2% to +39%).",
    "caveat": "Experienced developers on familiar repositories; narrow task and tool setting, not a claim about all developers or 2026 agents.",
    "scope": "Delivery evidence",
    "access": "Public text",
    "document_url": "https://metr.org/Early_2025_AI_Experienced_OS_Devs_Study-paper.pdf",
    "reviewed": "2026-09-05"
  },
  {
    "id": "E02",
    "title": "Updated developer productivity study",
    "publisher": "METR",
    "date": "2026-02-24",
    "kind": "Study update",
    "url": "https://metr.org/blog/2026-02-24-uplift-update/",
    "takeaway": "Point estimates suggest faster work: returning developers \u221218% time, new recruits \u22124%; both confidence intervals include no effect.",
    "caveat": "Selection effects and concurrent-agent time measurement weaken the estimates. This is not a clean temporal trend against the earlier trial.",
    "scope": "Delivery evidence",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "E03",
    "title": "The Effects of Generative AI on High-Skilled Work: Evidence from Three Field Experiments with Software Developers",
    "publisher": "Cui and colleagues / Management Science",
    "date": "2026-02-27",
    "kind": "Randomized study",
    "url": "https://pubsonline.informs.org/doi/10.1287/mnsc.2025.00535",
    "takeaway": "Across 4,867 developers, the pooled estimate was 26.08% more completed tasks, with standard error 10.3 percentage points.",
    "caveat": "Coding-assistant field experiments, not QE-specific labor savings. Public author PDF is the February 2025 working paper; publication followed in 2026.",
    "scope": "Delivery evidence",
    "access": "Publisher abstract + open author paper",
    "document_url": "https://economics.mit.edu/sites/default/files/inline-files/draft_copilot_experiments.pdf",
    "reviewed": "2026-09-05"
  },
  {
    "id": "E04",
    "title": "Automated Unit Test Improvement using Large Language Models at Meta",
    "publisher": "Alshahwan and colleagues / FSE 2024",
    "date": "2024-02-14",
    "kind": "Case study",
    "url": "https://arxiv.org/abs/2402.09171",
    "takeaway": "TestGen-LLM improves existing tests using build, reliability and coverage filters before developer review.",
    "caveat": "Single-company evaluation. Generated tests and test-a-thon recommendations use different denominators; acceptance does not establish defect reduction.",
    "scope": "AI for QE",
    "access": "Public text",
    "document_url": "https://arxiv.org/pdf/2402.09171",
    "reviewed": "2026-09-05"
  },
  {
    "id": "E05",
    "title": "LLM-powered bug catchers: Meta ACH",
    "publisher": "Meta Engineering",
    "date": "2025-02-05",
    "kind": "Case study",
    "url": "https://engineering.fb.com/2025/02/05/security/revolutionizing-software-testing-llm-powered-bug-catchers-meta-ach/",
    "takeaway": "Uses mutation-guided generation: candidate tests must detect a modeled fault while passing the original code.",
    "caveat": "Tests target selected fault concerns; effectiveness depends on mutation relevance and human review. Company-authored implementation report.",
    "scope": "AI for QE",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "E06",
    "title": "LLM-Based Automated Diagnosis Of Integration Test Failures At Google",
    "publisher": "Ziftci and colleagues / Google",
    "date": "2026-04-13",
    "kind": "Case study",
    "url": "https://arxiv.org/abs/2604.12108",
    "takeaway": "Reports 90.14% diagnosis accuracy on 71 manually evaluated failures and deployment on 52,635 distinct failing tests.",
    "caveat": "The small reviewed accuracy sample is different from the deployment population. Feedback usefulness is not diagnostic accuracy.",
    "scope": "AI for QE",
    "access": "Public text",
    "document_url": "https://arxiv.org/pdf/2604.12108",
    "reviewed": "2026-09-05"
  },
  {
    "id": "E07",
    "title": "FlakyGuard: Automatically Fixing Flaky Tests at Industry Scale",
    "publisher": "Li and colleagues / UT Austin and Uber",
    "date": "2025-11-18",
    "kind": "Case study",
    "url": "https://arxiv.org/abs/2511.14002",
    "takeaway": "Uses dynamic call-graph context and analysis to repair flaky tests in a Go monorepo.",
    "caveat": "Six-month enterprise case. Reported repair and acceptance percentages use different populations; weak fixes can change test semantics.",
    "scope": "AI for QE",
    "access": "Public text",
    "document_url": "https://arxiv.org/pdf/2511.14002",
    "reviewed": "2026-09-05"
  },
  {
    "id": "A01",
    "title": "Artificial Intelligence Risk Management Framework: Generative AI Profile",
    "publisher": "NIST",
    "date": "2024-07-26",
    "kind": "Framework",
    "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf",
    "takeaway": "Provides a cross-sector structure for identifying and managing generative-AI risks throughout the lifecycle.",
    "caveat": "Voluntary guidance; translate it into system-specific controls and testable requirements.",
    "scope": "QE for AI",
    "access": "Public text",
    "document_url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf",
    "reviewed": "2026-09-05"
  },
  {
    "id": "A02",
    "title": "AI RMF Core: Govern, Map, Measure, Manage",
    "publisher": "NIST",
    "date": "2023-01-26",
    "kind": "Framework",
    "url": "https://airc.nist.gov/airmf-resources/airmf/5-sec-core/",
    "takeaway": "Connects lifecycle risk management with representative evaluation, documented uncertainty and ongoing monitoring.",
    "caveat": "RMF 1.0 remains the referenced framework; NIST says it is being revised. It is not a product certification.",
    "scope": "QE for AI",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "A03",
    "title": "Top 10 for Agentic Applications 2026",
    "publisher": "OWASP GenAI Security Project",
    "date": "2025-12-09",
    "kind": "Framework",
    "url": "https://genai.owasp.org/2025/12/09/owasp-top-10-for-agentic-applications-the-benchmark-for-agentic-security-in-the-age-of-autonomous-ai/",
    "takeaway": "Threat guidance extends evaluation to agent goals, tool use, identity, supply chains and memory.",
    "caveat": "Community risk taxonomy; use it to design scenarios, not to infer measured incident rates for a bank.",
    "scope": "QE for AI",
    "access": "Public text",
    "document_url": "https://genai.owasp.org/download/52117/?tmstv=1765059207",
    "reviewed": "2026-09-05"
  },
  {
    "id": "A04",
    "title": "OWASP GenAI LLM Top 10 2026",
    "publisher": "OWASP GenAI Security Project",
    "date": "2026-08-03",
    "kind": "Framework",
    "url": "https://genai.owasp.org/resource/owasp-genai-llm-top-10-2026/",
    "takeaway": "The 2026 guide updates LLM application risks and explicitly pairs model-component risks with agentic-system risks.",
    "caveat": "Resource page dated August 3; release announcement September 1 has a September 2 dateline. Downloaded PDF retains date placeholders. Reviewed as the linked 2026 edition.",
    "scope": "QE for AI",
    "access": "Public text",
    "document_url": "https://genai.owasp.org/download/56857/?tmstv=1785822482",
    "reviewed": "2026-09-05"
  },
  {
    "id": "A05",
    "title": "Agent Control Standard (ACS)",
    "publisher": "OWASP GenAI Security Project",
    "date": "2026-09-01",
    "kind": "Framework",
    "url": "https://genai.owasp.org/resource/agent-control-standard-acs/",
    "takeaway": "Introduces common hooks for runtime policy enforcement and observability across agent frameworks.",
    "caveat": "Newly donated open standard; evaluate implementation coverage and integration maturity before relying on portability.",
    "scope": "QE for AI",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "R01",
    "title": "Generative and agentic AI: technology, cyber security and operational resilience",
    "publisher": "OSFI",
    "date": "2026-07",
    "kind": "Supervisory guidance",
    "url": "https://www.osfi-bsif.gc.ca/en/risks/technology-cyber-risk-management/technology-risk-bulletin/generative-agentic-artificial-intelligence-implications-technology-cyber-security-operational",
    "takeaway": "Discusses scoped agent identities, tool restrictions, testing, traceability, approval and fallback practices for financial institutions.",
    "caveat": "Technology Risk Bulletin offers sound practices that complement existing guidelines; it is not a new standalone binding AI rule.",
    "scope": "Financial services",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "R02",
    "title": "Guideline E-23: Model Risk Management (2027)",
    "publisher": "OSFI",
    "date": "2025-09-11",
    "kind": "Supervisory guidance",
    "url": "https://www.osfi-bsif.gc.ca/en/guidance/guidance-library/guideline-e-23-model-risk-management-2027",
    "takeaway": "Risk-based model governance and lifecycle expectations provide a planning reference for applicable AI systems.",
    "caveat": "Effective May 1, 2027. Assess applicability under institutional model policy; do not label every coding assistant a regulated model by default.",
    "scope": "Financial services",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "T01",
    "title": "Tosca Agentic AI documentation",
    "publisher": "Tricentis",
    "date": "Living documentation",
    "kind": "Product documentation",
    "url": "https://docs.tricentis.com/tosca-2026.1/en-us/content/agentic_ai/landing_page.htm",
    "takeaway": "Documents natural-language support for finding test assets, explaining results and generating test cases.",
    "caveat": "Capability description for Tosca 2026.1; not an independent assessment of accuracy, enterprise fit or return.",
    "scope": "Technology landscape",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "T02",
    "title": "Deterministic Visual AI tests",
    "publisher": "Applitools",
    "date": "Living documentation",
    "kind": "Product documentation",
    "url": "https://applitools.com/deterministic-visual-ai-tests/",
    "takeaway": "Describes visual comparison against approved baselines within existing testing frameworks.",
    "caveat": "Vendor product description. Validate false alerts, missed changes, dynamic content and baseline governance on your applications.",
    "scope": "Technology landscape",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "T03",
    "title": "Evaluation concepts and workflows",
    "publisher": "LangSmith / LangChain",
    "date": "Living documentation",
    "kind": "Product documentation",
    "url": "https://docs.langchain.com/langsmith/evaluation",
    "takeaway": "Documents offline datasets, evaluators and online feedback loops for AI applications.",
    "caveat": "Tool documentation does not establish evaluator validity. Calibrate judges and human labels to the target task.",
    "scope": "Technology landscape",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "T04",
    "title": "About GitHub Copilot code review",
    "publisher": "GitHub",
    "date": "Living documentation",
    "kind": "Product documentation",
    "url": "https://docs.github.com/en/copilot/concepts/agents/code-review",
    "takeaway": "Documents AI review suggestions within pull requests and editors.",
    "caveat": "Review suggestions supplement reviewer judgment; a product capability is not evidence of prevented defects.",
    "scope": "Technology landscape",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "T05",
    "title": "Red teaming for AI agents",
    "publisher": "Promptfoo",
    "date": "Living documentation",
    "kind": "Product documentation",
    "url": "https://www.promptfoo.dev/docs/red-team/agents/",
    "takeaway": "Documents adversarial testing of agent tool use, access boundaries and context.",
    "caveat": "A red-team suite samples a threat model. Passing it does not prove that an agent is secure.",
    "scope": "Technology landscape",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "T06",
    "title": "View and compare evaluation results",
    "publisher": "Microsoft Foundry",
    "date": "Living documentation",
    "kind": "Product documentation",
    "url": "https://learn.microsoft.com/en-us/azure/foundry/how-to/evaluate-results",
    "takeaway": "Documents run-level and row-level evaluation comparison, including quality and operational metrics.",
    "caveat": "Platform evaluation requires representative datasets, appropriate metrics and deployment-specific thresholds.",
    "scope": "Technology landscape",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  },
  {
    "id": "S01",
    "title": "Developer Survey 2025: AI",
    "publisher": "Stack Overflow",
    "date": "2025",
    "kind": "Survey",
    "url": "https://survey.stackoverflow.co/2025/ai",
    "takeaway": "Reported trust in AI accuracy remains mixed; distrust exceeds trust in the survey responses.",
    "caveat": "Self-selected developer survey. Overall sample size and AI-question response counts are different denominators.",
    "scope": "Industry",
    "access": "Public text",
    "document_url": null,
    "reviewed": "2026-09-05"
  }
]
