{"data":{"slug":"ai-red-team-test-plan","title":"AI Red-Team and Evaluation Test Plan","type":"assessment","formats":["xlsx","docx"],"topics":["risk","evidence"],"frameworks":["eu-ai-act","nist-ai-rmf"],"short":"A test plan and case log for adversarial and evaluation testing: accuracy, robustness, prompt injection, bias, harmful content, privacy leakage and misuse, with severity and retest tracking.","covers":{"categories":["safety_testing","accuracy_robustness_security"]},"version":1,"dataset_version":"bb068ecd9dad","generated_at":"2026-09-28T13:37:55+00:00","citations":16,"downloads":0,"files":[{"format":"xlsx","filename":"ai-red-team-test-plan-v1.xlsx","bytes":19063,"url":"https://aipolicytracker.org/templates/ai-red-team-test-plan#download"},{"format":"docx","filename":"ai-red-team-test-plan-v1.docx","bytes":13220,"url":"https://aipolicytracker.org/templates/ai-red-team-test-plan#download"}],"url":"https://aipolicytracker.org/templates/ai-red-team-test-plan","inside":["Test cases sheet with area, scenario, expected and observed behaviour, outcome and severity","Plan document: scope, independence, method, exit criteria","Testing duties sheet: safety-testing and robustness duties on record"],"legal_basis":["eu-ai-act","us-nist-ai-rmf"],"caveat":null,"covered_obligations":[{"slug":"us-california-sb-53-frontier-ai-framework","title":"Large frontier developers must publish a frontier AI framework","policy":"us-california-sb-53","source_reference":"Business and Professions Code, Chapter 25.1 (as added by SB 53)","url":"https://aipolicytracker.org/obligations/us-california-sb-53-frontier-ai-framework"},{"slug":"us-california-sb-53-catastrophic-risk-assessment-summaries","title":"Large frontier developers must send periodic summaries of catastrophic-risk assessments to the state","policy":"us-california-sb-53","source_reference":"Business and Professions Code Section 22757.12 (as added by SB 53)","url":"https://aipolicytracker.org/obligations/us-california-sb-53-catastrophic-risk-assessment-summaries"},{"slug":"eu-ai-act-accuracy-robustness-cybersecurity","title":"Achieve appropriate accuracy, robustness and cybersecurity","policy":"eu-ai-act","source_reference":"Article 15","url":"https://aipolicytracker.org/obligations/eu-ai-act-accuracy-robustness-cybersecurity"},{"slug":"eu-ai-act-gpai-systemic-risk","title":"Manage systemic risk for high-impact general-purpose models","policy":"eu-ai-act","source_reference":"Articles 51, 52 and 55","url":"https://aipolicytracker.org/obligations/eu-ai-act-gpai-systemic-risk"},{"slug":"eu-ai-act-art-55-systemic-risk-cybersecurity","title":"Providers of systemic-risk GPAI models must secure the model and its infrastructure","policy":"eu-ai-act","source_reference":"Article 55(1)(d)","url":"https://aipolicytracker.org/obligations/eu-ai-act-art-55-systemic-risk-cybersecurity"},{"slug":"us-new-york-city-local-law-144-bias-audit","title":"Employers and employment agencies must obtain an independent bias audit before using an automated employment decision tool","policy":"us-new-york-city-local-law-144-automated-employment-decision-tools","source_reference":"NYC Administrative Code Section 20-871(a)(1); 6 RCNY Section 5-301","url":"https://aipolicytracker.org/obligations/us-new-york-city-local-law-144-bias-audit"},{"slug":"south-korea-ai-basic-act-art-32-safety-measures-for-high-performance-ai","title":"Operators of AI above the compute threshold must run lifecycle risk management and report safety results","policy":"south-korea-framework-act-on-the-development-of-artificial-intelligence-and-establishment-of-a-foundation-for","source_reference":"Article 32","url":"https://aipolicytracker.org/obligations/south-korea-ai-basic-act-art-32-safety-measures-for-high-performance-ai"},{"slug":"us-nist-ai-rmf-measure","title":"Measure and test trustworthiness characteristics (Measure)","policy":"us-nist-ai-rmf","source_reference":"MEASURE function","url":"https://aipolicytracker.org/obligations/us-nist-ai-rmf-measure"}],"versions":[{"version":1,"generated_at":"2026-09-28T13:37:55+00:00","dataset_version":"bb068ecd9dad","changelog":"First version, built from dataset bb068ecd9dad.","stats":{"sheets":{"Test cases":0,"Testing duties":8},"blocks":59,"headings":14,"citations":16}}]},"meta":{"license":"CC BY 4.0","license_url":"https://creativecommons.org/licenses/by/4.0/","disclaimer":"This template is generated from the records on aipolicytracker.org. It is an informational resource, not legal advice, and completing it does not make an organisation compliant with any law or standard. Every row that cites a duty links to the record it came from; check the official source before relying on it."}}