PageSourceSearch

https://www.promptfoo.dev/assets/js/dfc4ebcb.85dbd87d.js

js promptfoo.dev collected 2026-09-24 17:59:47 UTC 8,698 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkpromptfoo_docs||=[]).push([[12318],{1425(e,n,i){i.r(n),i.d(n,{assets:()=>c,contentTitle:()=>l,default:()=>d,frontMatter:()=>t,metadata:()=>a,toc:()=>o});const a=JSON.parse('{"id":"red-team/plugins/harmbench","title":"HarmBench Plugin","description":"Red team LLM safety using HarmBench\'s 400 harmful behavior tests, with filters for specific semantic and functional categories","source":"@site/docs/red-team/plugins/harmbench.md","sourceDirName":"red-team/plugins","slug":"/red-team/plugins/harmbench","permalink":"/docs/red-team/plugins/harmbench","draft":false,"unlisted":false,"tags":[],"version":"current","lastUpdatedBy":"renovate[bot]","lastUpdatedAt":1790260018000,"frontMatter":{"sidebar_label":"HarmBench","description":"Red team LLM safety using HarmBench\'s 400 harmful behavior tests, with filters for specific semantic and functional categories"},"sidebar":"promptfoo","previous":{"title":"FERPA","permalink":"/docs/red-team/plugins/ferpa"},"next":{"title":"Aegis","permalink":"/docs/red-team/plugins/aegis"}}');var s=i(62540),r=i(43023);const t={sidebar_label:"HarmBench",description:"Red team LLM safety using HarmBench's 400 harmful behavior tests, with filters for specific semantic and functional categories"},l="HarmBench Plugin",c={},o=[{value:"Configuration",id:"configuration",level:2},{value:"Filtering by Category",id:"filtering-by-category",level:3},{value:"Filtering by Functional Category",id:"filtering-by-functional-category",level:3},{value:"References",id:"references",level:2},{value:"Related Concepts",id:"related-concepts",level:2}];function h(e){const n={a:"a",code:"code",h1:"h1",h2:"h2",h3:"h3",header:"header",li:"li",p:"p",pre:"pre",ul:"ul",...(0,r.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(n.header,{children:(0,s.jsx)(n.h1,{id:"harmbench-plugin",children:"HarmBench Plugin"})}),"\n",(0,s.jsxs)(n.p,{children:["The HarmBench plugin uses the ",(0,s.jsx)(n.a,{href:"https://github.com/centerforaisafety/HarmBench/tree/main/data/behavior_datasets",children:"HarmBench dataset"})," published by the ",(0,s.jsx)(n.a,{href:"https://www.safe.ai/",children:"Center for AI Safety"})," to test LLM responses to harmful prompts. The dataset evaluates models across 400 key harmful behaviors including:"]}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Chemical and biological threats (e.g., dangerous substances, weapons)"}),"\n",(0,s.jsx)(n.li,{children:"Illegal activities (e.g., theft, fraud, trafficking)"}),"\n",(0,s.jsx)(n.li,{children:"Misinformation and conspiracy theories"}),"\n",(0,s.jsx)(n.li,{children:"Harassment and hate speech"}),"\n",(0,s.jsx)(n.li,{children:"General harmful requests"}),"\n",(0,s.jsx)(n.li,{children:"Cybercrime (e.g., malware, system exploitation)"}),"\n",(0,s.jsx)(n.li,{children:"Copyright violations"}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["Unlike many of our other plugins, the HarmBench plugin uses a statically generated dataset vs dynamically generating cases such as those for ",(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/harmful",children:"Harmful behavior detection"})," (which was partially based on HarmBench's classifications for harmful behavior), ",(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/excessive-agency",children:"excessive agency"}),", ",(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/hallucination",children:"hallucination"})," and others. It is still however a great tool but not something that can be used in lieu of a constantly evolving, dynamically generated set of test cases."]}),"\n",(0,s.jsx)(n.h2,{id:"configuration",children:"Configuration"}),"\n",(0,s.jsx)(n.p,{children:"To include the HarmBench plugin in your LLM red teaming setup:"}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-yaml",children:"redteam:\n  plugins:\n    - harmbench\n"})}),"\n",(0,s.jsxs)(n.p,{children:["You can control the number of test cases using the ",(0,s.jsx)(n.code,{children:"numTests"})," parameter:"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-yaml",children:"redteam:\n  plugins:\n    - id: harmbench\n      numTests: 25 # The default is 5, but you can define the limit here.\n"})}),"\n",(0,s.jsx)(n.h3,{id:"filtering-by-category",children:"Filtering by Category"}),"\n",(0,s.jsx)(n.p,{children:"You can run a subset of HarmBench by filtering the dataset's semantic categories:"}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-yaml",children:"redteam:\n  plugins:\n    - id: harmbench\n      numTests: 20\n      config:\n        categories:\n          - cybercrime_intrusion\n          - misinformation_disinformation\n"})}),"\n",(0,s.jsx)(n.p,{children:"The available semantic categories are:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"chemical_biological"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"copyright"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"cybercrime_intrusion"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"harassment_bullying"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"harmful"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"illegal"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"misinformation_disinformation"})}),"\n"]}),"\n",(0,s.jsxs)(n.p,{children:["Common aliases such as ",(0,s.jsx)(n.code,{children:"cybercrime"}),", ",(0,s.jsx)(n.code,{children:"misinformation"}),", and ",(0,s.jsx)(n.code,{children:"chemical and biological"})," are also accepted and normalized to the canonical values above."]}),"\n",(0,s.jsx)(n.h3,{id:"filtering-by-functional-category",children:"Filtering by Functional Category"}),"\n",(0,s.jsxs)(n.p,{children:["HarmBench also distinguishes between ",(0,s.jsx)(n.code,{children:"standard"}),", ",(0,s.jsx)(n.code,{children:"contextual"}),", and ",(0,s.jsx)(n.code,{children:"copyright"}
1)," behaviors. You can filter by those functional slices as well:"]}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-yaml",children:"redteam:\n  plugins:\n    - id: harmbench\n      numTests: 20\n      config:\n        categories:\n          - misinformation\n        functionalCategories:\n          - contextual\n"})}),"\n",(0,s.jsx)(n.p,{children:"When you set both semantic and functional filters, Promptfoo generates tests from the matching intersection."}),"\n",(0,s.jsx)(n.p,{children:"The available functional categories are:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"standard"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"contextual"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.code,{children:"copyright"})}),"\n"]}),"\n",(0,s.jsx)(n.h2,{id:"references",children:"References"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://arxiv.org/abs/2402.04249",children:"HarmBench Paper"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://github.com/centerforaisafety/HarmBench/tree/main/data/behavior_datasets",children:"HarmBench Dataset"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://www.safe.ai/",children:"Center for AI Safety"})}),"\n"]}),"\n",(0,s.jsx)(n.h2,{id:"related-concepts",children:"Related Concepts"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.a,{href:"/docs/red-team/llm-vulnerability-types/",children:"Types of LLM vulnerabilities"})," - Full vulnerability and plugin directory with category mapping"]}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/guides/evaling-with-harmbench",children:"Evaluating LLM safety with HarmBench"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/harmful",children:"Harmful Content Plugin"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/beavertails",children:"BeaverTails Plugin"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/cyberseceval",children:"CyberSecEval Plugin"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/red-team/plugins/pliny",children:"Pliny Plugin"})}),"\n"]})]})}function d(e={}){const{wrapper:n}={...(0,r.R)(),...e.components};return n?(0,s.jsx)(n,{...e,children:(0,s.jsx)(h,{...e})}):h(e)}},43023(e,n,i){i.d(n,{R:()=>t,x:()=>l});var a=i(63696);const s={},r=a.createContext(s);function t(e){const n=a.useContext(r);return a.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function l(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:t(e.components),a.createElement(r.Provider,{value:n},e.children)}}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.