[{"data":1,"prerenderedAt":1285},["ShallowReactive",2],{"docs-nav":3,"docs-\u002Fdocs\u002Fweb-eval":78},[4,9,13,16,21,26,30,34,38,42,46,50,53,58,62,66,69,74],{"path":5,"title":6,"navTitle":6,"group":7,"order":8},"\u002Fdocs\u002Fcommands","Other commands","Guide",999,{"path":10,"title":11,"navTitle":12,"group":7,"order":8},"\u002Fdocs\u002Fcompaction","Context and compaction","Context & compaction",{"path":14,"title":15,"navTitle":15,"group":7,"order":8},"\u002Fdocs\u002Fconfiguration","Configuration",{"path":17,"title":18,"navTitle":18,"group":19,"order":20},"\u002Fdocs\u002Fenvironment","Environment variables","Reference",11,{"path":22,"title":23,"navTitle":23,"group":24,"order":25},"\u002Fdocs\u002Fflags","Flag reference","CLI",8,{"path":27,"title":28,"navTitle":28,"group":7,"order":29},"\u002Fdocs","Introduction",1,{"path":31,"title":32,"navTitle":32,"group":7,"order":33},"\u002Fdocs\u002Finstallation","Installation",2,{"path":35,"title":36,"navTitle":36,"group":19,"order":37},"\u002Fdocs\u002Fproviders","Providers",10,{"path":39,"title":40,"navTitle":40,"group":7,"order":41},"\u002Fdocs\u002Fquickstart","Quickstart",3,{"path":43,"title":44,"navTitle":45,"group":7,"order":8},"\u002Fdocs\u002Freleases","Releases and upgrading","Releases",{"path":47,"title":48,"navTitle":49,"group":7,"order":8},"\u002Fdocs\u002Frun","The run command","Running an audit",{"path":51,"title":52,"navTitle":52,"group":7,"order":8},"\u002Fdocs\u002Fsecurity","Security model",{"path":54,"title":55,"navTitle":56,"group":24,"order":57},"\u002Fdocs\u002Fsessions","Sessions and forking","Sessions",6,{"path":59,"title":60,"navTitle":61,"group":7,"order":8},"\u002Fdocs\u002Fskills-roles","Skills and roles","Skills & roles",{"path":63,"title":64,"navTitle":64,"group":19,"order":65},"\u002Fdocs\u002Ftools","Tool reference",12,{"path":67,"title":68,"navTitle":68,"group":7,"order":8},"\u002Fdocs\u002Ftroubleshooting","Troubleshooting",{"path":70,"title":71,"navTitle":72,"group":24,"order":73},"\u002Fdocs\u002Ftui","The interactive TUI","Interactive TUI",9,{"path":75,"title":76,"navTitle":77,"group":7,"order":8},"\u002Fdocs\u002Fweb-eval","Dashboard and evaluation","Dashboard & evaluation",{"id":79,"title":76,"body":80,"description":1275,"extension":1276,"group":7,"meta":1277,"navTitle":77,"navigation":1278,"order":8,"path":75,"seo":1279,"stem":1283,"__hash__":1284},"docs\u002Fdocs\u002Fweb-eval.md",{"type":81,"value":82,"toc":1256},"minimark",[83,87,91,98,103,142,145,150,235,238,242,249,518,524,562,566,569,611,621,625,631,642,645,649,652,656,659,663,666,674,719,722,757,764,771,840,843,850,856,880,883,890,910,917,923,928,1033,1049,1054,1058,1064,1196,1208,1218,1224,1230,1234,1243,1247,1252],[84,85,76],"h1",{"id":86},"dashboard-and-evaluation",[88,89,90],"p",{},"Two tools that sit beside a run rather than inside it: a browser UI over the session database, and an\noffline harness for measuring whether a change to the prompts or the pipeline actually helped.",[88,92,93,94,97],{},"The command-line surface for both is in ",[95,96,6],"a",{"href":5},". This page is what is\nunderneath.",[99,100,102],"h2",{"id":101},"the-dashboard","The dashboard",[104,105,110],"pre",{"className":106,"code":107,"language":108,"meta":109,"style":109},"language-bash shiki shiki-themes vitesse-dark","locac web\nlocac web --cwd \u002Fpath\u002Fto\u002Ftarget --enable-runs\n","bash","",[111,112,113,125],"code",{"__ignoreMap":109},[114,115,117,121],"span",{"class":116,"line":29},"line",[114,118,120],{"class":119},"sCK9x","locac",[114,122,124],{"class":123},"s7rlk"," web\n",[114,126,127,129,132,136,139],{"class":116,"line":33},[114,128,120],{"class":119},[114,130,131],{"class":123}," web",[114,133,135],{"class":134},"sXjYR"," --cwd",[114,137,138],{"class":123}," \u002Fpath\u002Fto\u002Ftarget",[114,140,141],{"class":134}," --enable-runs\n",[88,143,144],{},"It is a single-page app served by the binary over a session database. There is no build step, no\nbundler, and no dependency. The HTML and the client script are embedded in the executable.",[146,147,149],"h3",{"id":148},"what-it-renders","What it renders",[151,152,153,166],"table",{},[154,155,156],"thead",{},[157,158,159,163],"tr",{},[160,161,162],"th",{},"View",[160,164,165],{},"Contents",[167,168,169,177,185,193,200,208,216,224],"tbody",{},[157,170,171,174],{},[172,173,56],"td",{},[172,175,176],{},"Every session in the database, with its title, model and turn count",[157,178,179,182],{},[172,180,181],{},"Session detail",[172,183,184],{},"The full transcript, plus that session's findings",[157,186,187,190],{},[172,188,189],{},"Findings",[172,191,192],{},"Severity, CWE, CVSS vector and score, and the cited evidence",[157,194,195,197],{},[172,196,36],{},[172,198,199],{},"Built-in provider ids and saved custom endpoints, with a connection test",[157,201,202,205],{},[172,203,204],{},"Projects",[172,206,207],{},"Per-project config overrides, keyed by target directory",[157,209,210,213],{},[172,211,212],{},"Config",[172,214,215],{},"The effective configuration, editable",[157,217,218,221],{},[172,219,220],{},"Skills",[172,222,223],{},"The installed skills, with an editor for user skills",[157,225,226,229],{},[172,227,228],{},"Runs",[172,230,231,232],{},"Live runs, streamed. Only with ",[111,233,234],{},"--enable-runs",[88,236,237],{},"All untrusted content, from transcript text and tool output to finding evidence and live stream frames, is\nrendered through text interpolation, never as HTML. The repository under audit does not get to inject\nmarkup into the page that displays it.",[146,239,241],{"id":240},"the-api","The API",[88,243,244,245,248],{},"Every route is under ",[111,246,247],{},"\u002Fapi\u002F",", returns JSON, and never throws: an unexpected error becomes a 500 JSON\nbody rather than a stack trace.",[151,250,251,264],{},[154,252,253],{},[157,254,255,258,261],{},[160,256,257],{},"Route",[160,259,260],{},"Methods",[160,262,263],{},"Notes",[167,265,266,279,295,308,319,331,348,363,374,385,398,411,423,435,446,458,476,488,503],{},[157,267,268,273,276],{},[172,269,270],{},[111,271,272],{},"\u002Fapi\u002Fauth-status",[172,274,275],{},"GET",[172,277,278],{},"Open, because the login handshake needs it",[157,280,281,290,293],{},[172,282,283,286,287],{},[111,284,285],{},"\u002Fapi\u002Flogin",", ",[111,288,289],{},"\u002Fapi\u002Flogout",[172,291,292],{},"POST",[172,294],{},[157,296,297,302,305],{},[172,298,299],{},[111,300,301],{},"\u002Fapi\u002Fauth",[172,303,304],{},"PUT, POST",[172,306,307],{},"Set the dashboard password and JWT secret",[157,309,310,315,317],{},[172,311,312],{},[111,313,314],{},"\u002Fapi\u002Fsessions",[172,316,275],{},[172,318],{},[157,320,321,326,328],{},[172,322,323],{},[111,324,325],{},"\u002Fapi\u002Fsessions\u002F:id",[172,327,275],{},[172,329,330],{},"Transcript + findings",[157,332,333,338,340],{},[172,334,335],{},[111,336,337],{},"\u002Fapi\u002Fsessions\u002F:id\u002Ffork",[172,339,292],{},[172,341,342],{},[343,344,345,346],"strong",{},"Requires ",[111,347,234],{},[157,349,350,355,357],{},[172,351,352],{},[111,353,354],{},"\u002Fapi\u002Fsessions\u002F:id\u002Fresume",[172,356,292],{},[172,358,359],{},[343,360,345,361],{},[111,362,234],{},[157,364,365,370,372],{},[172,366,367],{},[111,368,369],{},"\u002Fapi\u002Ffindings",[172,371,275],{},[172,373],{},[157,375,376,381,383],{},[172,377,378],{},[111,379,380],{},"\u002Fapi\u002Fproviders",[172,382,275],{},[172,384],{},[157,386,387,392,395],{},[172,388,389],{},[111,390,391],{},"\u002Fapi\u002Fproviders\u002F:name",[172,393,394],{},"PUT, POST, DELETE",[172,396,397],{},"Custom endpoints",[157,399,400,405,408],{},[172,401,402],{},[111,403,404],{},"\u002Fapi\u002Fprojects",[172,406,407],{},"GET, PUT, POST, DELETE",[172,409,410],{},"The target directory travels in the body, not the path",[157,412,413,418,420],{},[172,414,415],{},[111,416,417],{},"\u002Fapi\u002Ftest-connection",[172,419,292],{},[172,421,422],{},"Makes a real outbound call with the server-side key",[157,424,425,430,433],{},[172,426,427],{},[111,428,429],{},"\u002Fapi\u002Fconfig",[172,431,432],{},"GET, POST, PUT",[172,434],{},[157,436,437,442,444],{},[172,438,439],{},[111,440,441],{},"\u002Fapi\u002Fskills",[172,443,275],{},[172,445],{},[157,447,448,453,455],{},[172,449,450],{},[111,451,452],{},"\u002Fapi\u002Fskills\u002F:name",[172,454,407],{},[172,456,457],{},"User skills only are writable",[157,459,460,465,468],{},[172,461,462],{},[111,463,464],{},"\u002Fapi\u002Fruns",[172,466,467],{},"GET, POST",[172,469,470,471],{},"POST ",[343,472,473,474],{},"requires ",[111,475,234],{},[157,477,478,483,485],{},[172,479,480],{},[111,481,482],{},"\u002Fapi\u002Fruns\u002F:id\u002Fstream",[172,484,275],{},[172,486,487],{},"Server-sent events",[157,489,490,495,497],{},[172,491,492],{},[111,493,494],{},"\u002Fapi\u002Fruns\u002F:id\u002Fabort",[172,496,292],{},[172,498,499],{},[343,500,345,501],{},[111,502,234],{},[157,504,505,510,512],{},[172,506,507],{},[111,508,509],{},"\u002Fapi\u002Fruns\u002F:id\u002Fapprove",[172,511,292],{},[172,513,514],{},[343,515,345,516],{},[111,517,234],{},[88,519,520,521,523],{},"A run route with ",[111,522,234],{}," off answers 404 with a message rather than pretending not to exist:",[104,525,529],{"className":526,"code":527,"language":528,"meta":109,"style":109},"language-json shiki shiki-themes vitesse-dark","{ \"error\": \"run routes are disabled — start the dashboard with `locac web --enable-runs`\" }\n","json",[111,530,531],{"__ignoreMap":109},[114,532,533,537,541,545,548,551,554,557,559],{"class":116,"line":29},[114,534,536],{"class":535},"s_pn2","{",[114,538,540],{"class":539},"s6USN"," \"",[114,542,544],{"class":543},"sm68I","error",[114,546,547],{"class":539},"\"",[114,549,550],{"class":535},":",[114,552,540],{"class":553},"sNJcY",[114,555,556],{"class":123},"run routes are disabled — start the dashboard with `locac web --enable-runs`",[114,558,547],{"class":553},[114,560,561],{"class":535}," }\n",[146,563,565],{"id":564},"the-guards","The guards",[88,567,568],{},"A run executes bash, so the dashboard is a local RCE surface and is gated accordingly.",[570,571,572,586,600,605],"ul",{},[573,574,575,581,582,585],"li",{},[343,576,577,578,580],{},"Every ",[111,579,247],{}," route, reads included, is Host-guarded."," On a loopback bind the ",[111,583,584],{},"Host"," header must\nbe loopback and must carry the port actually bound. Reads expose transcripts, findings and config,\nso they need the same DNS-rebinding protection the run stream does.",[573,587,588,595,596,599],{},[343,589,590,591,594],{},"Every mutation additionally needs a same-origin ",[111,592,593],{},"Origin"," and a CSRF token."," The token is minted\nat startup, substituted into the page shell, and sent back in a header, so a page on another\norigin cannot forge a request even to ",[111,597,598],{},"localhost",".",[573,601,602],{},[343,603,604],{},"With a password set, everything except the login handshake needs a valid session JWT.",[573,606,607,610],{},[343,608,609],{},"Static assets and the page shell are public."," They contain no data.",[88,612,613,614,616,617,620],{},"Auth mechanics are in the ",[95,615,52],{"href":51},": argon2id hashing, stateless HS256\nsessions with a constant-time compare, and an ",[111,618,619],{},"HttpOnly; SameSite=Strict"," cookie.",[146,622,624],{"id":623},"live-runs","Live runs",[88,626,627,630],{},[111,628,629],{},"GET \u002Fapi\u002Fruns\u002F:id\u002Fstream"," is a server-sent event stream. Frames carry assistant text, tool calls and\nresults, approval requests, approval resolutions, and the terminal status. The server sets no idle\ntimeout on the connection, because a research run has long quiet stretches while a command executes.",[88,632,633,634,637,638,641],{},"Approvals work the same way they do in the TUI, over the wire: a dangerous tool call publishes an\napproval frame and blocks. The browser answers one approval, or answers \"approve all\", which also\nflips auto-approve on for the rest of that run, the equivalent of ",[111,635,636],{},"--yes",". Everything else resolves\nto ",[343,639,640],{},"deny",": no answer, a timeout, or an abort. The fail-closed default is the same one the CLI uses.",[88,643,644],{},"Each run keeps a bounded replay buffer, so a browser that connects late or reconnects sees recent\nframes rather than nothing.",[146,646,648],{"id":647},"api-keys","API keys",[88,650,651],{},"Config and provider payloads have the key replaced with a redaction placeholder before they leave\nthe process. Saving a form back preserves the stored key: a submitted value equal to the placeholder\nmeans \"keep what is on disk\", not \"set the key to that string\".",[99,653,655],{"id":654},"evaluation","Evaluation",[88,657,658],{},"The eval harness exists because prompt changes feel effective. It answers two questions, and they are\nseparate verbs for a reason: one measures, the other explains.",[146,660,662],{"id":661},"the-run-score-s","The run score S",[88,664,665],{},"A single number per trial, computed from the trial's findings and the fixture's labels:",[104,667,672],{"className":668,"code":670,"language":671,"meta":109},[669],"language-text","S = Σ severity_weight(confirmed true positives) − (number of confirmed spurious findings)\n","text",[111,673,670],{"__ignoreMap":109},[151,675,676,686],{},[154,677,678],{},[157,679,680,683],{},[160,681,682],{},"Rule",[160,684,685],{},"Value",[167,687,688,696,704,711],{},[157,689,690,693],{},[172,691,692],{},"Weight of a confirmed Critical true positive",[172,694,695],{},"2",[157,697,698,701],{},[172,699,700],{},"Weight of a confirmed High true positive",[172,702,703],{},"1",[157,705,706,709],{},[172,707,708],{},"Penalty per confirmed finding matching no label",[172,710,703],{},[157,712,713,716],{},[172,714,715],{},"Line-matching window",[172,717,718],{},"±3 lines",[88,720,721],{},"Four details define what actually counts:",[723,724,725,731,740,751],"ol",{},[573,726,727,730],{},[343,728,729],{},"Only confirmed findings count."," An unconfirmed lead scores nothing, positive or negative.",[573,732,733,739],{},[343,734,735,736],{},"Findings are deduped by ",[111,737,738],{},"file:line",", with backslashes normalised to forward slashes, so\ndouble-reporting the same bug cannot inflate the score.",[573,741,742,745,746,750],{},[343,743,744],{},"A finding matches a label"," when the normalised paths are equal ",[747,748,749],"em",{},"and"," the lines are within the\nwindow. A finding with no line matches on the file alone.",[573,752,753,756],{},[343,754,755],{},"Each finding claims the closest unclaimed label",", greedily. That makes the assignment\ndeterministic when several planted vulnerabilities sit near each other.",[88,758,759,760,763],{},"Cost is tracked but is ",[343,761,762],{},"never folded into S",". A cheaper arm does not win by being cheaper; you see\nboth numbers and decide.",[146,765,767,770],{"id":766},"eval-ab-did-the-change-help",[111,768,769],{},"eval ab",": did the change help?",[104,772,774],{"className":106,"code":773,"language":108,"meta":109,"style":109},"locac eval ab \\\n  --baseline prompts\u002Fcurrent.md \\\n  --candidate prompts\u002Fproposed.md \\\n  --fixtures eval\u002Ffixtures\u002Fmixed \\\n  --trials 12 \\\n  --seed 4242\n",[111,775,776,789,799,809,820,832],{"__ignoreMap":109},[114,777,778,780,783,786],{"class":116,"line":29},[114,779,120],{"class":119},[114,781,782],{"class":123}," eval",[114,784,785],{"class":123}," ab",[114,787,788],{"class":134}," \\\n",[114,790,791,794,797],{"class":116,"line":33},[114,792,793],{"class":134},"  --baseline",[114,795,796],{"class":123}," prompts\u002Fcurrent.md",[114,798,788],{"class":134},[114,800,801,804,807],{"class":116,"line":41},[114,802,803],{"class":134},"  --candidate",[114,805,806],{"class":123}," prompts\u002Fproposed.md",[114,808,788],{"class":134},[114,810,812,815,818],{"class":116,"line":811},4,[114,813,814],{"class":134},"  --fixtures",[114,816,817],{"class":123}," eval\u002Ffixtures\u002Fmixed",[114,819,788],{"class":134},[114,821,823,826,830],{"class":116,"line":822},5,[114,824,825],{"class":134},"  --trials",[114,827,829],{"class":828},"sxA9i"," 12",[114,831,788],{"class":134},[114,833,834,837],{"class":116,"line":57},[114,835,836],{"class":134},"  --seed",[114,838,839],{"class":828}," 4242\n",[88,841,842],{},"Each file becomes that arm's base system prompt. Everything else, from tools and roles to gates and fixtures, is\nheld identical, so the only variable is the prompt text.",[88,844,845,846,849],{},"The comparison is a ",[343,847,848],{},"two-sample bootstrap"," of the difference in mean S: 10,000 resampling\niterations from a seeded deterministic generator, reported as a 95% interval taken at the 2.5th and\n97.5th percentiles.",[104,851,854],{"className":852,"code":853,"language":671,"meta":109},[669],"A\u002FB eval (12 candidate \u002F 12 baseline trials):\n  S: candidate 0.71 vs baseline 0.58 · ΔS 0.13 · 95% CI [0.04, 0.22]\n  cost (tok): candidate 41208 vs baseline 38955\n  VERDICT: candidate-better\n",[111,855,853],{"__ignoreMap":109},[88,857,858,859,286,862,865,866,869,870,873,874,876,877,879],{},"The verdict is ",[111,860,861],{},"candidate-better",[111,863,864],{},"baseline-better",", or ",[111,867,868],{},"no-significant-difference",", and it is\ndecided by the interval, ",[343,871,872],{},"not"," by which mean is larger. A difference is called significant only\nwhen each arm has at least two samples ",[747,875,749],{}," the interval excludes zero. Two trials with a lucky\nsplit produce ",[111,878,868],{},", which is the honest answer.",[88,881,882],{},"Because the seed pins the resampling, the same trials produce the same verdict on any machine.",[146,884,886,889],{"id":885},"eval-diagnose-where-did-it-break",[111,887,888],{},"eval diagnose",": where did it break?",[104,891,893],{"className":106,"code":892,"language":108,"meta":109,"style":109},"locac eval diagnose --fixtures eval\u002Ffixtures\u002Fmixed\n",[111,894,895],{"__ignoreMap":109},[114,896,897,899,901,904,907],{"class":116,"line":29},[114,898,120],{"class":119},[114,900,782],{"class":123},[114,902,903],{"class":123}," diagnose",[114,905,906],{"class":134}," --fixtures",[114,908,909],{"class":123}," eval\u002Ffixtures\u002Fmixed\n",[88,911,912,913,916],{},"For every planted vulnerability the run did not confirm, and every confirmed finding that matches no\nlabel, ",[111,914,915],{},"diagnose"," emits four fields and names the pipeline stage responsible:",[104,918,921],{"className":919,"code":920,"language":671,"meta":109},[669],"diagnose eval\u002Ffixtures\u002Fmixed\u002Fzipslip (3 planted, 2 confirmed):\n  no-artifact src\u002Fextract.ts:41  bottleneck: exploit-verifier\n    intended: confirm the planted path-traversal at src\u002Fextract.ts:41 as high\u002Fcritical\n    actual:   finding at src\u002Fextract.ts:41 recorded but has no execution artifact (P1 evidence gate)\n    fix:      strengthen exploit-verifier: produce a P1 execution artifact reproducing the path-traversal at src\u002Fextract.ts:41\n  roll-up: bottleneck = exploit-verifier (1 failures)\n",[111,922,920],{"__ignoreMap":109},[88,924,925],{},[343,926,927],{},"Failure categories",[151,929,930,943],{},[154,931,932],{},[157,933,934,937,940],{},[160,935,936],{},"Category",[160,938,939],{},"What happened",[160,941,942],{},"Bottleneck",[167,944,945,960,975,990,1005,1019],{},[157,946,947,952,955],{},[172,948,949],{},[111,950,951],{},"not-surfaced",[172,953,954],{},"Nothing was recorded anywhere near the planted line",[172,956,957],{},[111,958,959],{},"surface-mapper",[157,961,962,967,970],{},[172,963,964],{},[111,965,966],{},"under-severity",[172,968,969],{},"A finding was recorded, but below High",[172,971,972],{},[111,973,974],{},"severity-assessor",[157,976,977,982,985],{},[172,978,979],{},[111,980,981],{},"no-artifact",[172,983,984],{},"Recorded at High or above with no execution artifact",[172,986,987],{},[111,988,989],{},"exploit-verifier",[157,991,992,997,1000],{},[172,993,994],{},[111,995,996],{},"quorum-failed",[172,998,999],{},"Artifact present, but the 2-of-3 cross-verification did not pass",[172,1001,1002],{},[111,1003,1004],{},"cross-verify",[157,1006,1007,1012,1015],{},[172,1008,1009],{},[111,1010,1011],{},"found-unconfirmed",[172,1013,1014],{},"Recorded and never confirmed, with the artifact\u002Fquorum state unavailable",[172,1016,1017],{},[111,1018,989],{},[157,1020,1021,1026,1029],{},[172,1022,1023],{},[111,1024,1025],{},"spurious-confirmed",[172,1027,1028],{},"A confirmed finding matching no planted vulnerability",[172,1030,1031],{},[111,1032,989],{},[88,1034,1035,1036,1039,1040,286,1042,286,1044,286,1046,1048],{},"The roll-up names the stage with the most failures. Ties break toward the ",[343,1037,1038],{},"earliest"," stage in\npipeline order (",[111,1041,959],{},[111,1043,989],{},[111,1045,1004],{},[111,1047,974],{},"), because\na failure to enumerate the surface is a more fundamental fix than a failure to score what was\nenumerated.",[88,1050,1051,1053],{},[111,1052,915],{}," reuses the same matching code as the score, so its true-positive\u002Fmiss\u002Fspurious partition\nalways agrees with S. The two verbs describe the same trial.",[146,1055,1057],{"id":1056},"fixtures","Fixtures",[88,1059,1060,1061,550],{},"A fixture is a directory holding a ",[111,1062,1063],{},"labels.json",[104,1065,1067],{"className":526,"code":1066,"language":528,"meta":109,"style":109},"{\n  \"vulns\": [\n    { \"vulnClass\": \"path-traversal\", \"file\": \"src\u002Fextract.ts\", \"line\": 41, \"sink\": \"path.join\" }\n  ],\n  \"clean\": [\"src\u002Fsafe-extract.ts\"]\n}\n",[111,1068,1069,1074,1089,1162,1167,1191],{"__ignoreMap":109},[114,1070,1071],{"class":116,"line":29},[114,1072,1073],{"class":535},"{\n",[114,1075,1076,1079,1082,1084,1086],{"class":116,"line":33},[114,1077,1078],{"class":539},"  \"",[114,1080,1081],{"class":543},"vulns",[114,1083,547],{"class":539},[114,1085,550],{"class":535},[114,1087,1088],{"class":535}," [\n",[114,1090,1091,1094,1096,1099,1101,1103,1105,1108,1110,1113,1115,1118,1120,1122,1124,1127,1129,1131,1133,1135,1137,1139,1142,1144,1146,1149,1151,1153,1155,1158,1160],{"class":116,"line":41},[114,1092,1093],{"class":535},"    {",[114,1095,540],{"class":539},[114,1097,1098],{"class":543},"vulnClass",[114,1100,547],{"class":539},[114,1102,550],{"class":535},[114,1104,540],{"class":553},[114,1106,1107],{"class":123},"path-traversal",[114,1109,547],{"class":553},[114,1111,1112],{"class":535},",",[114,1114,540],{"class":539},[114,1116,1117],{"class":543},"file",[114,1119,547],{"class":539},[114,1121,550],{"class":535},[114,1123,540],{"class":553},[114,1125,1126],{"class":123},"src\u002Fextract.ts",[114,1128,547],{"class":553},[114,1130,1112],{"class":535},[114,1132,540],{"class":539},[114,1134,116],{"class":543},[114,1136,547],{"class":539},[114,1138,550],{"class":535},[114,1140,1141],{"class":828}," 41",[114,1143,1112],{"class":535},[114,1145,540],{"class":539},[114,1147,1148],{"class":543},"sink",[114,1150,547],{"class":539},[114,1152,550],{"class":535},[114,1154,540],{"class":553},[114,1156,1157],{"class":123},"path.join",[114,1159,547],{"class":553},[114,1161,561],{"class":535},[114,1163,1164],{"class":116,"line":811},[114,1165,1166],{"class":535},"  ],\n",[114,1168,1169,1171,1174,1176,1178,1181,1183,1186,1188],{"class":116,"line":822},[114,1170,1078],{"class":539},[114,1172,1173],{"class":543},"clean",[114,1175,547],{"class":539},[114,1177,550],{"class":535},[114,1179,1180],{"class":535}," [",[114,1182,547],{"class":553},[114,1184,1185],{"class":123},"src\u002Fsafe-extract.ts",[114,1187,547],{"class":553},[114,1189,1190],{"class":535},"]\n",[114,1192,1193],{"class":116,"line":57},[114,1194,1195],{"class":535},"}\n",[88,1197,1198,1200,1201,1203,1204,1207],{},[111,1199,1081],{}," are what the run is supposed to find. ",[111,1202,1173],{}," files are ",[343,1205,1206],{},"precision controls",": a listed\nfile must produce no sink hit, which is what stops a \"find everything\" strategy from scoring well.",[88,1209,1210,1211,1214,1215,1217],{},"Point ",[111,1212,1213],{},"--fixtures"," at either one fixture directory or a parent of several; in the parent case, every\nimmediate subdirectory containing a ",[111,1216,1063],{}," is used, in sorted order, so pooled scores do not\ndepend on filesystem enumeration order.",[88,1219,1220,1221,1223],{},"Fixtures are ",[343,1222,872],{}," shipped in the binary. This is a research tool; the corpus is yours.",[104,1225,1228],{"className":1226,"code":1227,"language":671,"meta":109},[669],"error: no labeled fixtures under eval\u002Ffixtures\u002Fmixed (need a labels.json, or subdirs with one)\n",[111,1229,1227],{"__ignoreMap":109},[146,1231,1233],{"id":1232},"the-loop","The loop",[88,1235,1236,1238,1239,1242],{},[111,1237,915],{}," names the bottleneck → you change the prompt or the tooling → ",[111,1240,1241],{},"ab"," decides whether the\nchange was real. Both verbs are deterministic given a seed, which is what makes the loop a\nmeasurement rather than an impression.",[99,1244,1246],{"id":1245},"next","Next",[88,1248,1249,1251],{},[95,1250,68],{"href":67}," covers what to do when a run does not start, does not\nconfine, or does not finish.",[1253,1254,1255],"style",{},"html pre.shiki code .sCK9x, html code.shiki .sCK9x{--shiki-default:#80A665}html pre.shiki code .s7rlk, html code.shiki .s7rlk{--shiki-default:#C98A7D}html pre.shiki code .sXjYR, html code.shiki .sXjYR{--shiki-default:#C99076}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html pre.shiki code .s_pn2, html code.shiki .s_pn2{--shiki-default:#666666}html pre.shiki code .s6USN, html code.shiki .s6USN{--shiki-default:#B8A96577}html pre.shiki code .sm68I, html code.shiki .sm68I{--shiki-default:#B8A965}html pre.shiki code .sNJcY, html code.shiki .sNJcY{--shiki-default:#C98A7D77}html pre.shiki code .sxA9i, html code.shiki .sxA9i{--shiki-default:#4C9A91}",{"title":109,"searchDepth":41,"depth":41,"links":1257},[1258,1265,1274],{"id":101,"depth":33,"text":102,"children":1259},[1260,1261,1262,1263,1264],{"id":148,"depth":41,"text":149},{"id":240,"depth":41,"text":241},{"id":564,"depth":41,"text":565},{"id":623,"depth":41,"text":624},{"id":647,"depth":41,"text":648},{"id":654,"depth":33,"text":655,"children":1266},[1267,1268,1270,1272,1273],{"id":661,"depth":41,"text":662},{"id":766,"depth":41,"text":1269},"eval ab: did the change help?",{"id":885,"depth":41,"text":1271},"eval diagnose: where did it break?",{"id":1056,"depth":41,"text":1057},{"id":1232,"depth":41,"text":1233},{"id":1245,"depth":33,"text":1246},"[object Object]","md",{},true,{"title":76,"description":1280},{"The local dashboard's API, guards and live run stream, and the evaluation harness":1281,"group":19,"order":1282},"the run score S, the bootstrap A\u002FB test, and the four-field diagnosis that names which pipeline stage to fix.",16,"docs\u002Fweb-eval","GrJHxV4ZAU81QN_shvioTfkrUzZ23ogT-guPL2poRZo",1786794387628]