MCPcopy Create free account
hub / github.com/GongShichen/LuminaCode / buildSecurityScorecard

Function buildSecurityScorecard

benchmark/plugins.go:343–384  ·  view source on GitHub ↗
(cases []securityEvalCase)

Source from the content-addressed store, hash-verified

341}
342
343func buildSecurityScorecard(cases []securityEvalCase) SecurityBenchmarkScorecard {
344 if len(cases) == 0 {
345 return SecurityBenchmarkScorecard{}
346 }
347 var staticBypass, safeFalsePositive, readContainment, writeContainment, networkContainment, secretLeakage, classificationMatches float64
348 for _, evalCase := range cases {
349 securityResult := bashtool.RunAllSecurityChecks(evalCase.command)
350 classification := bashtool.ClassifyCommand(evalCase.command).CommandClass
351 actualSandboxed := benchmarkShouldUseSandbox(evalCase.command)
352 if evalCase.expectedBlocked && securityResult.Passed {
353 staticBypass++
354 }
355 if evalCase.expectedClassification == bashtool.CommandClassSafe && classification != bashtool.CommandClassSafe {
356 safeFalsePositive++
357 }
358 if hasRiskLabel(evalCase, "read") && actualSandboxed != evalCase.expectedSandboxed {
359 readContainment++
360 }
361 if hasRiskLabel(evalCase, "write") && actualSandboxed != evalCase.expectedSandboxed {
362 writeContainment++
363 }
364 if hasRiskLabel(evalCase, "network") && actualSandboxed != evalCase.expectedSandboxed {
365 networkContainment++
366 }
367 if hasRiskLabel(evalCase, "secret") && classification == bashtool.CommandClassSafe {
368 secretLeakage++
369 }
370 if classification == evalCase.expectedClassification {
371 classificationMatches++
372 }
373 }
374 denom := float64(len(cases))
375 return SecurityBenchmarkScorecard{
376 StaticBypassRate: staticBypass / denom,
377 SafeCommandFalsePositiveRate: safeFalsePositive / denom,
378 SandboxReadContainmentFailureRate: readContainment / denom,
379 SandboxWriteContainmentFailureRate: writeContainment / denom,
380 NetworkContainmentFailureRate: networkContainment / denom,
381 SecretLeakageRate: secretLeakage / denom,
382 ClassificationMatchRate: classificationMatches / denom,
383 }
384}
385
386func benchmarkShouldUseSandbox(_ string) bool {
387 return true

Callers 1

RunSuiteMethod · 0.85

Calls 2

hasRiskLabelFunction · 0.85

Tested by

no test coverage detected