[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"skill-sourcegraph-status":3,"mdc--ureg20-key":29,"related-org-sourcegraph-status":414,"related-repo-sourcegraph-status":531},{"slug":4,"name":4,"fn":5,"description":6,"org":7,"tags":11,"stars":19,"repoUrl":20,"updatedAt":21,"license":22,"forks":23,"topics":24,"repo":25,"sourceUrl":27,"mdContent":28},"status","monitor benchmark execution and task status","Monitor active runs, check task completion status, and watch benchmark execution progress.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},"sourcegraph","Sourcegraph","https:\u002F\u002Fpexgzepcugksgbtrxkhf.supabase.co\u002Fstorage\u002Fv1\u002Fobject\u002Fpublic\u002Forg-logos\u002Fsourcegraph.png",[12,16],{"name":13,"slug":14,"type":15},"Benchmarking","benchmarking","tag",{"name":17,"slug":18,"type":15},"Monitoring","monitoring",31,"https:\u002F\u002Fgithub.com\u002Fsourcegraph\u002FCodeScaleBench","2026-07-16T06:02:37.137106",null,4,[],{"repoUrl":20,"stars":19,"forks":23,"topics":26,"description":22},[],"https:\u002F\u002Fgithub.com\u002Fsourcegraph\u002FCodeScaleBench\u002Ftree\u002FHEAD\u002Fskills\u002Fstatus","---\nname: status\ndescription: Monitor active runs, check task completion status, and watch benchmark execution progress.\n---\n\n# Skill: Status & Monitoring\n\n## Scope\n\nUse this skill when the user asks to:\n- Check the status of active or completed runs\n- Monitor benchmark execution progress in real-time\n- List task completion rates by suite and configuration\n- Watch for failures or bottlenecks during runs\n- Analyze run fingerprints and anomalies\n\n## Canonical Commands\n\n```bash\n# Overall staging status\npython3 scripts\u002Fanalysis\u002Faggregate_status.py --staging\n\n# Watch active runs (refresh every 10s)\npython3 scripts\u002Fanalysis\u002Faggregate_status.py --watch\n\n# Status by suite\npython3 scripts\u002Fanalysis\u002Faggregate_status.py --staging --suite csb_sdlc_debug\n\n# Status fingerprints (failure patterns)\npython3 scripts\u002Fanalysis\u002Fstatus_fingerprints.py runs\u002Fstaging\u002Frun_dir\u002Fresult.json\n```\n\n## Key Metrics\n\n- **Completion rate** — tasks passed \u002F total tasks\n- **Model success rate** — tasks where agent succeeded\n- **Verification success** — tasks that passed post-run validation\n- **Timeouts & OOM** — infrastructure failure rates\n- **Anomalies** — tasks with unexpected patterns (fingerprint match)\n\n## Monitoring Patterns\n\n- **Real-time watch**: `--watch` mode updates every 10 seconds\n- **Batch status**: Full scan across runs\u002Fstaging with counts by suite\n- **Fingerprint triage**: Use `status_fingerprints` to bucket failures (timeout, OOM, auth, etc.)\n\n## Related Skills\n\n- `\u002Frun` — launch and manage runs\n- `\u002Faudit` — deeper validation and health checks\n- `\u002Ftriage` — investigate specific task failures\n",{"data":30,"body":31},{"name":4,"description":6},{"type":32,"children":33},"root",[34,43,50,56,86,92,251,257,311,317,366,372,408],{"type":35,"tag":36,"props":37,"children":39},"element","h1",{"id":38},"skill-status-monitoring",[40],{"type":41,"value":42},"text","Skill: Status & Monitoring",{"type":35,"tag":44,"props":45,"children":47},"h2",{"id":46},"scope",[48],{"type":41,"value":49},"Scope",{"type":35,"tag":51,"props":52,"children":53},"p",{},[54],{"type":41,"value":55},"Use this skill when the user asks to:",{"type":35,"tag":57,"props":58,"children":59},"ul",{},[60,66,71,76,81],{"type":35,"tag":61,"props":62,"children":63},"li",{},[64],{"type":41,"value":65},"Check the status of active or completed runs",{"type":35,"tag":61,"props":67,"children":68},{},[69],{"type":41,"value":70},"Monitor benchmark execution progress in real-time",{"type":35,"tag":61,"props":72,"children":73},{},[74],{"type":41,"value":75},"List task completion rates by suite and configuration",{"type":35,"tag":61,"props":77,"children":78},{},[79],{"type":41,"value":80},"Watch for failures or bottlenecks during runs",{"type":35,"tag":61,"props":82,"children":83},{},[84],{"type":41,"value":85},"Analyze run fingerprints and anomalies",{"type":35,"tag":44,"props":87,"children":89},{"id":88},"canonical-commands",[90],{"type":41,"value":91},"Canonical Commands",{"type":35,"tag":93,"props":94,"children":99},"pre",{"className":95,"code":96,"language":97,"meta":98,"style":98},"language-bash shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","# Overall staging status\npython3 scripts\u002Fanalysis\u002Faggregate_status.py --staging\n\n# Watch active runs (refresh every 10s)\npython3 scripts\u002Fanalysis\u002Faggregate_status.py --watch\n\n# Status by suite\npython3 scripts\u002Fanalysis\u002Faggregate_status.py --staging --suite csb_sdlc_debug\n\n# Status fingerprints (failure patterns)\npython3 scripts\u002Fanalysis\u002Fstatus_fingerprints.py runs\u002Fstaging\u002Frun_dir\u002Fresult.json\n","bash","",[100],{"type":35,"tag":101,"props":102,"children":103},"code",{"__ignoreMap":98},[104,116,137,147,155,172,180,189,216,224,233],{"type":35,"tag":105,"props":106,"children":109},"span",{"class":107,"line":108},"line",1,[110],{"type":35,"tag":105,"props":111,"children":113},{"style":112},"--shiki-light:#90A4AE;--shiki-light-font-style:italic;--shiki-default:#546E7A;--shiki-default-font-style:italic;--shiki-dark:#676E95;--shiki-dark-font-style:italic",[114],{"type":41,"value":115},"# Overall staging status\n",{"type":35,"tag":105,"props":117,"children":119},{"class":107,"line":118},2,[120,126,132],{"type":35,"tag":105,"props":121,"children":123},{"style":122},"--shiki-light:#E2931D;--shiki-default:#FFCB6B;--shiki-dark:#FFCB6B",[124],{"type":41,"value":125},"python3",{"type":35,"tag":105,"props":127,"children":129},{"style":128},"--shiki-light:#91B859;--shiki-default:#C3E88D;--shiki-dark:#C3E88D",[130],{"type":41,"value":131}," scripts\u002Fanalysis\u002Faggregate_status.py",{"type":35,"tag":105,"props":133,"children":134},{"style":128},[135],{"type":41,"value":136}," --staging\n",{"type":35,"tag":105,"props":138,"children":140},{"class":107,"line":139},3,[141],{"type":35,"tag":105,"props":142,"children":144},{"emptyLinePlaceholder":143},true,[145],{"type":41,"value":146},"\n",{"type":35,"tag":105,"props":148,"children":149},{"class":107,"line":23},[150],{"type":35,"tag":105,"props":151,"children":152},{"style":112},[153],{"type":41,"value":154},"# Watch active runs (refresh every 10s)\n",{"type":35,"tag":105,"props":156,"children":158},{"class":107,"line":157},5,[159,163,167],{"type":35,"tag":105,"props":160,"children":161},{"style":122},[162],{"type":41,"value":125},{"type":35,"tag":105,"props":164,"children":165},{"style":128},[166],{"type":41,"value":131},{"type":35,"tag":105,"props":168,"children":169},{"style":128},[170],{"type":41,"value":171}," --watch\n",{"type":35,"tag":105,"props":173,"children":175},{"class":107,"line":174},6,[176],{"type":35,"tag":105,"props":177,"children":178},{"emptyLinePlaceholder":143},[179],{"type":41,"value":146},{"type":35,"tag":105,"props":181,"children":183},{"class":107,"line":182},7,[184],{"type":35,"tag":105,"props":185,"children":186},{"style":112},[187],{"type":41,"value":188},"# Status by suite\n",{"type":35,"tag":105,"props":190,"children":192},{"class":107,"line":191},8,[193,197,201,206,211],{"type":35,"tag":105,"props":194,"children":195},{"style":122},[196],{"type":41,"value":125},{"type":35,"tag":105,"props":198,"children":199},{"style":128},[200],{"type":41,"value":131},{"type":35,"tag":105,"props":202,"children":203},{"style":128},[204],{"type":41,"value":205}," --staging",{"type":35,"tag":105,"props":207,"children":208},{"style":128},[209],{"type":41,"value":210}," --suite",{"type":35,"tag":105,"props":212,"children":213},{"style":128},[214],{"type":41,"value":215}," csb_sdlc_debug\n",{"type":35,"tag":105,"props":217,"children":219},{"class":107,"line":218},9,[220],{"type":35,"tag":105,"props":221,"children":222},{"emptyLinePlaceholder":143},[223],{"type":41,"value":146},{"type":35,"tag":105,"props":225,"children":227},{"class":107,"line":226},10,[228],{"type":35,"tag":105,"props":229,"children":230},{"style":112},[231],{"type":41,"value":232},"# Status fingerprints (failure patterns)\n",{"type":35,"tag":105,"props":234,"children":236},{"class":107,"line":235},11,[237,241,246],{"type":35,"tag":105,"props":238,"children":239},{"style":122},[240],{"type":41,"value":125},{"type":35,"tag":105,"props":242,"children":243},{"style":128},[244],{"type":41,"value":245}," scripts\u002Fanalysis\u002Fstatus_fingerprints.py",{"type":35,"tag":105,"props":247,"children":248},{"style":128},[249],{"type":41,"value":250}," runs\u002Fstaging\u002Frun_dir\u002Fresult.json\n",{"type":35,"tag":44,"props":252,"children":254},{"id":253},"key-metrics",[255],{"type":41,"value":256},"Key Metrics",{"type":35,"tag":57,"props":258,"children":259},{},[260,271,281,291,301],{"type":35,"tag":61,"props":261,"children":262},{},[263,269],{"type":35,"tag":264,"props":265,"children":266},"strong",{},[267],{"type":41,"value":268},"Completion rate",{"type":41,"value":270}," — tasks passed \u002F total tasks",{"type":35,"tag":61,"props":272,"children":273},{},[274,279],{"type":35,"tag":264,"props":275,"children":276},{},[277],{"type":41,"value":278},"Model success rate",{"type":41,"value":280}," — tasks where agent succeeded",{"type":35,"tag":61,"props":282,"children":283},{},[284,289],{"type":35,"tag":264,"props":285,"children":286},{},[287],{"type":41,"value":288},"Verification success",{"type":41,"value":290}," — tasks that passed post-run validation",{"type":35,"tag":61,"props":292,"children":293},{},[294,299],{"type":35,"tag":264,"props":295,"children":296},{},[297],{"type":41,"value":298},"Timeouts & OOM",{"type":41,"value":300}," — infrastructure failure rates",{"type":35,"tag":61,"props":302,"children":303},{},[304,309],{"type":35,"tag":264,"props":305,"children":306},{},[307],{"type":41,"value":308},"Anomalies",{"type":41,"value":310}," — tasks with unexpected patterns (fingerprint match)",{"type":35,"tag":44,"props":312,"children":314},{"id":313},"monitoring-patterns",[315],{"type":41,"value":316},"Monitoring Patterns",{"type":35,"tag":57,"props":318,"children":319},{},[320,338,348],{"type":35,"tag":61,"props":321,"children":322},{},[323,328,330,336],{"type":35,"tag":264,"props":324,"children":325},{},[326],{"type":41,"value":327},"Real-time watch",{"type":41,"value":329},": ",{"type":35,"tag":101,"props":331,"children":333},{"className":332},[],[334],{"type":41,"value":335},"--watch",{"type":41,"value":337}," mode updates every 10 seconds",{"type":35,"tag":61,"props":339,"children":340},{},[341,346],{"type":35,"tag":264,"props":342,"children":343},{},[344],{"type":41,"value":345},"Batch status",{"type":41,"value":347},": Full scan across runs\u002Fstaging with counts by suite",{"type":35,"tag":61,"props":349,"children":350},{},[351,356,358,364],{"type":35,"tag":264,"props":352,"children":353},{},[354],{"type":41,"value":355},"Fingerprint triage",{"type":41,"value":357},": Use ",{"type":35,"tag":101,"props":359,"children":361},{"className":360},[],[362],{"type":41,"value":363},"status_fingerprints",{"type":41,"value":365}," to bucket failures (timeout, OOM, auth, etc.)",{"type":35,"tag":44,"props":367,"children":369},{"id":368},"related-skills",[370],{"type":41,"value":371},"Related Skills",{"type":35,"tag":57,"props":373,"children":374},{},[375,386,397],{"type":35,"tag":61,"props":376,"children":377},{},[378,384],{"type":35,"tag":101,"props":379,"children":381},{"className":380},[],[382],{"type":41,"value":383},"\u002Frun",{"type":41,"value":385}," — launch and manage runs",{"type":35,"tag":61,"props":387,"children":388},{},[389,395],{"type":35,"tag":101,"props":390,"children":392},{"className":391},[],[393],{"type":41,"value":394},"\u002Faudit",{"type":41,"value":396}," — deeper validation and health checks",{"type":35,"tag":61,"props":398,"children":399},{},[400,406],{"type":35,"tag":101,"props":401,"children":403},{"className":402},[],[404],{"type":41,"value":405},"\u002Ftriage",{"type":41,"value":407}," — investigate specific task failures",{"type":35,"tag":409,"props":410,"children":411},"style",{},[412],{"type":41,"value":413},"html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"items":415,"total":218},[416,429,446,461,475,489,503,513,518],{"slug":417,"name":417,"fn":418,"description":419,"org":420,"tags":421,"stars":19,"repoUrl":20,"updatedAt":428},"audit","audit repository health and benchmark integrity","Run repo health checks, validate benchmark tasks, and audit run integrity.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[422,424,425],{"name":423,"slug":417,"type":15},"Audit",{"name":13,"slug":14,"type":15},{"name":426,"slug":427,"type":15},"QA","qa","2026-07-17T06:07:07.220218",{"slug":430,"name":430,"fn":431,"description":432,"org":433,"tags":434,"stars":19,"repoUrl":20,"updatedAt":445},"evaluate","score traces and evaluate benchmark results","Extract metrics, score traces, and evaluate benchmark task results.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[435,436,439,442],{"name":13,"slug":14,"type":15},{"name":437,"slug":438,"type":15},"Data Analysis","data-analysis",{"name":440,"slug":441,"type":15},"Engineering","engineering",{"name":443,"slug":444,"type":15},"Evals","evals","2026-07-16T06:04:41.910821",{"slug":447,"name":447,"fn":448,"description":449,"org":450,"tags":451,"stars":19,"repoUrl":20,"updatedAt":460},"infra","check infrastructure and system dependencies","Check infrastructure readiness, manage MCP tools, and audit system dependencies.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[452,453,454,457],{"name":423,"slug":417,"type":15},{"name":440,"slug":441,"type":15},{"name":455,"slug":456,"type":15},"Infrastructure","infrastructure",{"name":458,"slug":459,"type":15},"MCP","mcp","2026-07-16T06:04:41.188274",{"slug":462,"name":462,"fn":463,"description":464,"org":465,"tags":466,"stars":19,"repoUrl":20,"updatedAt":474},"next","plan benchmarking and coverage tasks","Plan upcoming work, analyze coverage gaps, and recommend next steps for benchmarking.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[467,468,471],{"name":13,"slug":14,"type":15},{"name":469,"slug":470,"type":15},"Planning","planning",{"name":472,"slug":473,"type":15},"Strategy","strategy","2026-07-17T06:06:57.69018",{"slug":476,"name":476,"fn":477,"description":478,"org":479,"tags":480,"stars":19,"repoUrl":20,"updatedAt":488},"report","generate CodeScaleBench evaluation reports","Generate evaluation reports, analyze run costs, and compare configurations.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[481,484,485],{"name":482,"slug":483,"type":15},"Analytics","analytics",{"name":13,"slug":14,"type":15},{"name":486,"slug":487,"type":15},"Reporting","reporting","2026-07-16T06:02:36.809556",{"slug":490,"name":490,"fn":491,"description":492,"org":493,"tags":494,"stars":19,"repoUrl":20,"updatedAt":502},"run","manage CodeScaleBench benchmark runs","Launch and manage CodeScaleBench benchmark runs with paired-run guardrails, quick reruns, and execution orchestration.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[495,498,499],{"name":496,"slug":497,"type":15},"Automation","automation",{"name":13,"slug":14,"type":15},{"name":500,"slug":501,"type":15},"Testing","testing","2026-07-16T06:04:40.848817",{"slug":504,"name":504,"fn":505,"description":506,"org":507,"tags":508,"stars":19,"repoUrl":20,"updatedAt":512},"scaffold","create and validate benchmark tasks","Create, mine, and validate new benchmark tasks and task suites.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[509,510,511],{"name":13,"slug":14,"type":15},{"name":440,"slug":441,"type":15},{"name":500,"slug":501,"type":15},"2026-07-16T06:04:41.525889",{"slug":4,"name":4,"fn":5,"description":6,"org":514,"tags":515,"stars":19,"repoUrl":20,"updatedAt":21},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[516,517],{"name":13,"slug":14,"type":15},{"name":17,"slug":18,"type":15},{"slug":519,"name":519,"fn":520,"description":521,"org":522,"tags":523,"stars":19,"repoUrl":20,"updatedAt":530},"triage","triage and analyze failed benchmark tasks","Investigate and triage failed benchmark tasks, analyze root causes, and plan reruns.",{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[524,525,528],{"name":13,"slug":14,"type":15},{"name":526,"slug":527,"type":15},"Debugging","debugging",{"name":529,"slug":519,"type":15},"Triage","2026-07-16T06:02:37.472337",{"items":532,"total":218},[533,539,546,553,559,565,571],{"slug":417,"name":417,"fn":418,"description":419,"org":534,"tags":535,"stars":19,"repoUrl":20,"updatedAt":428},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[536,537,538],{"name":423,"slug":417,"type":15},{"name":13,"slug":14,"type":15},{"name":426,"slug":427,"type":15},{"slug":430,"name":430,"fn":431,"description":432,"org":540,"tags":541,"stars":19,"repoUrl":20,"updatedAt":445},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[542,543,544,545],{"name":13,"slug":14,"type":15},{"name":437,"slug":438,"type":15},{"name":440,"slug":441,"type":15},{"name":443,"slug":444,"type":15},{"slug":447,"name":447,"fn":448,"description":449,"org":547,"tags":548,"stars":19,"repoUrl":20,"updatedAt":460},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[549,550,551,552],{"name":423,"slug":417,"type":15},{"name":440,"slug":441,"type":15},{"name":455,"slug":456,"type":15},{"name":458,"slug":459,"type":15},{"slug":462,"name":462,"fn":463,"description":464,"org":554,"tags":555,"stars":19,"repoUrl":20,"updatedAt":474},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[556,557,558],{"name":13,"slug":14,"type":15},{"name":469,"slug":470,"type":15},{"name":472,"slug":473,"type":15},{"slug":476,"name":476,"fn":477,"description":478,"org":560,"tags":561,"stars":19,"repoUrl":20,"updatedAt":488},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[562,563,564],{"name":482,"slug":483,"type":15},{"name":13,"slug":14,"type":15},{"name":486,"slug":487,"type":15},{"slug":490,"name":490,"fn":491,"description":492,"org":566,"tags":567,"stars":19,"repoUrl":20,"updatedAt":502},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[568,569,570],{"name":496,"slug":497,"type":15},{"name":13,"slug":14,"type":15},{"name":500,"slug":501,"type":15},{"slug":504,"name":504,"fn":505,"description":506,"org":572,"tags":573,"stars":19,"repoUrl":20,"updatedAt":512},{"slug":8,"name":9,"logoUrl":10,"githubOrg":8},[574,575,576],{"name":13,"slug":14,"type":15},{"name":440,"slug":441,"type":15},{"name":500,"slug":501,"type":15}]