{"$schema":"https://static.modelcontextprotocol.io/schemas/2025-06-18/server.schema.json","name":"io.github.aniruddhaadak80.crucible","description":"Crucible is a benchmark forge. Paste a model failure, forge it into a scored evaluation task, re-run the deterministic grader across recorded transcripts, and publish a suite anyone can reproduce.","version":"0.1.0","websiteUrl":"https://crucibleforge.vercel.app","repository":{"url":"https://github.com/aniruddhaadak80/crucible","source":"github"},"capabilities":{"tools":{"listChanged":false}},"servers":{"crucible":{"type":"http","url":"https://crucibleforge.vercel.app/api/mcp","headers":{}}},"tools":[{"name":"list_tasks","title":"List forged tasks","description":"Every benchmark task owned by the calling session, newest first, each with its engine verdict. Retired tasks are excluded unless includeRetired is true.","inputSchema":{"type":"object","properties":{"limit":{"type":"integer","minimum":1,"maximum":50,"default":20},"before":{"type":"string","description":"ISO timestamp cursor for the next page"},"status":{"type":"string","enum":["forging","poured","retired"]},"includeRetired":{"type":"boolean","default":false}},"additionalProperties":false},"annotations":{"readOnlyHint":true,"destructiveHint":false,"idempotentHint":true}},{"name":"get_task","title":"Read one task","description":"Full record plus the engine verdict, every transcript grade with per-assertion evidence, and the head of the audit chain.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":true,"destructiveHint":false,"idempotentHint":true}},{"name":"grade_task","title":"Run the engine and seal the verdict","description":"Re-runs crucible-grade over a task, consulting the live Hugging Face Hub for the target model's revision and gating. Appends a grade event to the chain and returns the new seal. Deterministic: identical input yields an identical score.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":false,"destructiveHint":false,"idempotentHint":true}},{"name":"rank_lineup","title":"Rank the recorded lineup","description":"Grades every recorded transcript of a task with the shared grader and ranks the models by score. Reports the population sigma, so a task that cannot separate models is visible as such.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":true,"destructiveHint":false,"idempotentHint":true}},{"name":"verify_integrity","title":"Replay the audit chain","description":"Recomputes every seal in a task's SHA-384 chain from genesis and reports the first broken link, if any.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":true,"destructiveHint":false,"idempotentHint":true}},{"name":"live_signals","title":"Read live model and preprint signals","description":"Live Hugging Face Hub facts for the benchmark lineup, including whether each model's weights are gated and whether a pinnable revision exists, plus the newest arXiv capability-evaluation papers.","inputSchema":{"type":"object","properties":{"paperLimit":{"type":"integer","minimum":1,"maximum":20,"default":6}},"additionalProperties":false},"annotations":{"readOnlyHint":true,"destructiveHint":false,"idempotentHint":true}},{"name":"export_bundle","title":"Generate the Kaggle Benchmarks bundle","description":"Returns a runnable kaggle_benchmarks task file, a self-contained Python grader, a self-check and the CLI commands to push and run it. Grading is implemented in the bundle itself, so the exported file grades identically.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":true,"destructiveHint":false,"idempotentHint":true}},{"name":"forge_task","title":"Forge a new benchmark task","description":"Creates a task from a failure mode, a prompt, assertions and recorded transcripts. Returns the persisted record with its verdict and the first audit seal. Supplying the same idempotencyKey twice returns the first result and performs no second mutation.","inputSchema":{"type":"object","properties":{"name":{"type":"string","minLength":3,"maxLength":120},"failureMode":{"type":"string","minLength":10,"maxLength":600},"prompt":{"type":"string","minLength":10,"maxLength":8000},"sealedFixtures":{"type":"array","items":{"type":"string"},"maxItems":40},"assertions":{"type":"array","items":{"type":"object"},"maxItems":40},"transcripts":{"type":"array","items":{"type":"object"},"maxItems":60},"seed":{"type":["integer","null"]},"targetModel":{"type":["string","null"]},"targetTemp":{"type":["number","null"],"minimum":0,"maximum":2},"targetRevision":{"type":["string","null"]},"tokenBudget":{"type":"integer","minimum":0,"maximum":10000000},"idempotencyKey":{"type":"string","maxLength":200}},"required":["name","failureMode","prompt"],"additionalProperties":false},"annotations":{"readOnlyHint":false,"destructiveHint":false,"idempotentHint":true}},{"name":"revise_task","title":"Revise a task","description":"Patches any subset of a task's fields through the same service the UI uses. Records the pre-change engine score on the audit event. Supplying the same idempotencyKey twice returns the first result and performs no second mutation.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1},"name":{"type":"string","minLength":3,"maxLength":120},"failureMode":{"type":"string","minLength":10,"maxLength":600},"prompt":{"type":"string","minLength":10,"maxLength":8000},"sealedFixtures":{"type":"array","items":{"type":"string"},"maxItems":40},"assertions":{"type":"array","items":{"type":"object"},"maxItems":40},"transcripts":{"type":"array","items":{"type":"object"},"maxItems":60},"seed":{"type":["integer","null"]},"targetModel":{"type":["string","null"]},"targetTemp":{"type":["number","null"],"minimum":0,"maximum":2},"targetRevision":{"type":["string","null"]},"tokenBudget":{"type":"integer","minimum":0,"maximum":10000000},"status":{"type":"string","enum":["forging","poured","retired"]},"idempotencyKey":{"type":"string","maxLength":200}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":false,"destructiveHint":false,"idempotentHint":true}},{"name":"record_decision","title":"Record an adopt/iterate/discard decision","description":"Files a decision with a note. The engine score at the moment of the decision is stored on the record and sealed into the chain. Supplying the same idempotencyKey twice returns the first result and performs no second mutation.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1},"verdict":{"type":"string","enum":["adopt","iterate","discard"]},"note":{"type":"string","minLength":3,"maxLength":600},"idempotencyKey":{"type":"string","maxLength":200}},"required":["id","verdict","note"],"additionalProperties":false},"annotations":{"readOnlyHint":false,"destructiveHint":false,"idempotentHint":true}},{"name":"retire_task","title":"Retire a task as a tombstone","description":"Soft-deletes a task. The row and its audit events are retained so the chain still replays, and the replay result is returned as proof. Supplying the same idempotencyKey twice returns the first result and performs no second mutation.","inputSchema":{"type":"object","properties":{"id":{"type":"string","minLength":1},"idempotencyKey":{"type":"string","maxLength":200}},"required":["id"],"additionalProperties":false},"annotations":{"readOnlyHint":false,"destructiveHint":true,"idempotentHint":true}}],"protocolVersion":"2025-06-18","engine":{"version":"crucible-grade-v1.0.0","grader":"crucible-grader-v1.0.0"}}