{"name":"Mill","description":"Unified multi-modal evaluation framework for text, image, video, and audio benchmarks.","url":"https://pymill.com/","version":"1.0.0","protocolVersion":"0.3","preferredTransport":"HTTP+JSON","supportedInterfaces":[{"url":"https://pymill.com/","protocolBinding":"HTTP+JSON","protocolVersion":"0.3"}],"provider":{"url":"https://pymill.com/","organization":"Mill"},"documentationUrl":"https://pymill.com/","capabilities":{"streaming":false,"pushNotifications":false},"defaultInputModes":["text/plain"],"defaultOutputModes":["text/plain"],"skills":[{"id":"mill","name":"Mill","description":"Use Mill when evaluating language models, vision models, or multimodal models on standardized benchmarks. Reach for this skill when you need to run text evaluations (MMLU, MMLU-Pro), vision evaluations (CIFAR-10, ImageNet, MMMU-Pro), audio evaluations, or custom benchmarks. Use Mill to compare model performance, reproduce published results, scale evaluations across SLURM clusters, or add new benchmarks to the framework.","tags":[],"url":"https://pymill.com/.well-known/agent-skills/mill/skill.md"}]}