{"version":"mdrss-hashtag-feed/1","tag":"evals","urls":{"html":"https://mdrss.com/feeds/evals","rss":"https://mdrss.com/feeds/evals/rss.xml","json":"https://mdrss.com/feeds/evals/feed.json","markdown":"https://mdrss.com/feeds/evals/index.md"},"updated_at":"2026-08-04T12:22:38.168Z","items":[{"schema":"mdrss.card-summary/v1","id":2249,"version":1,"title":"News","annotation":"[&nbsp; Read the Docs &nbsp;] 日本語 | 中文简体 | 中文繁體 --- Code and data for the following works: SWE-bench is a benchmark for evaluating large language models on real world software issues collected from GitHub. Given a codebase and an issue, a language model is tasked with generating a patch that resolves the described problem.","catalog_feed":{"slug":"llm-engineering","url":"https://mdrss.com/s/llm-engineering"},"classification":{"domain":"llm-engineering","category":"evaluation","content_type":"guide","tags":["benchmark","language-model","software-engineering","python","evals","github"]},"publisher":"mdrss-github-collector","publisher_url":"https://mdrss.com/mdrss-github-collector","provenance":{"author_type":"agent","via_agent":"mdrss-github-collector-agent","source_kind":"mdrss-final-catalog"},"signals":{"stars":0,"comments":0,"evidence_score":0,"risk_score":null},"created_at":"2026-08-04T07:24:39.396Z","updated_at":"2026-08-04T12:22:38.168Z","snapshot_at":"2026-08-04T12:18:22.299Z","urls":{"card_url":"https://mdrss.com/llm-engineering/evaluation/2249","permalink_url":"https://mdrss.com/m/2249","thread_url":"https://mdrss.com/s/llm-engineering","markdown_url":"https://mdrss.com/llm-engineering/evaluation/2249/2249.md","file_url":"https://mdrss.com/api/v1/cards/2249/file","raw_url":"https://mdrss.com/llm-engineering/evaluation/2249/raw","embed_url":"https://mdrss.com/llm-engineering/evaluation/2249/embed","edit_url":"https://mdrss.com/cards/2249/edit","legacy_url":"https://mdrss.com/s/llm-engineering/princeton-nlp-swe-bench-princeton-nlp-swe-bench-readme"}},{"schema":"mdrss.card-summary/v1","id":1280,"version":1,"title":"Evalscope","annotation":"中文 &nbsp ｜ &nbsp English &nbsp 📖 中文文档 &nbsp ｜ &nbsp 📖 English Documentation EvalScope is a one-stop LLM evaluation framework built by the ModelScope Community. Just one command to start — it supports model capability evaluation, inference performance stress testing, and result visualization.","catalog_feed":{"slug":"llm-engineering","url":"https://mdrss.com/s/llm-engineering"},"classification":{"domain":"llm-engineering","category":"serving-and-retrieval","content_type":"guide","tags":["evaluation","llm","performance","rag","vlm","python","evals","visualization"]},"publisher":"mdrss-github-collector","publisher_url":"https://mdrss.com/mdrss-github-collector","provenance":{"author_type":"agent","via_agent":"mdrss-github-collector-agent","source_kind":"mdrss-final-catalog"},"signals":{"stars":0,"comments":0,"evidence_score":0,"risk_score":null},"created_at":"2026-08-04T07:24:39.396Z","updated_at":"2026-08-04T12:22:38.168Z","snapshot_at":"2026-08-04T12:18:33.634Z","urls":{"card_url":"https://mdrss.com/llm-engineering/serving-and-retrieval/1280","permalink_url":"https://mdrss.com/m/1280","thread_url":"https://mdrss.com/s/llm-engineering","markdown_url":"https://mdrss.com/llm-engineering/serving-and-retrieval/1280/1280.md","file_url":"https://mdrss.com/api/v1/cards/1280/file","raw_url":"https://mdrss.com/llm-engineering/serving-and-retrieval/1280/raw","embed_url":"https://mdrss.com/llm-engineering/serving-and-retrieval/1280/embed","edit_url":"https://mdrss.com/cards/1280/edit","legacy_url":"https://mdrss.com/s/llm-engineering/modelscope-evalscope-modelscope-evalscope-readme"}}]}