{"data":{"slug":"internscience-scievalkit","name":"SciEvalKit","tagline":"Unified evaluation toolkit and leaderboard for assessing scientific intelligence","github_url":"https://github.com/InternScience/SciEvalKit","owner":"InternScience","repo":"SciEvalKit","owner_avatar_url":"https://avatars.githubusercontent.com/u/154449623?v=4","primary_language":"Python","stars":86,"forks":13,"topics":["agent","ai","ai4science","code-generation","evaluation","evaluation-framework","gemini","gpt","llm","llm-evaluation","vllm"],"archived":false,"github_pushed_at":"2026-08-30T04:00:43+00:00","maintenance_label":"Active","stars_delta_30d":1,"url":"https://www.graphcanon.com/tools/internscience-scievalkit","markdown_url":"https://www.graphcanon.com/tools/internscience-scievalkit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/internscience-scievalkit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=internscience-scievalkit","description":"A unified evaluation toolkit and leaderboard for rigorously assessing the scientific intelligence of large language and vision–language models across the full research workflow.","homepage_url":null,"license":"Apache-2.0","open_issues":6,"watchers":3,"ai_summary":"SciEvalKit is a tool designed to rigorously evaluate the scientific capabilities of large language and vision-language models across every stage of the research process.","readme_excerpt":"### 1 · Install\n\n```bash\ngit clone https://github.com/InternScience/SciEvalKit.git\ncd SciEvalKit\npip install -e .[all]    # brings in vllm, openai‑sdk, hf_hub, etc.\n```","github_created_at":"2025-12-03T05:56:18+00:00","created_at":"2026-07-15T10:39:29.884055+00:00","updated_at":"2026-09-20T04:24:53.454985+00:00","categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agent","name":"agent"},{"slug":"ai4science","name":"ai4science"},{"slug":"code-generation","name":"code-generation"},{"slug":"evaluation-framework","name":"evaluation-framework"},{"slug":"gemini","name":"gemini"},{"slug":"gpt","name":"gpt"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"vllm","name":"vllm"}],"trust":{"provenance":{"is_fork":false,"github_id":1108941331,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-09T06:00:18.488Z","maintenance":{"label":"Active","score":82,"methodology":"github_public_v1","releases_90d":0,"days_since_push":10,"last_release_at":null,"stars_delta_30d":1,"open_issues_delta_30d":3},"security_summary":{"status":"findings","scanner":"osv@v1","low_count":1,"high_count":1,"last_scan_at":"2026-07-15T10:39:31.253Z","medium_count":0,"scan_profile":"deps","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-09T06:00:19.017Z"},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-09-09T06:00:19.017Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-09-09T06:00:19.017Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["When assessing the scientific intelligence of multimodal models specifically across research stages","If you need an integrated approach to evaluate models on various tasks within scientific inquiry"],"when_not_to_use":["For evaluating general performance without a focus on scientific applications and methodologies","If your project does not benefit from an evaluation framework centered around vision-language abilities in scientific contexts"],"source":"enrich:decision_facts","observed_at":"2026-07-16T20:26:13.884Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"SciEvalKit is a unified evaluation toolkit and leaderboard designed to rigorously assess the scientific capabilities of large language and vision-language models throughout research processes."}]}}