原始 JSON 报告 机器可读
{
"data": {
"repo": {
"topics": [
"agent-benchmark",
"agent-evaluation",
"ai-agents",
"crewai",
"evaluation",
"langgraph",
"openai-assistants",
"testing",
"pytest",
"anthropic",
"agentic-ai",
"langchain-agent",
"autogen",
"cli",
"llm",
"mcp",
"python",
"regression-testing"
],
"is_fork": false,
"size_kb": 123325,
"has_wiki": false,
"homepage": "https://evalview.com",
"languages": {
"Shell": 3862,
"Python": 4575659,
"Makefile": 8574,
"Dockerfile": 124,
"JavaScript": 3796,
"TypeScript": 7438
},
"pushed_at": "2026-07-26T19:53:19Z",
"created_at": "2025-11-17T22:21:38Z",
"owner_type": "User",
"updated_at": "2026-07-26T19:57:31Z",
"description": "Regression testing for AI agents. Snapshot behavior,diff tool calls,catch regressions in CI. Works with LangGraph, CrewAI, OpenAI, Anthropic.",
"is_archived": false,
"is_disabled": false,
"license_spdx": "Apache-2.0",
"default_branch": "main",
"license_spdx_raw": "Apache-2.0",
"primary_language": "Python",
"significant_languages": [
"Python"
]
},
"owner": {
"blog": "hidaibarmor.com",
"name": "Hidai Bar-Mor",
"type": "User",
"login": "hidai25",
"company": null,
"location": "Tel-Aviv, Israel",
"followers": 13,
"avatar_url": "https://avatars.githubusercontent.com/u/31502796?v=4",
"created_at": "2017-08-31T07:32:46Z",
"is_verified": null,
"public_repos": 44,
"account_age_days": 3251
},
"license": {
"state": "standard",
"spdx_id": "Apache-2.0",
"raw_spdx": "Apache-2.0",
"file_present": true,
"scorecard_found": true,
"profile_has_license": true
},
"activity": {
"releases": [
{
"tag": "v0.8.1",
"kind": "patch",
"published_at": "2026-07-26T19:53:19Z"
},
{
"tag": "v0.8.0",
"kind": "minor",
"published_at": "2026-05-15T10:21:31Z"
},
{
"tag": "v0.7.1",
"kind": "patch",
"published_at": "2026-05-02T19:38:53Z"
},
{
"tag": "v0.7.0",
"kind": "minor",
"published_at": "2026-04-22T21:14:27Z"
},
{
"tag": "v0.6.2",
"kind": "patch",
"published_at": "2026-04-11T20:25:04Z"
},
{
"tag": "v0.6.1",
"kind": "patch",
"published_at": "2026-03-28T22:22:07Z"
},
{
"tag": "v0.6.0",
"kind": "minor",
"published_at": "2026-03-27T09:22:23Z"
},
{
"tag": "v0.5.5",
"kind": "patch",
"published_at": "2026-03-25T09:23:15Z"
},
{
"tag": "v0.5.4",
"kind": "patch",
"published_at": "2026-03-23T15:32:43Z"
},
{
"tag": "v0.5.3",
"kind": "patch",
"published_at": "2026-03-18T22:10:45Z"
},
{
"tag": "v0.5.2",
"kind": "patch",
"published_at": "2026-03-17T08:57:02Z"
},
{
"tag": "v0.5.1",
"kind": "patch",
"published_at": "2026-03-13T20:13:34Z"
},
{
"tag": "v0.5.0",
"kind": "minor",
"published_at": "2026-03-12T12:05:36Z"
},
{
"tag": "v0.4.1",
"kind": "patch",
"published_at": "2026-03-09T09:47:58Z"
},
{
"tag": "v0.4.0",
"kind": "minor",
"published_at": "2026-03-05T10:55:10Z"
},
{
"tag": "v0.3.2",
"kind": "patch",
"published_at": "2026-02-27T11:10:02Z"
},
{
"tag": "v0.3.0",
"kind": "minor",
"published_at": "2026-02-20T10:41:31Z"
},
{
"tag": "v0.2.9",
"kind": "patch",
"published_at": "2026-02-19T06:39:37Z"
},
{
"tag": "v0.2.8",
"kind": "patch",
"published_at": "2026-02-19T06:33:09Z"
},
{
"tag": "v0.2.7",
"kind": "patch",
"published_at": "2026-02-19T06:23:59Z"
},
{
"tag": "v0.2.6",
"kind": "patch",
"published_at": "2026-02-19T06:18:15Z"
},
{
"tag": "v0.2.5",
"kind": "patch",
"published_at": "2026-02-15T10:41:01Z"
},
{
"tag": "v0.2.4",
"kind": "patch",
"published_at": "2026-02-02T05:28:59Z"
},
{
"tag": "v0.2.3",
"kind": "patch",
"published_at": "2026-01-24T22:42:16Z"
},
{
"tag": "v0.2.1",
"kind": "patch",
"published_at": "2026-01-11T22:10:19Z"
},
{
"tag": "v0.2.0",
"kind": "minor",
"published_at": "2026-01-10T13:29:24Z"
},
{
"tag": "v0.1.9",
"kind": "patch",
"published_at": "2026-01-03T14:57:25Z"
},
{
"tag": "v0.1.8",
"kind": "patch",
"published_at": "2025-12-31T21:46:30Z"
},
{
"tag": "v0.1.7",
"kind": "patch",
"published_at": "2025-12-29T14:25:28Z"
},
{
"tag": "v0.1.6",
"kind": "patch",
"published_at": "2025-12-29T14:12:47Z"
},
{
"tag": "v0.1.5",
"kind": "patch",
"published_at": "2025-12-19T11:45:04Z"
},
{
"tag": "v0.1.4",
"kind": "patch",
"published_at": "2025-12-09T23:05:03Z"
},
{
"tag": "v0.1.3",
"kind": "patch",
"published_at": "2025-12-09T09:36:37Z"
}
],
"recent_commits": [
{
"oid": "afbc7f6c715d7cd1ab1ea71aefd5e26a9bbcd767",
"body": "The roadmap had drifted out of sync with what actually shipped:\n\n- \"Where we are\" still said June 2026 and 14+ adapters; bumped to\n July / v0.8.1, added capture, py.typed, and the Vercel AI SDK adapter\n- \"Next batch\" still listed the Vercel adapter that merged in June (#250)\n- Claimed every listed \n[…]\nunderlying\nwatcher already exists. All four were stale `help wanted` issues that\ngave arriving contributors nothing real to pick up.\n\nCo-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(roadmap): refresh for July 2026 and credit contributors (#262)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-26T19:49:41Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "8ffdfdf069822d8e54e160b88b93f847847ebbd8",
"body": "* feat(check): add --watch, and fix the watcher it delegates to\n\nCloses #242.\n\n`evalview check --watch` re-runs the check on every file change by\ndelegating to the existing watcher, so the snapshot/check loop gains a\nwatch mode without reaching for a third command. Watch mode supports\nTEST_PATH, --t\n[…]\nt the release being cut\nrather than the previous one.\n\nCo-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "0.8.1: check --watch, watcher fix, loader fix (#263)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-26T19:47:27Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "761018a49bdaf4875c57bfb32082a40e2ed2f296",
"body": "The daily self-testing workflow is the project's strongest trust signal\nbut lived only in docs/. Surface it: add the live workflow-status badge\nto the header row and a dedicated section explaining that EvalView runs\nits own snapshot/check loop against itself every day, with failures\nposted to a public rolling issue.\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(readme): feature the daily dogfood workflow",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T14:23:18Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "6286c756cd3ec853e560344cc860a3a40418a12d",
"body": "CI installs the optional cohere and mistralai extras, so these two\nadapters type-check against real SDK types there (locally they were\nsilenced by ignore_missing_imports). Reproduced by installing the\nextras:\n\n- Use the SDKs' typed message classes instead of raw role/content\n dicts.\n- cohere: messa\n[…]\nnt[0].text.\n- mistral: message content may be a chunk list rather than a string\n (join the text-bearing chunks), and token counts are Optional.\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(types): fix cohere/mistral adapters against real SDK stubs",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "f68101d068de13c6857e18a2811db42ce466909f",
"body": "Replace the 999-line feature tour with a 127-line README that leads\nwith the core value proposition: snapshot your agent's behavior, get\ntold when it silently changes. Positions EvalView as a merge-time\nregression gate (vs observability and metric-scoring tools), keeps\nQuick Start to two commands, and pushes power features to the docs.\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(readme): rewrite around the snapshot/check loop",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "3cb319625d64471c67578b0b02c890de94ecea44",
"body": "Remove the ignore_errors escape hatch for evalview.reporters.* and fix\nthe 52 errors it was hiding. Most were one missing declaration: the\nConsoleReporter mixins read self.console but never declared it, which\nalone accounted for 43 errors. The remaining nine were real bugs —\nattribute names that don\n[…]\nuality_hints: evaluations.tool_calls →\n tool_accuracy, and .score → .accuracy (scaled to 0-100 to match\n the threshold it's compared against).\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(types): full mypy coverage for reporters",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "6cd145803bfe8ea888423a838299d8756d4a2feb",
"body": "Remove the ignore_errors escape hatch for evalview.evaluators.* and\nfix the 39 errors it was hiding:\n\n- ExpectedBehavior.output is Union[ExpectedOutput, Dict] (YAML loading\n can leave a raw dict), but evaluators accessed model attributes on it\n directly — an AttributeError waiting for the dict arm\n[…]\ntions, and a\n None-typed SequenceMode parameter.\n\nPre-existing failure in test_snapshot_json_output is unrelated (fails\non the clean tree too).\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(types): full mypy coverage for evaluators",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "68d4a1e7a2fbc6ee23a1ff5aa17d14e38003c00d",
"body": "Remove the ignore_errors escape hatch for evalview.adapters.* and fix\nthe 21 errors it was hiding. Two were latent runtime bugs:\n\n- ollama: TokenUsage(total_tokens=...) passed a kwarg that isn't a\n field (it's a computed property) — silently dropped by pydantic\n- langgraph: tool_name=step_data.get(\n[…]\ntoken counts parsed from untyped\nresponse payloads, Optional narrowing for subprocess pipes, and\nmissing annotations on empty-list initializers.\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(types): full mypy coverage for adapters",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "ef69830350601adf0713a6dba659ae65781ffac6",
"body": "Without the marker, mypy/pyright treat installed evalview as untyped\nand consumers of the public API (evalview.gate et al.) get no type\nchecking. Verified the marker lands in the built wheel.\n\nAlso syncs version strings that had drifted from pyproject:\nevalview.__version__ was 0.7.0, server.json was 0.6.1.\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(types): ship PEP 561 py.typed marker — release 0.8.1",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "bf4b48590b222f1a0477847864ad72bd2f4d7c94",
"body": "The 0.1.0 entry predated the project's first commit by ten months.\nPyPI records the 0.1.0 upload at 2025-12-03, two days before 0.1.1.\n\nCo-Authored-By: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(changelog): fix 0.1.0 release date (2025-01-24 → 2025-12-03)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-07-03T13:43:53Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "99e1e8472f655229b44984116aef2f604873a947",
"body": "* feat: Add Vercel AI SDK adapter\n\n* Fix lint and type check errors",
"is_bot": false,
"headline": "feat: Add Vercel AI SDK adapter (#250)",
"author_name": "Gaurav Kumar Thakur",
"author_login": "gauravxthakur",
"committed_at": "2026-06-15T04:56:57Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "47753a89da5ab8dd4412f2177aa89c69fe94d397",
"body": "… (#249)\n\n* docs: standardize GitHub Action pin to v0.8.0\n\nThe README pinned the action to @main while docs/CI_CD.md and others\npinned @v0.6.1, giving users inconsistent and outdated setup snippets.\nUnify all 14 references across README, docs, examples, and llms-full.txt\nto the latest release (v0.8.\n[…]\n, public-docs disclaimer with a corrections link and a\n trademark/non-affiliation notice to the README, COMPARISONS, and the\n llms.* files.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs: add prompt-as-migration guide and update comparisons for v0.8.0…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-06-03T19:41:41Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "c0a48053bebacbece25216fa7be95057e6fd656c",
"body": "…247)\n\nLift the duplicated _STOPWORDS frozenset out of freshness, goal_drift,\nand retrieval_lineage into a shared evalview.core.text module. Convert\nPEP 585 builtins (set[...], frozenset[...]) to typing.Set / FrozenSet\nacross the six new modules to match CLAUDE.md's Python-3.9 style.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor(core): dedupe stoplist and normalize generics to typing.* (#…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-27T15:49:02Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "2dc0e2b5d4d5bfa928ab11db939ae8666b3d13b5",
"body": "Frame the next batch around real May 2026 agent-testing pains\n(prompt-as-migration, non-determinism, CI-native tool diffing) and link\nto six new help-wanted issues so newcomers have an obvious entry point.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs: add ROADMAP to attract contributors (#245)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-24T07:36:28Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "8154f95b6e3e3347360890a0847d760a7e87544e",
"body": "* feat(cloud): unify CLI auth with the SaaS via /cli-auth loopback flow\n\n`evalview login` previously OAuthed against a standalone Supabase\nproject that the cloud dashboard never read from, so users could\n\"log in\" successfully and still have every subsequent push silently\ndrop on the floor. The login\n[…]\ngged in\nthe new module/test). No behavior change.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(cloud): unify CLI auth with SaaS via /cli-auth loopback flow (#236)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-18T19:08:04Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "c996a89324800c0d68b6d9629dd46adb0637c91b",
"body": "Bump pyproject.toml, package.json, and uv.lock to 0.8.0 and roll the\n[Unreleased] cassette entries into a dated CHANGELOG section alongside\nthe new --schedule cron syntax for evalview monitor (#224, #227) and\nthe lockfile sync (#230). Headline features: record/replay cassettes\nfor hermetic tool replay (#228) and cron-based monitor scheduling.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(release): 0.8.0 (#234)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-15T10:20:51Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "4253ad50a7ed890e55b20ae5008f95a945d848e0",
"body": "…cause modules (#233)\n\n* feat(freshness): detect production-query coverage gaps in the eval suite\n\n`evalview autopr` already closes the loop from a failing production trace\nto a pinned regression test. `evalview freshness` closes the complementary\nloop — from drifted traffic to a new capability test\n[…]\nsame seed = same scenario, different\nseed = different scenario), step-bound respect, registry\nconsistency, and frozen-dataclass immutability.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Add freshness, fleet, chaos, goal-drift, retrieval-lineage, and root-…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-15T09:39:10Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "f647590a0348e91a0f88fa424367d1c043d71ba8",
"body": "Lockfile drift from version bump (0.5.3 → 0.7.1) and the new `schedule` extra (croniter). Brings runtime deps jinja2/jsonschema/tomli into the locked core set.",
"is_bot": false,
"headline": "chore(deps): sync uv.lock with pyproject (#230)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-10T21:26:36Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "61147799ab88a9245f9034f4bb601311b267f900",
"body": "* feat(simulate): record/replay cassettes for hermetic tool replay\n\nAdds VCR-style record/replay so users can capture real tool calls\nonce and replay them deterministically forever after — closing the\n\"replay reproducibility\" gap where tests were really testing infra\nbecause the only way to reproduc\n[…]\nional[Any];\nwrap with bool() so a missing/None capability renders as ✗.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Add record/replay cassettes for hermetic tool call simulation (#228)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-04T20:45:54Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "70ba9f69ca0703dc12205a0d107502b6a3c41825",
"body": "…low-up) (#227)\n\n- _run_monitor_dashboard now accepts cadence_label (matching _run_monitor_loop)\n and shows it in the header, so the cron schedule is visible from cycle 0\n instead of only after the first cycle completes.\n- In the dashboard's exception path, move live.update() after next_check_time\n is recomputed so the rendered footer reflects the next scheduled run.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(monitor): dashboard parity with --schedule cadence (PR #224 fol…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-03T19:23:21Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "cf56d0e44d0bf7a4221d22414fd8647ee86d4113",
"body": "* feat(monitor): add --schedule cron syntax to evalview monitor\n\nCloses #84\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>\n\n* fix(monitor): silence mypy import-untyped for croniter\n\ncroniter has no type stubs, so mypy errors on the lazy import even with\nignore_missing_imports \n[…]\n Opus 4.7 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(monitor): add --schedule cron syntax to evalview monitor (#224)",
"author_name": "Matt Van Horn",
"author_login": "mvanhorn",
"committed_at": "2026-05-03T19:14:01Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "20e346e38a6276fae6015ce44046edf76ac291a6",
"body": "Bump pyproject.toml + package.json to 0.7.1 and roll the [Unreleased]\npolish into a dated CHANGELOG section together with the new shipping\nitems: TOML test cases (#209), CSV log import (#216), `parse_csv` warn\ntype tightening (#221), the four-file split (#215), root polish (#218),\nand the `AGENTS.md`/`docs/guides/` rename (#219).\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(release): 0.7.1 (#222)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-02T19:38:15Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "e30153f6b1b1ddc135339a3145b31538d2d20a57",
"body": "Replaces `warn: Any = None` with `Optional[Callable[[str], None]] = None`\nso the contract is explicit at the call site and mypy can catch misuse.\nBehavior unchanged.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(importers): tighten parse_csv warn callable type (#221)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-05-02T19:12:32Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "3a15b17810a009cc341e9bfc1143a0e69c5aa950",
"body": "* feat(importers): add CSV support for generate --from-log (#94)\n\nAdds a fourth log-import format alongside JSONL, OpenAI, and EvalView\ncapture. Many teams export agent interactions as CSV (spreadsheet\ntooling, analytics dumps), so this is the lowest-friction format for\nseeding test suites.\n\nChanges\n[…]\n------\n\nCo-authored-by: Hidai Bar-Mor <31502796+hidai25@users.noreply.github.com>\nCo-authored-by: Hidai Bar-Mor <hidai25@gmail.com>\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(importers): add CSV support for generate --from-log (#94) (#216)",
"author_name": "Matt Van Horn",
"author_login": "mvanhorn",
"committed_at": "2026-05-02T19:08:37Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "112d0b5daa7ad741b07a31f5e1fa627b0293a7fa",
"body": "…uides/` under `docs/` (#219)\n\nTier-2 cleanup, conservative pass — only the moves where I could verify\nnothing critical breaks:\n\n- **`AGENT_INSTRUCTIONS.md` → `AGENTS.md`** matches the emerging\n convention that AI coding agents look for. Updated 6 references\n (README.md, Makefile, docs/INTERNAL_DO\n[…]\nMerging would harm\n discoverability without a real readability win.\n\nPure rename/move. Ruff, mypy, and 1842-test suite all green.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(repo): rename `AGENT_INSTRUCTIONS.md` → `AGENTS.md` and move `g…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-30T12:39:07Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "bba3bdfe2bf0cf5ffeed0f069884f43f9159a761",
"body": "Tier-1 cleanup from the project assessment — purely cosmetic, no behavior\nor public-API changes:\n\n1. **`.gitignore`**: `**/.DS_Store` (was top-level only); `.aider*` glob\n replaces the two specific aider history files; new `assets/*.mp4` /\n `assets/*.mov` entries to prevent re-tracking.\n2. **`de\n[…]\nready\n gitignored, just stale local files.\n\nAfter: `ls` at root shows ~28 entries (down from ~35), each with an\nobvious purpose.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(repo): root-directory polish pass (#218)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-30T12:28:08Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "602e5ef05ec00c7aa23b95579305644b22ea4f9a",
"body": null,
"is_bot": false,
"headline": "refactor: split 4 remaining 1k+ files into focused submodules (#215)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-29T17:13:36Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "1c66b716f0db3ed1036e09ec97f85b09057c1a85",
"body": "`inspect` and `visualize` had largely overlapping functionality —\nboth generate a visual HTML report from a results file, both\nauto-open in the browser, and both support --title/--notes/--no-open/\n--output. `visualize` is a strict superset: it adds --compare for\nmulti-run side-by-side comparison.\n\nC\n[…]\nprawl of \"show me the results\" commands from 3\n(report/inspect/visualize) to 2 (report for console + plain HTML,\nvisualize for rich HTML + comparisons).\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor(cli): merge `inspect` into `visualize` (#214)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-29T04:37:41Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "4c5a9b73520a8c61c8e7f9f0993bffaff80ec0d1",
"body": "Closes #143.\n\nevalview/core/loader.py\n - _parse_test_case_file dispatches on file suffix: .toml -> tomllib\n (Python 3.11+) or tomli (3.9/3.10), everything else -> yaml.\n - load_from_file uses the dispatcher; existing YAML callers are\n unaffected.\n - load_from_directory now picks up .yml AND\n[…]\nts/test_loader.py -v -> 28 passed (23 prior + 5 new)\n\nCo-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>\nCo-authored-by: Hidai Bar-Mor <31502796+hidai25@users.noreply.github.com>",
"is_bot": false,
"headline": "feat(loader): support TOML test cases alongside YAML (#209)",
"author_name": "Matt Van Horn",
"author_login": "mvanhorn",
"committed_at": "2026-04-29T04:29:25Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "994ecb28b132563f5cf547244487a6c7a60b0642",
"body": "Before this commit a fresh contributor running pytest saw 12 errors\nand 1 hard failure that had nothing to do with their changes. Fix:\n\n- Add `requires_api_key` pytest marker. tests/conftest.py auto-skips\n any test marked with it when neither OPENAI_API_KEY nor\n ANTHROPIC_API_KEY is set, with a cl\n[…]\nfull suite goes from `1818 passed, 12 errors, 1 failed` to\n`1831 passed, 6 cleanly-skipped, 0 errors, 0 failures` on a vanilla\nmachine with no API keys.\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: make the suite hermetic — `git clone && make test` is green (#211)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T12:15:01Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "e03e400fd52b59f170daa1e1e3324b844ab9edc8",
"body": "Brings in the PR #211 hermetic-tests changes plus a fix for the CI\nfailures the original PR introduced on Python 3.10/3.11/3.12.\n\nRoot cause of the CI failure: PR #211 added an `env: OPENAI_API_KEY`\nforward to the pytest step in ci.yml. With pytest-asyncio 1.3.0\n(used on 3.10+), class-level `@pytest\n[…]\nown-good model and gates on\nthe secret existing.\n\nAlso updates the CHANGELOG entry to remove the now-obsolete bullet\nabout ci.yml forwarding the secret.\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: hermetic suite + drop CI OPENAI_API_KEY forwarding (#212)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T12:04:13Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "eda2c22c672ff41c3afa9b0928bde0edbae82e1b",
"body": "* refactor(chat): derive command allowlist from the Click registry\n\nReplace the hand-maintained VALID_EVALVIEW_COMMANDS / VALID_RUN_FLAGS /\nVALID_DEMO_FLAGS / VALID_ADAPTERS_FLAGS / VALID_LIST_FLAGS sets in\nchat.py with a single derive_chat_allowlists() call in chat_runtime.py\nthat walks main.comman\n[…]\n's changes in CHANGELOG [Unreleased]:\nchat-allowlist auto-derivation, docs/README.md index expansion, and\nthe demo-dir cross-linking READMEs.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Derive chat validator allowlist from Click registry (#210)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T11:20:54Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "06d8f66fae43de232908a9336003145ceb1db52d",
"body": "* fix(lint): drop unused asyncio/sys imports from init_cmd.py\n\nThe quickstart removal in #206 left two now-unused imports in\ninit_cmd.py — asyncio and sys were only used by the deleted shim.\nThis broke the Lint & Type Check job on main.\n\n* chore: remove requirements.txt in favor of pyproject.toml + \n[…]\nANGELOG entries for this branch's\nchanges (quickstart removal, requirements.txt removal, dogfood\nrolling issue, view surfacing, CI lint fix).\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Surface `evalview view` command and clean up dependencies (#207)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T10:06:36Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "fd3efaf09bd8dd35c2e0a5640379c9d4e686f377",
"body": "* ci(dogfood): roll daily failures into one persistent issue\n\nReplace the per-day \"Create issue on failure\" step with a rolling-issue\npattern: find an open issue titled \"🐕 Daily dogfood is failing\" and\neither comment on it (failure) or close it (success). When no rolling\nissue exists yet and dogfood\n[…]\nDrop the ~400-line shim and\nits registration from cli.py. Update docs/CLI_REFERENCE.md and a guide\nthat still pointed at the removed command.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Claude/dogfood rollup and quickstart removal (#206)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T09:03:52Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "647da389c85844efc55c34b081bb8bb930ae8836",
"body": "* chore: clean up stale root-level files\n\nRemove generated HTML reports (report.html, diff-report.html,\nbenchmark_compare.html), planning docs for the cloud project (now in\na separate repo), the goosebench task leftover artifact, and an unused\nscratch script. Add gitignore patterns for the report fi\n[…]\n.md was filed as a real GitHub issue\n(#106-#131) and the vast majority are now closed. The markdown file is\nredundant with the issue tracker.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Clean up planning documents and generated reports from repository (#205)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T08:43:50Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "47e103567c03cb90aa3bf536f0828829e5fe1357",
"body": "* refactor(cli): move validate-adapter under `adapters validate`\n\nConvert top-level `adapters` from a command into a Click group with two\nsubcommands: `list` (default behavior, preserves existing UX) and\n`validate` (formerly `validate-adapter`). The standalone `validate-adapter`\ncommand is kept as a\n[…]\n).\n\nStrip the long command-list blocks from the root docstring so the help\ntext and the sectioned command listing don't duplicate each other.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Reorganize CLI help with curated command sections (#204)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T08:12:24Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "92a87a506388152ac99bf85b7392d5777d0ce93e",
"body": "Eight tests were failing on a developer machine (any environment with\nLLM provider API keys set):\n\n- test_init_cmd.py: 3 tests\n- test_snapshot_generated_workflow.py: 5 tests\n\nRoot cause: both `evalview init` and `evalview snapshot` added new\ninteractive prompts after these tests were written —\n- sna\n[…]\nthan changing what\nCI verifies.\n\n1837 tests pass with no `--ignore` flags (was 1792 with 3 files\nignored). Ruff clean, mypy clean.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: fix interactive-prompt mocking in init/snapshot tests",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-28T03:36:09Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "5a89d038fe37647070f4cef07eb92cfa8a53fa16",
"body": "CI on main flagged 9 ruff errors and 1 mypy error introduced by the\nrefactor merge. All are import-hygiene fallout from extracting helpers\ninto submodules:\n\n- check_cmd.py: drop now-unused `_parse_fail_statuses`, `Tuple`,\n `TraceDiff`; mark `_format_snapshot_timestamp` as a noqa re-export\n- _securi\n[…]\n infer the\n list element type (project convention is `List[...]` not `list[...]`)\n\nZero functional change. 1792 tests still pass.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(refactor): clear ruff and mypy errors from PR #202 merge",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-27T20:48:23Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "7760ffe751fff08b3194d4d492912fcc7e396c0b",
"body": "* chore(diff): apply review follow-ups to PR #198\n\nSix small corrections to the diff command merged in #198. The bug-fixy\nones are #1 and #2 (the test fixture wasn't exercising real production\ndata shape, and the helper coupled itself to the CLI); the rest are\nconvention/coverage hardening.\n\n1. Fix \n[…]\nng\nall 96 tests in tests/skills/test_security.py.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor: split 6 large files into focused submodules (#202)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-27T20:23:13Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "aac61f47b0af68f5b66929ae1ea09a2c6aa82e63",
"body": "Six small corrections to the diff command merged in #198. The bug-fixy\nones are #1 and #2 (the test fixture wasn't exercising real production\ndata shape, and the helper coupled itself to the CLI); the rest are\nconvention/coverage hardening.\n\n1. Fix the test fixture's latency field. The original test\n[…]\nix: `evalview/core/observability.py:223`\ntreats `total_latency` as seconds and multiplies by 1000 — that's\na 1000x bug there, since `total_latency` is in ms per the adapters.\nOut of scope for this PR.",
"is_bot": false,
"headline": "chore(diff): apply review follow-ups to PR #198 (#201)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-27T06:35:24Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "aabe8486d2b3c763483b1c000841027eacba2477",
"body": "Resolves #141.\n\nAdds 'evalview diff <run1.json> <run2.json>' that pretty-prints what\nchanged between two result files:\n\n- Tool-sequence diff per test case\n- Output similarity via difflib.SequenceMatcher (color-coded:\n >= 95% dim, 80-94 yellow, < 80 red)\n- Cost delta and latency delta (negative = gr\n[…]\nren't a list of test-case objects\n\nTests cover the happy path, --json output shape, and the three\nfile-validation error paths.\n\nCo-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>",
"is_bot": false,
"headline": "feat(diff): add evalview diff command (#198)",
"author_name": "Matt Van Horn",
"author_login": "mvanhorn",
"committed_at": "2026-04-27T05:27:09Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "f1ac5513b983a337aca223c254303c808892130b",
"body": "- demo/fixtures/aider/undoc.py: undocumented fib() — fixture for the\n docstring-add task.\n- demo/fixtures/aider/untyped.py: untyped merge_dicts() — fixture for\n the type-hint-add task.\n- demo/tests/aider/docstring.yaml: aider+sonnet adds Google-style\n docstring to fib(); asserts read_file/edit_fi\n[…]\nrompt.\n- .gitignore: exclude .aider.chat.history.md and .aider.input.history\n so local aider state stops appearing in git status.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "demo: add aider docstring/types fixtures + local-deep-researcher test",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-26T20:09:39Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "4ebb1389197ee37cd974a0a4d3b04b5aa6365e30",
"body": "GitHub deprecated Node 20 in JavaScript actions. Starting June 2,\n2026, all actions are force-upgraded to Node 24 regardless of what\nthey declare; on September 16, 2026 Node 20 is removed entirely.\n\nactions/checkout@v4 and actions/setup-python@v5 ship with Node 20\nbinaries — current CI emits a depre\n[…]\nodecov-action@v4, upload-artifact@v4,\ngithub-script@v7) were not flagged by the deprecation notice and are\nleft for a future bump.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "ci: bump checkout to v5 and setup-python to v6 for Node 24",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-26T19:58:11Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "05cffa0340e0fafd1b8cfe053b4072a46c4eed0a",
"body": "Five small corrections to the validate command merged in #195. None\nchange observable behavior on the happy path; #1 and #2 are real bug\nfixes, the rest are convention/coverage hardening.\n\n1. Dedupe `_CONFIG_FILES`: import `CONFIG_FILE_PATTERNS` from\n evalview.core.loader instead of redeclaring th\n[…]\nainst future discovery refactors)\n\n13 tests pass (was 8); ruff clean; mypy clean across 212 source\nfiles; full suite 1505 passing.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(validate): apply review follow-ups to PR #195 (#199)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-26T19:43:31Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "9e3fe9ac36bbbecbfc0e41c871362f0540929518",
"body": "Closes #144\n\nAdds `evalview validate [path]` that lints test YAML/TOML files against\nthe TestCase schema without running any agent or making API calls.\n\n- Reports all errors across files, not just the first\n- Shows filename and specific field errors\n- `--json` outputs machine-readable results\n- Exit\n[…]\n in\nevalview/commands/skill_cmd.py: rich vs JSON output split, per-file\niteration, and Click CliRunner-friendly exit handling.\n\nCo-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>",
"is_bot": false,
"headline": "feat: add validate command for linting test files (#144) (#195)",
"author_name": "Matt Van Horn",
"author_login": "mvanhorn",
"committed_at": "2026-04-26T19:34:33Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "36833ec09425b96d62a93bf8a8469215d0c6976f",
"body": "push.py was emitting `run_type=\"check\"` and putting per-test\nobservability arrays only in the legacy `observability` envelope.\nCloud's Zod schema rejects \"check\" (accepts only \"standard\" /\n\"simulation\") and reads per-test arrays at the top level\n(`behavioral_anomalies`, `trust_scores`, `coherence_an\n[…]\n\nApril but didn't exist. Documents what cloud stores, what it\nnever runs, and which constants need lockstep updates between\nrepos.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(cloud): align push payload with cloud schema v2 (#197)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-25T19:56:29Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "b6798e2236349121443a502aed40cb60e7bc7275",
"body": "The repo had accumulated 95 ruff errors because CI ran mypy but not\nruff — contributors who ran `ruff check` locally caught their own\nissues, contributors who didn't caught nothing. The previous commit\ncleared the backlog; this commit closes the loop so it can't re-form.\n\nRenames the job from \"Type \n[…]\nble via .pre-commit-config.yaml —\ncontributors who run `pre-commit install` will catch issues before\npush, but CI is the backstop.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "ci: add ruff to lint job to prevent backlog regrowth",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-25T13:14:46Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "98b2bf3254f94e10324ca6f381f13922eea00ba2",
"body": "Resolves a long-standing backlog that had accumulated because CI runs\nmypy but not ruff. The next commit wires ruff into CI so this stays\nclean going forward.\n\nAuto-fixed (77, via ruff check --fix):\n- 68 unused imports across 33 files\n- 8 f-string-missing-placeholders (e.g. f\"static text\" → \"static \n[…]\n# type: ignore[name-defined]`.\n\nVerified: ruff clean across evalview/ tests/, mypy clean across 211\nsource files, 1492 tests pass.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(lint): clear ruff backlog (95 errors → 0)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-25T13:14:46Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "aa72020d0056e3ee7436e889141f689d45ef294d",
"body": "…ict conventions\n\nThe 504405f trust-hint commit shipped a TypedDict mistake and a missing\ntest for a new public constant — both would have been caught at PR\nreview time if the checklist prompted for them. This wires up\nmechanical and human enforcement so the same gap doesn't reopen.\n\n- .pre-commit-c\n[…]\ner total=False + Required — silent\n degradation matters) and the Literal-mirroring-constant rule\n (must ship with a drift test).\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(quality): pre-commit hooks, PR template invariant check, TypedD…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-25T13:14:46Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "ba41cfe307f8206d526c7520e73319e774566860",
"body": "Follow-up to 504405f. The new public surface — GamingFlag.hint and\nDECISION_TYPE_DESCRIPTIONS — shipped without test coverage and with\none TypedDict whose `total=False` flag did the opposite of what its\ndocstring claimed.\n\n- TrustFlagDict: switch from `total=False` to default `total=True`\n with Not\n[…]\n errors in the touched test\nfiles — unused imports and a mid-file import — to keep ruff output\nquiet on these files going forward.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(observability): pin trust-hint invariants and tighten TrustFlagDict",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-25T13:14:46Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "c3abd49a129b3cf8aacaf0707b0d9acf6361a158",
"body": "…cision_type descriptions (#194)\n\nCloud was inventing plain-language meanings for TrustIndicator flags\nand rationale decision types because the OSS side didn't carry them.\nThis lifts that information up to the single source of truth.\n\n- TrustFlagDict (observability.py): add optional `hint` field\n- G\n[…]\nmpatible on the wire: cloud's Zod schema accepts unknown\nkeys and will render the new `hint` when present without any schema\nbump.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(observability): add optional `hint` to GamingFlag + canonical de…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-23T05:47:33Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "60ac9f46b218efe5efe6978a2e5f0ec85fbf4b71",
"body": "Minor bump (33 commits since 0.6.2). Headline additions:\nAider adapter, flake quarantine, autopr loop, release verdict +\n`evalview since`, progress/drift/slack-digest investigative commands,\nnoise confirmation gate, observability signals, improvement\nrecommendations, simulation harness (schema v2), `snapshot --json`,\n`check --explain`, token cost breakdown. See CHANGELOG for the full\nlist.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore(release): 0.7.0 (#193)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-22T21:13:25Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "c5f6c9481545ec5ff1bcbfaae6b5e2b68dc261c0",
"body": "…192)\n\nRecurring dogfood failures (#181, #184, #187, #190, #191) traced to:\n - Test 01 missing expected.tools: [evalview_cli] even though the agent\n legitimately calls it to answer \"What adapters does EvalView support?\".\n - Tests 01/02 max_latency: 30000ms too tight for real OpenAI/Anthropic\n \n[…]\nved 47–76s end-to-end).\n\nTests 03/04 already use tools: [evalview_cli] and 180000ms — this commit\naligns 01/02 with that baseline.\n\nCo-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(dogfood): align tests 01/02 with CI LLM latency and tool usage (#…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-22T20:30:00Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "952faa8060fe6af4f7f1612f646d0ec369d64b1c",
"body": null,
"is_bot": false,
"headline": "Remove generated artifacts from git tracking (#189)",
"author_name": "Gaurav Kumar Thakur",
"author_login": "gauravxthakur",
"committed_at": "2026-04-22T20:01:35Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "da06956e751a8907a7b2f01582f7c26d3a720f77",
"body": "* Show token cost breakdown in check output\n\n* Fix type check error\n\n* Fix type check error\n\n* Restore accidentally deleted files\n\n* Fix type error by adding explicit annotation",
"is_bot": false,
"headline": "Show token cost breakdown in check output (#185)",
"author_name": "Gaurav Kumar Thakur",
"author_login": "gauravxthakur",
"committed_at": "2026-04-22T10:28:25Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "b3a4f927b187c4550ba0402be66f200935597e55",
"body": "* feat(schema): add v2 types for simulation + rationale capture\n\nIntroduces the wire-format foundation for two agent-eval gaps the\ncommunity flagged in April 2026:\n- pre-deployment simulation harness (mock-injection, what-if runs)\n- structured decision-rationale logging (why did the agent branch?)\n\n\n[…]\nture-table rows and one docs-table link pointing at the new\n docs. No section reorder; existing structure untouched.\n\n222 tests still green.\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Add simulation harness and decision-rationale capture (schema v2) (#188)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-20T11:51:15Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "c0d57582f72aae331999870a87068e8e1cbdc8e0",
"body": "Follow-up to Matt's --json flag:\n\n- Keep stdout clean in JSON mode by plumbing json_output into\n _execute_snapshot_tests and running without the Rich live spinner.\n Previously per-test prints and spinner frames leaked into the\n payload, making stdout unparseable.\n- Track the real set of saved tes\n[…]\nal save failures), variant-aware paths, --preview\n rejection, and the empty-suite error shape.\n\nhttps://claude.ai/code/session_01YPVyciLBGFEpKoMV7y8zFQ\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(snapshot): harden --json output for CI consumers (#186)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-20T07:50:28Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "c73182fc019bcf2cf504bb444e2676f219fda9f7",
"body": "Adds machine-readable JSON output to the snapshot command, matching\nthe pattern already used by evalview check --json. When --json is\nactive, all Rich console output is suppressed and structured JSON\nis printed to stdout with test names, scores, golden file paths,\nvariant info, and timestamp.\n\nCloses #145\n\nCo-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>",
"is_bot": false,
"headline": "feat: add --json flag to evalview snapshot for CI pipelines (#182)",
"author_name": "Matt Van Horn",
"author_login": "mvanhorn",
"committed_at": "2026-04-20T07:48:10Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "6cfe8fa031b9d9caaf0b426545b8f730cb12ede5",
"body": "Fixes bugs found during review, then builds the --explain feature on top\nof the cleaned-up foundation.\n\nBug fixes:\n- HTML report now accepts pre-computed root causes (ai_explanation and\n narrative_root_cause) instead of re-running analyze_root_cause() from\n scratch, so --ai-root-cause / --explain \n[…]\nbjects\n\nTotal: 65/65 root_cause tests pass; full suite 1733 passed (same 12\npre-existing isolation failures in test_evaluators.py unchanged).\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(explain): add --explain flag for deep trace narrative analysis",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-18T10:16:58Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "dd32b5a484d512098700b9454227a28758595942",
"body": "…olds\n\nTwo causes of the daily Dogfood workflow perpetually filing \"failed\"\nissues despite the underlying features working:\n\n1. Monitor smoke-test step: `timeout --preserve-status 35s` sends\n SIGTERM at the deadline and the monitor exits 143. That's the\n INTENDED termination path — we cap the sm\n[…]\nd\n as \"Unexpected tools\". Align with 03-run-command.yaml — declare\n expected.tools: [evalview_cli] and raise max_latency to 180000.\n\nRoot causes are both test-calibration, not product regressions.",
"is_bot": false,
"headline": "fix(dogfood): accept SIGTERM as clean monitor exit + loosen 04 thresh…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-17T15:48:07Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "c37176a2c06362788c25c34d0d2711cf537a91ca",
"body": "Five issues surfaced in review of PR #179 + commit 38aa744 (\"production-\nhardening\"). Address all five here.\n\n1. turn_coherence._detect_tool_regression was O(n²) and emitted one\n near-duplicate issue per (i, j) pair. A 10-turn conversation with a\n consistent tool drop produced ~45 warnings. Pick\n[…]\nwiring). Every test would fail\n against a no-op implementation.\n\n1728 tests pass (up from 1712); mypy clean on 208 source files.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor: finish production-hardening the observability stack",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-17T08:06:20Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "f53f3cda6a871eaf554329a039d79e26e534198c",
"body": "The new tests from PR #177 passed `-r` to `evalview skill doctor`, but\nrecursion is already the default (`default=True` on the flag). On Python\n3.9 — where uv pins click 8.1.8 because click 8.3 requires 3.10+ —\n`is_flag=True, default=True` with `-r` inverts to False, so the recursive\nglob never find\n[…]\nssed on 3.10/3.11/3.12 but broke 3.9 CI.\n\nThe `-r` was redundant on every Python version; removing it is the\nminimal, correct fix.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "test(skill-doctor): drop redundant -r that trips click 8.1 flag quirk",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-17T07:09:53Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "ef462839bf062611c9ad385eaa4abc5a52694483",
"body": "… budget (#177)\n\nevalview skill doctor previously summed every valid skill description\ninto the 15k Claude Code character budget. Skills declared with\n`disable-model-invocation: true` are manual-only (never auto-invoked by\nthe model) so their descriptions are not loaded into the skill-\ndescription c\n[…]\nline when such skills are present\n- update docs/SKILLS_TESTING.md\n- add regression tests for parsing and budget-exclusion behavior\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(skill-doctor): exclude disable-model-invocation skills from char…",
"author_name": "Iraklis Georgas",
"author_login": "iraklisg",
"committed_at": "2026-04-17T06:56:51Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "e7f8743344e24f23da2f1e1a14a1d75138a806f4",
"body": "mypy does not narrow getattr(obj, name, None) truthy checks, so accessing\n.get() on anomaly_report / trust_report / coherence_report triggered\nunion-attr and dict-item errors. Switch to direct attribute access with\nan `is not None` guard to restore narrowing.\n\nCo-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(mypy): narrow Optional[dict] reports via direct is-not-None",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-17T06:46:12Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "38aa7441e775f516782b89b55fa8b76ede95c17d",
"body": "Fix crash bugs in replay pipeline (wrong adapter kwarg, wrong execute\narg type, dead code), extract shared ObservabilitySummary to eliminate\n6-way DRY violation across check/display/api/cloud/mcp/slack surfaces,\nfix logic bugs (hardcoded reference_turn, latency=0 false flags,\nsubstring extension mat\n[…]\ntemplate from to_dict() internals via normalizer layer, add\nTypedDict schemas for report types, and add 58 new tests (1706 total).\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor: production-harden PR #179 observability modules",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-16T19:16:17Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "54d6e0b3f7a84aa2da8e448538d1607dc7f057bc",
"body": "* feat: add agent observability and benchmark trust modules\n\nAddress the core gaps identified in market research:\n\n1. behavioral_anomalies.py — Silent failure detection via pattern analysis:\n tool loops, progress stalls, brittle recovery, excessive retries,\n skipped required steps. Catches \"look\n[…]\ne`\n\nThis completes 10/10 user-facing surface wiring.\n1583 tests pass, 0 regressions.\n\nhttps://claude.ai/code/session_01PdiCwZqKS9dXHcgWsSQNnJ\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Add observability modules for agent evaluation quality signals (#179)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-16T11:50:49Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "2a6b0e6bb281786e0286989ad19e23ece3b42158",
"body": "Follow-ups to #171:\n- Extract `_execute_agent_with_slow_warning` helper that measures real\n elapsed time via `time.monotonic()`, so the printed \"Xs elapsed\" stays\n accurate if the event loop slips instead of just echoing `timeout * 0.5`.\n- Apply the helper to `_execute_snapshot_tests` so `evalview\n[…]\narametrized test plus a\n dedicated elapsed-format assertion. Bumps the sleep margins so the\n tests are less flaky under CI load.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(slow-agent-warning): real elapsed time + snapshot path + DRY (#175)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-14T21:20:42Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "18381d35fefe8915e4adb57a24d919d5c2a1ffeb",
"body": "* Add slow-agent warning (fixes #142)\n\n* Add docstrings for slow-agent-warning functions",
"is_bot": false,
"headline": "Slow agent warning (#171)",
"author_name": "Gaurav Kumar Thakur",
"author_login": "gauravxthakur",
"committed_at": "2026-04-14T21:12:55Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ab9836c2ee6d6c535bd75aabc97a0b684c8f5c8b",
"body": "Bind monitor_cfg.incidents_path via a walrus on the truthiness guard so mypy sees a single narrowed local before passing it to Path(). Restores green Type Check on main after #172.",
"is_bot": false,
"headline": "fix(monitor): satisfy mypy on incidents_path narrowing (#173)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-14T08:33:21Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8905c78115f01b9f472309c41512bc3a35e3e821",
"body": "* feat(autopr): close the prod-incident → regression-test → PR loop\n\nShip the \"glue\" the README promises: `evalview monitor --incidents`\nnow writes one JSON record per confirmed regression to\n`.evalview/incidents.jsonl`, and the new `evalview autopr` command\nsynthesizes a pinned regression test YAML\n[…]\ncrubbed from the feature\nbranch's history too, that requires an explicit force-push.\n\nhttps://claude.ai/code/session_013CyqJYqkzaRQeXBFPNnBjD\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(autopr): close the prod-incident → regression-test → PR loop (#172)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-14T08:23:05Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "12e3d83edc04ca722361dd38344b05b4fe8e1026",
"body": "Drop the user-specific absolute aider binary path from the demo YAMLs\nand resolve it in the adapter instead: explicit kwarg > $AIDER_PATH >\n\"aider\" on PATH. Makes the test suite portable across machines and CI\nwithout every user patching the YAMLs.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor(aider): resolve aider_path from AIDER_PATH env var",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-14T06:26:38Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "3a3d1c67f33c5eceb092fc59b02ed889df9ab4d5",
"body": "Wraps `aider --message` as an AgentAdapter so Aider becomes a\nfirst-class drift-detection target in EvalView. Parses `Applied edit`\nmarkers and computes unified diffs against in-memory file snapshots,\nthen restores fixtures after each run so repeated snapshots are\nidempotent — a requirement for drif\n[…]\nrrored tasks under demo/tests/aider/ — refactor, bug-fix,\nimplement — pointing at freshly broken fixtures in demo/fixtures/aider/.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(adapters): add Aider CLI adapter + narrow deterministic test set",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-14T06:21:38Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "bf56cfae60c9f41d2fd8b363aa66b81075b1fae7",
"body": "Week 4 README pass — four targeted additions, no rewrite. The existing\nREADME was already 80% of the way to the Week-4 target state after\nyesterday's tweaks; two gaps needed closing.\n\nGap 1 — `evalview since` was completely absent from the README even\nthough it's the daily habit anchor for four of s\n[…]\nrison,\n Model Drift Detection, pyproject description, and everything\n below Production Monitoring are untouched as intended.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(readme): surface `evalview since` + verdict panel in the hero path",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T15:54:18Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "b687a9924e064e8242585efa89f6fe8ed8a4f539",
"body": "Self-review pass across the Week 3 progress/drift/slack-digest work\nand the noise-tracker follow-ups. Four issues fixed, three regression\ntests added. Nothing requires a refactor — the architecture holds up.\n\nnoise_tracker.ConfirmationGate:\n - Strict-bucket leak: a test previously pending in relaxe\n[…]\nom 1515 by 3 new regression tests, zero\nregressions). mypy clean on noise_tracker + slack_digest_cmd +\nprogress_cmd + monitor_cmd.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(noise): strict-bucket leak + digest markdown escape + doc drift",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T15:44:10Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "a0d429e8b057db9625bcc1fc4a54e547f4db0bf2",
"body": "…s (#170)\n\nEvery PR since #166 has shipped to main with the `Type Check` CI job\nfailing — no one was blocking merges on it, but the badge has been red\non every run. All 7 errors live in two commands that were added in the\nWeek 2-3 investigative-loop work:\n\nslack_digest_cmd.py\n - Reused the loop var\n[…]\npy + tests/test_noise_tracker.py\n → 70 passed (the suites that cover the touched code paths)\n\nhttps://claude.ai/code/session_017sc1uHapk5aRJjpCK81eJ7\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(mypy): resolve 7 Type Check CI failures in slack_digest + progres…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T13:26:35Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "b6b8812fa2b7dec0a8794c42023e565181869754",
"body": "… CI (#169)\n\nThe README linked to docs/CLI_REFERENCE.md for monitor config options\nbut the file had no evalview monitor section at all. This adds a full\nreference covering the confirmation gate, noise tracking, config keys,\nenv vars, prerequisites, and the related .evalview/noise.jsonl file.\n\nAlso w\n[…]\nis the\nprerequisite for unlocking the deferred repeat-suppression and\nauto-quarantine features.\n\nhttps://claude.ai/code/session_01KHpgDmMjZDHvjPDmQNCZEd\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(cli): add evalview monitor reference + wire monitor into dogfood…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T12:07:54Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "27735c93143e7d5342ca833197f3ef5d5479a376",
"body": "… Monitoring (#168)\n\nThe noise-reduction work shipped in #166 is invisible to users until they read\nthe source — add two short paragraphs to the Production Monitoring section so\nthe behavior change (n>=2 confirmation, gate:strict bypass, visible Noise\nsection in slack-digest) is discoverable from the README. Links to\nnoise_tracker.py for readers who want the full design.\n\nhttps://claude.ai/code/session_01LXXGAki7tURsKxBNmNfzXf\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(readme): explain confirmation gate + strict bypass in Production…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T12:03:16Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "769c17237684d2518623450248cf594a22a4de5d",
"body": "… Monitoring (#167)\n\nThe noise-reduction work shipped in #166 is invisible to users until they read\nthe source — add two short paragraphs to the Production Monitoring section so\nthe behavior change (n>=2 confirmation, gate:strict bypass, visible Noise\nsection in slack-digest) is discoverable from the README. Links to\nnoise_tracker.py for readers who want the full design.\n\nhttps://claude.ai/code/session_01LXXGAki7tURsKxBNmNfzXf\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(readme): explain confirmation gate + strict bypass in Production…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T11:52:50Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "ab29fe1738ed075d6d46f42ba4a079d48d67a595",
"body": "…c + strict bypass (#166)\n\n* feat(noise): confirmation gate, incident collapse, public noise metric\n\nShip the three highest-leverage noise-reduction moves for the beta:\n\n1. ConfirmationGate — never alert on n=1. A failure seen in a single\n monitor cycle is parked as \"pending\" and only promoted to \n[…]\ntest list with × count annotation, list capped at\n top-5 with \"and N more\" tail.\n\nhttps://claude.ai/code/session_01Pj3qevTSnyAWHEztiws44J\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(noise): confirmation gate, incident collapse, public noise metri…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T10:54:37Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "4112c399518841539d253d6f07b5f1b4775baef1",
"body": "…igest / quarantine (#164)\n\nREADME now documents the Week 2-3 workflow that was shipping in code but missing\nfrom the front door:\n\n- Lead paragraph calls out investigative loop + graded confidence\n- Quick Start shows the three new commands right after init/snapshot/check\n- New \"Daily Workflow\" secti\n[…]\nrdict, recommendation engine now sit above model-check\n- \"How It Works\" extended from 5 to 8 steps covering the new loop\n\nNo code changes — README-only.\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(readme): surface investigative loop — progress / drift / slack-d…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T07:02:46Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "88069438a68b527db5795db598616e2032bdbdce",
"body": "Week 3 of the 30-day push — adds the three commands that complete\nthe habit path (progress, drift, slack-digest), tightens the drift\nconfidence wire into the verdict layer, and injects a replay\nstability check when the verdict lands on INVESTIGATE.\n\nNew commands:\n\n - `evalview progress --since <sha\n[…]\nP0.1 regression guard).\n - BLOCK_RELEASE does NOT inject the stability rec.\n\nSuite: 1482 passed (up from 1444, zero regressions).\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: progress / drift / slack-digest commands + graded drift verdict",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-13T03:43:55Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "0005931ab7393f436701f8da81fd1415fab57893",
"body": "Week 1 + Week 2 of the 30-day \"release gate for AI agents\" push. Adds\nfour primitives that together turn `evalview check` from a diff tool\ninto a ship/don't-ship decision surface:\n\n 1. Release verdict layer (core/verdict.py)\n - Verdict enum: SAFE_TO_SHIP / SHIP_WITH_QUARANTINE / INVESTIGATE /\n \n[…]\n batch_update, markdown escape, code-fence fallback, stale_tests\n cap.\n\nSuite: 1444 passed (up from 1356, zero regressions).\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: release verdict layer, quarantine governance, and since brief",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-12T21:13:26Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "b43cb0900d7a8b30757d9b5c258fb74efaf32bcb",
"body": "- QuarantineStore.increment_flaky(): tracks how many times a test is\n classified as flaky across runs\n- QuarantineStore.should_quarantine(): returns True after 3+ flaky\n classifications — triggers auto-suggestion in CLI\n- check_display: suggests \"evalview quarantine add <test>\" when a test\n hits \n[…]\n\n re-creating per iteration\n- quarantine list: now shows flaky_count\n- Improved UX copy: \"block deployment\" instead of \"block CI\"\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: wire up flaky_count tracking, fix quarantine perf + UX",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-11T21:07:30Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "db3c2571bf7f84f54d71283d300d16d2b315adef",
"body": "- New evalview/core/quarantine.py: YAML-backed quarantine store\n- New evalview quarantine add/remove/list CLI commands\n- check_cmd: quarantined test failures excluded from exit code\n- check_display: quarantined tests shown with ⏸ badge, summary line\n- --strict flag overrides quarantine (all failures block)\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: flake quarantine — known-flaky tests don't block CI",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-11T20:49:31Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "84fbbc1f128e396232f053df192df77575669827",
"body": "- New evalview/core/recommendations.py: generates specific, actionable\n recommendations based on failure patterns (tool changes, score drops,\n hallucination, model drift, cost spikes, PII, safety)\n- Wired into CLI check_display.py: prints recommendations below each\n failing test's root cause anal\n[…]\n Wired into HTML reporter: renders recommendations in test detail view\n- recommend_from_trace_diff() adapter for TraceDiff objects\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: improvement recommendation engine",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-11T20:44:31Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "3679ca51b7994c4eff6d0d77e78f9477a41178d7",
"body": "Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore: bump node SDK to 0.6.2",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-11T20:26:18Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "f41c39090ab4d37849208f55df0c709780016e96",
"body": "- evalview model-check: closed-model drift detection with canary suite\n- DriftKind + DriftConfidence enums for unified drift taxonomy\n- model_snapshots: timestamped store with auto-pin and pruning\n- structural scorers: tool_choice, json_schema, refusal, exact_match\n- canary_suite loader with hash-pi\n[…]\napter_factory\n- 80 net new tests (snapshot, scorers, canary, integration)\n- TraceDiff gains drift_kind and drift_confidence fields\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore: bump version to 0.6.2",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-11T20:24:19Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "de83e3810be31a0b591a7df667178d0ae48bb1f8",
"body": "Replace **dict unpacking (typed as dict[str, float]) with explicit\nkeyword arguments so mypy can verify the medium_flip_count: int\nparameter correctly.\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: resolve mypy type errors in classify kwargs (CI fix)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-11T19:01:36Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "b7b61efa95e0a4b996643fb48fac147c483dbf72",
"body": "…st reduction (#161)\n\n* refactor(model-check): split cmd module, add negation-aware scoring, configurable thresholds, concurrency\n\nSix improvements to the model health check feature:\n\n1. Extract drift classification into core/drift_classifier.py and rendering\n into commands/model_check_render.py —\n[…]\nompt would change model behavior\nand invalidate existing snapshots.\n\n101 tests pass.\n\nhttps://claude.ai/code/session_01BnjZvicyfxaymmAAmCnQx4\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "Improve model-check: split modules, fix scoring, add concurrency & co…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-10T21:30:17Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "1f82292600b4411b77684fb82f836544fbaf5046",
"body": null,
"is_bot": false,
"headline": "fix: add jsonschema to core dependencies",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-10T09:41:55Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "f7504cf32c3269ffe315d06e014de3e9bf52ff62",
"body": "Code fixes:\n- scoring: replace greedy regex JSON extraction with bracket-counting\n _extract_first_json_object() — avoids false failures on multi-object\n responses and JSON wrapped in trailing prose\n- cmd: render actual cost (outcome.total_cost_usd) in header, not estimate\n- cmd: require BOTH snaps\n[…]\nte model-check to top row of Key Features table\n- Trim Quick Start blurb to a two-liner with link to the new section\n\nAll 80 unit tests pass.\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(model-check): 5 correctness fixes + world-class README section",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-10T07:33:58Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "c8ef600c9cd80f55c5cebd56a44cb51ce335c128",
"body": "* feat(core): unified DriftKind taxonomy + adapter factory cleanups\n\nFoundations for `evalview model-check` (closed-model drift detection):\n\n- Add core/drift_kind.py with DriftKind {NONE, MODEL, CONTRACT, BEHAVIORAL}\n and DriftConfidence {STRONG, MEDIUM, WEAK}. This is an orthogonal\n dimension to \n[…]\nampling, top_p=1.0 is redundant\nand has been removed from the _run_anthropic call.\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(model-check): closed-model drift detection command (#159)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-10T07:22:02Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "59e8d016060b465ff30e86d2eeb018d4ef7aece2",
"body": "…shold)\n\n- dogfood/agent.py: reduce subprocess timeout 60s → 30s so evalview run\n fails fast instead of blocking for a full minute\n- dogfood/test-cases/03-run-command.yaml: raise max_latency 60s → 180s to\n accommodate realistic CLI command execution time\n- tests/test_evaluator_accuracy.py: loosen \n[…]\n_scores_low\n threshold 20 → 40; gpt-5.4-mini scores \"I don't know.\" at 35 (honest\n but empty), still well below the passing threshold of 70\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: resolve daily dogfood failures (latency timeout + LLM score thre…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-09T23:27:52Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "0aaedb9e804dab43b94051fd764168d0c3969c8d",
"body": "- Switch Gemma demo tests from llamacpp to ollama/gemma4:26b\n- Rewrite start_gemma.sh to use ollama pull instead of llama-server\n- Add input_key + graph_config to LangGraph adapter for agents with\n non-messages input schemas (e.g. local-deep-researcher)\n- Pass input_key/graph_config through build_a\n[…]\n- Add langgraph to no_http_adapters to skip global connectivity pre-check\n- Parse running_summary/output/result fields from graph node updates\n so non-messages LangGraph agents return captured output",
"is_bot": false,
"headline": "feat: Ollama support + LangGraph adapter improvements",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-08T09:31:51Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "23347006915a0a5cf7514f84246b040e538cb246",
"body": "… flags\n\n- html_reporter.py: restore json/Counter/Dict/Any imports, plotly try/except\n block, and HTMLReporter = _DeprecatedHTMLReporter backward-compat alias\n- snapshot_cmd.py: add --no-judge flag, pass skip_llm_judge to evaluator\n- check_cmd.py: add --no-judge flag, pass skip_llm_judge to _execut\n[…]\nests\n- shared.py: add skip_llm_judge param to _execute_snapshot_tests\n- run/_cmd.py: add --timeout flag that overrides adapter timeout config\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: restore html_reporter imports/alias and add --no-judge/--timeout…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-08T08:41:09Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "02dfad3ee327e5845a94bce8d96aa0028a4b1c3b",
"body": "- Switch demo tests from ollama/ to llamacpp-gemma/llamacpp-qwen model IDs\n- Show \"free (local)\" instead of $0.0000 for opencode/goose/ollama adapters in console and summary\n- Suppress false $0 cost warning for local adapters in cost evaluator\n- Route `evalview report --html` and `evalview run --html` through new visual report generator\n- Expand HTML reporter and visualization generators with richer charts\n- Add benchmark_compare.html and demo start scripts for Gemma/Qwen",
"is_bot": false,
"headline": "feat: improve local model support + visual HTML report",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-06T21:11:31Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "bb080236b5b285885db3b518fb2681c096234d9f",
"body": "… Qwen3-Coder 30B)\n\n- New opencode_adapter.py: runs opencode CLI non-interactively, parses NDJSON\n event stream into ExecutionTrace with tool calls, latency, cost, diffs\n- Fixed shared.py: snapshot + check commands now support non-HTTP adapters\n (opencode, goose, etc.) via _build_adapter_for_tc he\n[…]\non step\n- demo/run_benchmark.sh: sequential local model execution to avoid Ollama contention\n- Sonnet baseline: 87.5–100/100, 13–33s per task\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add OpenCode adapter + model benchmark (Sonnet vs Gemma4 26B vs…",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-06T11:01:32Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "d61a6fe476da5c272c7884507ebf235e37692a6e",
"body": "- Add OpenCodeAdapter parsing NDJSON event stream from `opencode run --format json`\n- Maps tool_use events (read, edit, bash, glob) to ExecutionTrace steps\n- Captures diffs from edit tool metadata, per-step latency, token usage\n- Wire into adapter_factory.py, registry.py, and types.py validator\n- Ad\n[…]\n (bug-fix, refactor, implement YAMLs) for evalview run/snapshot/check\n- Add 10 unit tests covering NDJSON parsing, tool names, tokens, errors\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add OpenCode adapter with Gemma 4 / Qwen 3.5 demo fixtures",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-06T09:18:59Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "f2706a16ae029dc74c6d52aaae8d03010f0f4977",
"body": "dict[str, str] is incompatible with OpenAI SDK's ChatCompletion message union type.\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: use List[Any] for OpenAI messages to satisfy mypy",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-05T12:51:33Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "12f6844aec636827661a02d458fa00753a41559f",
"body": "Implements the pytest parametrize pattern for testing multiple models\non the same task side-by-side:\n\n @pytest.mark.parametrize(\"model\", [\"claude-opus-4-6\", \"gpt-4o\", \"claude-sonnet-4-6\"])\n def test_my_task(model):\n result = evalview.run_eval(model, query=\"summarize this contract\")\n \n[…]\nALVIEW_API_TOKEN is set\n- examples/model_comparison_test.py: full working example\n- README.md: new Model Comparison section with code samples\n\nCo-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add model comparison API (run_eval, compare_models, score)",
"author_name": "Hidai Bar-Mor",
"author_login": "hidai25",
"committed_at": "2026-04-05T12:41:51Z",
"body_truncated": true,
"is_coding_agent": true
}
],
"releases_count": 33,
"commits_last_year": 804,
"latest_release_at": "2026-07-26T19:53:19Z",
"latest_release_tag": "v0.8.1",
"releases_from_tags": false,
"days_since_last_push": 0,
"active_weeks_last_year": 31,
"days_since_latest_release": 0,
"mean_days_between_releases": 14.4
},
"community": {
"has_readme": true,
"has_license": true,
"has_description": true,
"has_contributing": true,
"health_percentage": 100,
"has_issue_template": false,
"has_code_of_conduct": true,
"has_pull_request_template": true
},
"ecosystem": {
"packages": [
{
"name": "evalview",
"exists": true,
"license": "Apache-2.0",
"keywords": [],
"ecosystem": "npm",
"matches_repo": null,
"registry_url": "https://www.npmjs.com/package/evalview",
"is_deprecated": false,
"latest_version": "0.8.0",
"repository_url": null,
"versions_count": 13,
"total_downloads": null,
"dependents_count": null,
"deprecation_note": null,
"maintainers_count": 1,
"monthly_downloads": 114,
"first_published_at": "2026-01-19T07:02:52.767000Z",
"latest_published_at": "2026-05-15T10:24:04.268000Z",
"latest_version_yanked": null,
"days_since_latest_publish": 72
},
{
"name": "evalview",
"exists": true,
"license": "Apache-2.0",
"keywords": [
"ai",
"agents",
"testing",
"evaluation",
"llm",
"langchain",
"langgraph",
"crewai",
"openai",
"anthropic",
"claude",
"huggingface",
"ollama",
"mcp",
"multi-agent",
"pytest-ai",
"ai-agent-testing",
"llm-testing",
"agent-evaluation",
"regression-testing",
"golden-baseline",
"ci-cd-testing",
"yaml-testing",
"tool-calling",
"agent-regression",
"llm-regression",
"skill-testing",
"mcp-contract-testing",
"pass-at-k",
"non-deterministic-testing",
"ai-ci-cd",
"Development Status :: 4 - Beta",
"Intended Audience :: Developers",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.10",
"Programming Language :: Python :: 3.11",
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.9",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Topic :: Software Development :: Quality Assurance",
"Topic :: Software Development :: Testing"
],
"ecosystem": "pypi",
"matches_repo": true,
"registry_url": "https://pypi.org/project/evalview/",
"is_deprecated": false,
"latest_version": "0.8.1",
"repository_url": "https://github.com/hidai25/eval-view.git",
"versions_count": 38,
"total_downloads": null,
"dependents_count": null,
"deprecation_note": null,
"maintainers_count": null,
"monthly_downloads": 2093,
"first_published_at": "2025-12-03T20:29:32.074034Z",
"latest_published_at": "2026-07-26T20:34:14.044393Z",
"latest_version_yanked": null,
"days_since_latest_publish": 0
}
]
},
"popularity": {
"forks": 21,
"stars": 124,
"watchers": 2,
"fork_history": {
"days": [
{
"date": "2025-12-11",
"count": 1
},
{
"date": "2025-12-16",
"count": 1
},
{
"date": "2025-12-26",
"count": 1
},
{
"date": "2026-01-26",
"count": 1
},
{
"date": "2026-02-23",
"count": 1
},
{
"date": "2026-03-01",
"count": 1
},
{
"date": "2026-03-03",
"count": 1
},
{
"date": "2026-03-05",
"count": 1
},
{
"date": "2026-03-07",
"count": 1
},
{
"date": "2026-03-11",
"count": 2
},
{
"date": "2026-03-12",
"count": 2
},
{
"date": "2026-03-17",
"count": 1
},
{
"date": "2026-03-25",
"count": 1
},
{
"date": "2026-03-31",
"count": 1
},
{
"date": "2026-04-15",
"count": 1
},
{
"date": "2026-04-16",
"count": 1
},
{
"date": "2026-04-18",
"count": 1
},
{
"date": "2026-04-21",
"count": 1
},
{
"date": "2026-06-15",
"count": 1
}
],
"complete": true,
"collected": 21,
"total_forks": 21
},
"star_history": null,
"open_issues_and_prs": 2
},
"ai_readiness": {
"has_nix": false,
"example_dirs": [
"examples"
],
"has_llms_txt": true,
"has_dockerfile": true,
"has_mcp_signal": true,
"bootstrap_files": [
"Makefile"
],
"api_schema_files": [],
"has_devcontainer": false,
"typecheck_configs": [
"evalview/py.typed",
"sdks/node/tsconfig.json"
],
"toolchain_manifests": [],
"largest_source_bytes": 78287,
"source_files_sampled": 370,
"oversized_source_files": 1,
"agent_instruction_files": [
"AGENTS.md",
"docs/AGENTS.md"
],
"agent_instruction_max_bytes": 10846
},
"dependencies": {
"manifests": [
"demo-agent/requirements.txt",
"package.json",
"pyproject.toml"
],
"advisories": {
"error": null,
"scope": "published_package",
"source": "osv",
"findings": [],
"collected": true,
"malicious": [],
"truncated": false,
"by_severity": {},
"advisory_count": 0,
"affected_count": 0,
"assessed_count": 60,
"malicious_count": 0,
"assessed_package": "npm:evalview@0.8.0",
"unassessed_count": 0,
"direct_affected_count": 0
},
"ecosystems": [
"npm",
"pypi"
],
"dependencies": [
{
"name": "commander",
"manifest": "package.json",
"ecosystem": "npm",
"version_constraint": "^11.1.0"
},
{
"name": "yaml",
"manifest": "package.json",
"ecosystem": "npm",
"version_constraint": "^2.3.4"
},
{
"name": "openai",
"manifest": "package.json",
"ecosystem": "npm",
"version_constraint": "^4.28.0"
},
{
"name": "chalk",
"manifest": "package.json",
"ecosystem": "npm",
"version_constraint": "^5.3.0"
},
{
"name": "ora",
"manifest": "package.json",
"ecosystem": "npm",
"version_constraint": "^8.0.1"
},
{
"name": "click",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=8.1.0"
},
{
"name": "pydantic",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=2.5.0"
},
{
"name": "pyyaml",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=6.0"
},
{
"name": "openai",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=1.12.0"
},
{
"name": "anthropic",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.39.0"
},
{
"name": "rich",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=13.7.0"
},
{
"name": "prompt_toolkit",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=3.0.0"
},
{
"name": "httpx",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.26.0"
},
{
"name": "python-dateutil",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=2.8.2"
},
{
"name": "python-dotenv",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=1.0.0"
},
{
"name": "jinja2",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=3.0"
},
{
"name": "jsonschema",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=4.0.0"
},
{
"name": "tomli",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=2.0"
}
],
"all_dependencies": {
"error": "GitHub dependency-graph SBOM unavailable (404); the dependency graph may be disabled for this repository",
"source": null,
"packages": [],
"collected": false,
"truncated": false,
"total_count": null,
"direct_count": null,
"indirect_count": null
}
},
"maintainership": {
"issues": {
"open_prs": 0,
"merged_prs": 93,
"open_issues": 2,
"closed_ratio": 0.987,
"closed_issues": 156,
"closed_unmerged_prs": 7
},
"bus_factor": 1,
"bot_contributors": 0,
"top_contributors": [
{
"type": "User",
"login": "hidai25",
"commits": 747,
"avatar_url": "https://avatars.githubusercontent.com/u/31502796?v=4"
},
{
"type": "User",
"login": "claude",
"commits": 27,
"avatar_url": "https://avatars.githubusercontent.com/u/81847?v=4"
},
{
"type": "User",
"login": "mvanhorn",
"commits": 6,
"avatar_url": "https://avatars.githubusercontent.com/u/455140?v=4"
},
{
"type": "User",
"login": "XJ789",
"commits": 6,
"avatar_url": "https://avatars.githubusercontent.com/u/204276375?v=4"
},
{
"type": "User",
"login": "gauravxthakur",
"commits": 6,
"avatar_url": "https://avatars.githubusercontent.com/u/68263456?v=4"
},
{
"type": "User",
"login": "illbeurs",
"commits": 3,
"avatar_url": "https://avatars.githubusercontent.com/u/91798939?v=4"
},
{
"type": "User",
"login": "gruckion",
"commits": 2,
"avatar_url": "https://avatars.githubusercontent.com/u/4559650?v=4"
},
{
"type": "User",
"login": "DanishShaikh18",
"commits": 1,
"avatar_url": "https://avatars.githubusercontent.com/u/109476012?v=4"
},
{
"type": "User",
"login": "iraklisg",
"commits": 1,
"avatar_url": "https://avatars.githubusercontent.com/u/3668360?v=4"
},
{
"type": "User",
"login": "ninjadevils4587",
"commits": 1,
"avatar_url": "https://avatars.githubusercontent.com/u/69968253?v=4"
}
],
"contributors_sampled": 13,
"top_contributor_share": 0.93
},
"quality_signals": {
"has_ci": true,
"has_tests": true,
"ci_workflows": [
"ci.yml",
"dogfood.yml",
"evalview.yml",
"publish.yml"
],
"has_docs_dir": true,
"linter_configs": [],
"has_editorconfig": false,
"has_linter_config": true,
"has_precommit_config": true
},
"security_signals": {
"lockfiles": [
"package-lock.json",
"uv.lock"
],
"scorecard": {
"checks": [
{
"name": "Binary-Artifacts",
"score": 10,
"reason": "no binaries found in the repo",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#binary-artifacts"
},
{
"name": "Branch-Protection",
"score": 1,
"reason": "branch protection is not maximal on development and all release branches",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#branch-protection"
},
{
"name": "CI-Tests",
"score": 10,
"reason": "24 out of 24 merged PRs checked by a CI test -- score normalized to 10",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#ci-tests"
},
{
"name": "CII-Best-Practices",
"score": 0,
"reason": "no effort to earn an OpenSSF best practices badge detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#cii-best-practices"
},
{
"name": "Code-Review",
"score": 1,
"reason": "Found 4/24 approved changesets -- score normalized to 1",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#code-review"
},
{
"name": "Contributors",
"score": 6,
"reason": "project has 2 contributing companies or organizations -- score normalized to 6",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#contributors"
},
{
"name": "Dangerous-Workflow",
"score": 10,
"reason": "no dangerous workflow patterns detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#dangerous-workflow"
},
{
"name": "Dependency-Update-Tool",
"score": 0,
"reason": "no update tool detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#dependency-update-tool"
},
{
"name": "Fuzzing",
"score": 0,
"reason": "project is not fuzzed",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#fuzzing"
},
{
"name": "License",
"score": 10,
"reason": "license file detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#license"
},
{
"name": "Maintained",
"score": 10,
"reason": "30 commit(s) and 7 issue activity found in the last 90 days -- score normalized to 10",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#maintained"
},
{
"name": "Packaging",
"score": null,
"reason": "packaging workflow not detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#packaging"
},
{
"name": "Pinned-Dependencies",
"score": 0,
"reason": "dependency not pinned by hash detected -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#pinned-dependencies"
},
{
"name": "SAST",
"score": 0,
"reason": "SAST tool is not run on all commits -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#sast"
},
{
"name": "Security-Policy",
"score": 10,
"reason": "security policy file detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#security-policy"
},
{
"name": "Signed-Releases",
"score": null,
"reason": "no releases found",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#signed-releases"
},
{
"name": "Token-Permissions",
"score": 0,
"reason": "detected GitHub workflow tokens with excessive permissions",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#token-permissions"
},
{
"name": "Vulnerabilities",
"score": 0,
"reason": "17 existing vulnerabilities detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#vulnerabilities"
}
],
"commit": "afbc7f6c715d7cd1ab1ea71aefd5e26a9bbcd767",
"ran_at": "2026-07-26T20:35:21Z",
"aggregate_score": 4.1,
"scorecard_version": "v5.5.0"
},
"has_codeql_workflow": false,
"has_security_policy": true,
"has_dependabot_config": false
},
"contribution_flow": {
"collected": true,
"ci_last_run_at": "2026-07-26T19:53:38Z",
"oldest_open_prs": [],
"last_merged_pr_at": "2026-07-26T19:49:41Z",
"ci_last_conclusion": "FAILURE",
"oldest_open_issues": [
{
"number": 148,
"created_at": "2026-04-01T04:36:23Z",
"last_comment_at": null,
"last_comment_author": null
},
{
"number": 240,
"created_at": "2026-05-24T07:24:49Z",
"last_comment_at": null,
"last_comment_author": null
}
]
}
},
"config": {
"disabled_metrics": [],
"disabled_categories": [],
"disabled_components": {}
},
"source": {
"url": "https://github.com/hidai25/eval-view",
"host": "github.com",
"name": "eval-view",
"owner": "hidai25"
},
"metrics": {
"overall": {
"key": "overall",
"band": "excellent",
"name": "Overall health",
"note": "The weighted overall 74 is calibrated to 88 on the published index scale (record calibration 2026-08-02).",
"notes": [
{
"code": "overall_calibration",
"params": {
"raw": 74,
"calibrated": 88,
"calibration": "2026-08-02"
}
}
],
"value": 88,
"inputs": {
"security": 53,
"vitality": 92,
"community": 64,
"governance": 62,
"calibration": "2026-08-02",
"engineering": 92,
"ai_readiness": 84,
"weighted_overall_raw": 74
},
"components": []
},
"categories": [
{
"key": "vitality",
"band": "excellent",
"name": "Vitality",
"value": 92,
"weight": 0.21,
"metrics": [
{
"key": "development_activity",
"band": "excellent",
"name": "Development activity",
"note": null,
"notes": [],
"value": 86,
"inputs": {
"commits_last_year": 804,
"human_commit_share": 1,
"days_since_last_push": 0,
"active_weeks_last_year": 31
},
"components": [
{
"key": "push_recency",
"name": "Push recency",
"detail": "last push 0 days ago",
"points": 36,
"status": "met",
"details": [
{
"code": "push_recency",
"params": {
"days": 0
}
}
],
"max_points": 36
},
{
"key": "commit_cadence",
"name": "Commit cadence",
"detail": "31/52 weeks with commits",
"points": 21.5,
"status": "partial",
"details": [
{
"code": "commit_cadence_weeks",
"params": {
"weeks": 31
}
}
],
"max_points": 36
},
{
"key": "commit_volume",
"name": "Commit volume",
"detail": "804 commits in the last year",
"points": 18,
"status": "met",
"details": [
{
"code": "commits_last_year",
"params": {
"count": 804
}
}
],
"max_points": 18
},
{
"key": "openssf_scorecard_maintained",
"name": "OpenSSF Scorecard: Maintained",
"detail": "30 commit(s) and 7 issue activity found in the last 90 days -- score normalized to 10",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
}
]
},
{
"key": "release_discipline",
"band": "exceptional",
"name": "Release discipline",
"note": "Excluded from scoring (no data or not applicable): OpenSSF Scorecard: Signed-Releases. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"openssf_scorecard_signed_releases"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 100,
"inputs": {
"releases_count": 33,
"latest_release_tag": "v0.8.1",
"releases_from_tags": false,
"days_since_latest_release": 0,
"mean_days_between_releases": 14.4
},
"components": [
{
"key": "ships_releases",
"name": "Ships releases",
"detail": "33 releases published",
"points": 27,
"status": "met",
"details": [
{
"code": "releases_published",
"params": {
"count": 33
}
}
],
"max_points": 27
},
{
"key": "release_recency",
"name": "Release recency",
"detail": "latest release 0 days ago",
"points": 36,
"status": "met",
"details": [
{
"code": "release_recency",
"params": {
"days": 0
}
}
],
"max_points": 36
},
{
"key": "release_cadence",
"name": "Release cadence",
"detail": "a release every ~14.4 days",
"points": 27,
"status": "met",
"details": [
{
"code": "release_cadence",
"params": {
"gap": 14.4
}
}
],
"max_points": 27
},
{
"key": "openssf_scorecard_signed_releases",
"name": "OpenSSF Scorecard: Signed-Releases",
"detail": "no releases found",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 10
}
]
},
{
"key": "abandonment",
"band": "exceptional",
"name": "Abandonment",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"cap": null,
"state": "maintained",
"guards": [],
"signals": [],
"red_flag": false,
"multiplier_pct": 100,
"declared_reason": null,
"unverified_reason": null,
"unanswered_open_prs": null,
"unanswered_open_issues": null,
"days_since_last_merged_pr": null,
"days_since_last_human_commit": 8,
"days_since_last_human_commit_is_floor": false
},
"components": [
{
"key": "project_is_still_maintained",
"name": "Project is still maintained",
"detail": "last human commit 8 days ago",
"points": 100,
"status": "met",
"details": [
{
"code": "abandonment_maintained",
"params": {
"days": 8
}
}
],
"max_points": 100
}
]
}
],
"description": "Is the project alive — is code being written and are releases shipping?"
},
{
"key": "community",
"band": "moderate",
"name": "Community & Adoption",
"value": 64,
"weight": 0.17,
"metrics": [
{
"key": "popularity",
"band": "weak",
"name": "Popularity & adoption",
"note": null,
"notes": [],
"value": 45,
"inputs": {
"forks": 21,
"stars": 124,
"watchers": 2,
"growth_state": "unverified",
"growth_factor_pct": 100,
"growth_unverified_reason": "no_history"
},
"components": [
{
"key": "stars",
"name": "Stars",
"detail": "124 stars",
"points": 33.9,
"status": "partial",
"details": [
{
"code": "stars",
"params": {
"count": 124
}
}
],
"max_points": 60
},
{
"key": "forks",
"name": "Forks",
"detail": "21 forks",
"points": 10.8,
"status": "partial",
"details": [
{
"code": "forks",
"params": {
"count": 21
}
}
],
"max_points": 25
},
{
"key": "watchers",
"name": "Watchers",
"detail": "2 watchers",
"points": 0,
"status": "missed",
"details": [
{
"code": "watchers",
"params": {
"count": 2
}
}
],
"max_points": 15
}
]
},
{
"key": "community_health",
"band": "excellent",
"name": "Community health",
"note": null,
"notes": [],
"value": 92,
"inputs": {
"has_readme": true,
"has_license": true,
"readme_badges": null,
"has_contributing": true,
"has_issue_template": false,
"has_code_of_conduct": true,
"readme_badge_services": [],
"has_pull_request_template": true
},
"components": [
{
"key": "readme",
"name": "README",
"detail": null,
"points": 22.5,
"status": "met",
"details": [],
"max_points": 22.5
},
{
"key": "license",
"name": "License",
"detail": "recognized license (Apache-2.0)",
"points": 22.5,
"status": "met",
"details": [
{
"code": "license_standard",
"params": {}
},
{
"code": "license_spdx",
"params": {
"spdx": "Apache-2.0"
}
}
],
"max_points": 22.5
},
{
"key": "contributing_guide",
"name": "CONTRIBUTING guide",
"detail": null,
"points": 18,
"status": "met",
"details": [],
"max_points": 18
},
{
"key": "code_of_conduct",
"name": "Code of conduct",
"detail": null,
"points": 13.5,
"status": "met",
"details": [],
"max_points": 13.5
},
{
"key": "issue_template",
"name": "Issue template",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.2
},
{
"key": "pr_template",
"name": "PR template",
"detail": null,
"points": 6.3,
"status": "met",
"details": [],
"max_points": 6.3
}
]
},
{
"key": "ecosystem_adoption",
"band": "moderate",
"name": "Ecosystem adoption (downloads)",
"note": "Excluded from scoring (no data or not applicable): Registry dependents. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"registry_dependents"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 56,
"inputs": {
"packages": [
"evalview",
"evalview"
],
"dependents": null,
"ecosystems": "npm, pypi",
"total_downloads": null,
"monthly_downloads": 2207
},
"components": [
{
"key": "monthly_downloads",
"name": "Monthly downloads",
"detail": "2,207 downloads/month across npm, pypi",
"points": 44.6,
"status": "partial",
"details": [
{
"code": "downloads_monthly",
"params": {
"count": 2207,
"ecosystems": "npm, pypi"
}
}
],
"max_points": 80
},
{
"key": "registry_dependents",
"name": "Registry dependents",
"detail": "not reported by this ecosystem",
"points": 0,
"status": "excluded",
"details": [
{
"code": "not_reported_by_this_ecosystem",
"params": {}
}
],
"max_points": 20
}
]
}
],
"description": "Does the project have users, downloads, attention, and a welcoming setup for contributors?"
},
{
"key": "governance",
"band": "moderate",
"name": "Sustainability & Governance",
"value": 62,
"weight": 0.23,
"metrics": [
{
"key": "maintainer_resilience",
"band": "at_risk",
"name": "Maintainer resilience (bus factor)",
"note": null,
"notes": [],
"value": 30,
"inputs": {
"bus_factor": 1,
"contributors_sampled": 13,
"top_contributor_share": 0.93
},
"components": [
{
"key": "bus_factor",
"name": "Bus factor",
"detail": "1 contributor(s) cover half of all commits",
"points": 9,
"status": "partial",
"details": [
{
"code": "bus_factor",
"params": {
"count": 1
}
}
],
"max_points": 54
},
{
"key": "commit_distribution",
"name": "Commit distribution",
"detail": "top contributor authored 93% of commits",
"points": 1.6,
"status": "partial",
"details": [
{
"code": "top_contributor_share",
"params": {
"share": 93
}
}
],
"max_points": 22.5
},
{
"key": "contributor_breadth",
"name": "Contributor breadth",
"detail": "13 contributors",
"points": 13.5,
"status": "met",
"details": [
{
"code": "contributors_sampled",
"params": {
"count": 13
}
}
],
"max_points": 13.5
},
{
"key": "openssf_scorecard_contributors",
"name": "OpenSSF Scorecard: Contributors",
"detail": "project has 2 contributing companies or organizations -- score normalized to 6",
"points": 6,
"status": "partial",
"details": [],
"max_points": 10
}
]
},
{
"key": "responsiveness",
"band": "excellent",
"name": "Issue & PR responsiveness",
"note": "Excluded from scoring (no data or not applicable): Newcomer PR acceptance. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"newcomer_pr_acceptance"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 81,
"inputs": {
"merged_prs": 93,
"open_issues": 2,
"closed_issues": 156,
"prs_merged_7d": null,
"prs_decided_7d": null,
"prs_merged_30d": null,
"prs_decided_30d": null,
"issue_closed_ratio": 0.987,
"closed_unmerged_prs": 7,
"first_time_authors_30d": null,
"first_time_prs_merged_30d": null,
"first_time_prs_decided_30d": null
},
"components": [
{
"key": "issue_resolution",
"name": "Issue resolution",
"detail": "99% of issues closed",
"points": 41.5,
"status": "partial",
"details": [
{
"code": "issues_closed_share",
"params": {
"share": 99
}
}
],
"max_points": 42
},
{
"key": "pr_acceptance",
"name": "PR acceptance",
"detail": "93/100 decided PRs merged",
"points": 27.9,
"status": "partial",
"details": [
{
"code": "decided_prs_merged",
"params": {
"merged": 93,
"decided": 100
}
}
],
"max_points": 30
},
{
"key": "newcomer_pr_acceptance",
"name": "Newcomer PR acceptance",
"detail": "no first-time contributor's PR decided in 30d",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_newcomer_prs",
"params": {
"days": 30
}
}
],
"max_points": 13
},
{
"key": "openssf_scorecard_code_review",
"name": "OpenSSF Scorecard: Code-Review",
"detail": "Found 4/24 approved changesets -- score normalized to 1",
"points": 1.5,
"status": "partial",
"details": [],
"max_points": 15
}
]
},
{
"key": "stewardship",
"band": "moderate",
"name": "Ownership & stewardship",
"note": "Excluded from scoring (no data or not applicable): Verified domain. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"verified_domain"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 53,
"inputs": {
"followers": 13,
"owner_type": "User",
"is_verified": null,
"owner_login": "hidai25",
"public_repos": 44,
"account_age_days": 3251
},
"components": [
{
"key": "ownership_backing",
"name": "Ownership backing",
"detail": "personal (user) account",
"points": 10,
"status": "partial",
"details": [
{
"code": "owner_personal",
"params": {}
}
],
"max_points": 30
},
{
"key": "verified_domain",
"name": "Verified domain",
"detail": "not applicable to user accounts",
"points": 0,
"status": "excluded",
"details": [
{
"code": "not_applicable_to_user_accounts",
"params": {}
}
],
"max_points": 20
},
{
"key": "owner_reach",
"name": "Owner reach",
"detail": "13 followers of hidai25",
"points": 8.2,
"status": "partial",
"details": [
{
"code": "owner_followers",
"params": {
"count": 13,
"login": "hidai25"
}
}
],
"max_points": 25
},
{
"key": "track_record",
"name": "Track record",
"detail": "44 public repos, account ~8 yr old",
"points": 24,
"status": "partial",
"details": [
{
"code": "public_repos",
"params": {
"count": 44
}
},
{
"code": "account_age_years",
"params": {
"years": 8
}
}
],
"max_points": 25
}
]
},
{
"key": "package_maintenance",
"band": "exceptional",
"name": "Package maintenance",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"packages": [
"evalview",
"evalview"
],
"ecosystems": "npm, pypi",
"any_deprecated": false,
"min_days_since_publish": 0
},
"components": [
{
"key": "published_resolvable",
"name": "Published & resolvable",
"detail": "2 package(s) on npm, pypi",
"points": 25,
"status": "met",
"details": [
{
"code": "packages_published",
"params": {
"count": 2,
"ecosystems": "npm, pypi"
}
}
],
"max_points": 25
},
{
"key": "publish_recency",
"name": "Publish recency",
"detail": "latest publish 0 days ago",
"points": 35,
"status": "met",
"details": [
{
"code": "publish_recency",
"params": {
"days": 0
}
}
],
"max_points": 35
},
{
"key": "version_history",
"name": "Version history",
"detail": "38 published versions",
"points": 20,
"status": "met",
"details": [
{
"code": "published_versions",
"params": {
"count": 38
}
}
],
"max_points": 20
},
{
"key": "not_deprecated",
"name": "Not deprecated",
"detail": "active, not deprecated or yanked",
"points": 20,
"status": "met",
"details": [
{
"code": "package_not_deprecated",
"params": {}
}
],
"max_points": 20
}
]
}
],
"description": "Will the project survive its people — bus factor, responsiveness, who backs it, and package upkeep?"
},
{
"key": "engineering",
"band": "excellent",
"name": "Engineering Quality",
"value": 92,
"weight": 0.19,
"metrics": [
{
"key": "engineering_practices",
"band": "exceptional",
"name": "Engineering practices",
"note": null,
"notes": [],
"value": 94,
"inputs": {
"has_ci": true,
"has_tests": true,
"has_editorconfig": false,
"has_linter_config": true,
"has_precommit_config": true
},
"components": [
{
"key": "ci_workflows",
"name": "CI workflows",
"detail": "4 workflow(s)",
"points": 24,
"status": "met",
"details": [
{
"code": "ci_workflows",
"params": {
"count": 4
}
}
],
"max_points": 24
},
{
"key": "tests_present",
"name": "Tests present",
"detail": null,
"points": 24,
"status": "met",
"details": [],
"max_points": 24
},
{
"key": "linter_config",
"name": "Linter config",
"detail": null,
"points": 16,
"status": "met",
"details": [],
"max_points": 16
},
{
"key": "pre_commit_hooks",
"name": "Pre-commit hooks",
"detail": null,
"points": 9.6,
"status": "met",
"details": [],
"max_points": 9.6
},
{
"key": "editorconfig",
"name": ".editorconfig",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 6.4
},
{
"key": "openssf_scorecard_ci_tests",
"name": "OpenSSF Scorecard: CI-Tests",
"detail": "24 out of 24 merged PRs checked by a CI test -- score normalized to 10",
"points": 20,
"status": "met",
"details": [],
"max_points": 20
}
]
},
{
"key": "documentation",
"band": "excellent",
"name": "Documentation",
"note": null,
"notes": [],
"value": 90,
"inputs": {
"topics": [
"agent-benchmark",
"agent-evaluation",
"ai-agents",
"crewai",
"evaluation",
"langgraph",
"openai-assistants",
"testing",
"pytest",
"anthropic",
"agentic-ai",
"langchain-agent",
"autogen",
"cli",
"llm",
"mcp",
"python",
"regression-testing"
],
"has_wiki": false,
"homepage": "https://evalview.com",
"has_readme": true,
"has_docs_dir": true,
"has_description": true
},
"components": [
{
"key": "readme",
"name": "README",
"detail": null,
"points": 30,
"status": "met",
"details": [],
"max_points": 30
},
{
"key": "documentation_directory",
"name": "Documentation directory",
"detail": null,
"points": 25,
"status": "met",
"details": [],
"max_points": 25
},
{
"key": "documentation_homepage_site",
"name": "Documentation / homepage site",
"detail": "https://evalview.com",
"points": 15,
"status": "met",
"details": [],
"max_points": 15
},
{
"key": "repository_description",
"name": "Repository description",
"detail": null,
"points": 10,
"status": "met",
"details": [],
"max_points": 10
},
{
"key": "topics",
"name": "Topics",
"detail": "18 topics",
"points": 10,
"status": "met",
"details": [
{
"code": "topics_count",
"params": {
"count": 18
}
}
],
"max_points": 10
},
{
"key": "wiki",
"name": "Wiki",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 10
}
]
}
],
"description": "Are baseline engineering and documentation practices in place?"
},
{
"key": "security",
"band": "moderate",
"name": "Security",
"value": 53,
"weight": 0.16,
"metrics": [
{
"key": "security_posture",
"band": "weak",
"name": "Security posture",
"note": "Excluded from scoring (no data or not applicable): Packaging, Signed-Releases. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"packaging",
"signed_releases"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 41,
"inputs": {
"source": "openssf_scorecard",
"checks_evaluated": 16,
"scorecard_version": "v5.5.0",
"checks_inconclusive": 2,
"scorecard_aggregate": 4.1
},
"components": [
{
"key": "binary_artifacts",
"name": "Binary-Artifacts",
"detail": "no binaries found in the repo",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "branch_protection",
"name": "Branch-Protection",
"detail": "branch protection is not maximal on development and all release branches",
"points": 0.8,
"status": "partial",
"details": [],
"max_points": 7.5
},
{
"key": "ci_tests",
"name": "CI-Tests",
"detail": "24 out of 24 merged PRs checked by a CI test -- score normalized to 10",
"points": 2.5,
"status": "met",
"details": [],
"max_points": 2.5
},
{
"key": "cii_best_practices",
"name": "CII-Best-Practices",
"detail": "no effort to earn an OpenSSF best practices badge detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 2.5
},
{
"key": "code_review",
"name": "Code-Review",
"detail": "Found 4/24 approved changesets -- score normalized to 1",
"points": 0.8,
"status": "partial",
"details": [],
"max_points": 7.5
},
{
"key": "contributors",
"name": "Contributors",
"detail": "project has 2 contributing companies or organizations -- score normalized to 6",
"points": 1.5,
"status": "partial",
"details": [],
"max_points": 2.5
},
{
"key": "dangerous_workflow",
"name": "Dangerous-Workflow",
"detail": "no dangerous workflow patterns detected",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
},
{
"key": "dependency_update_tool",
"name": "Dependency-Update-Tool",
"detail": "no update tool detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "fuzzing",
"name": "Fuzzing",
"detail": "project is not fuzzed",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "license",
"name": "License",
"detail": "license file detected",
"points": 2.5,
"status": "met",
"details": [],
"max_points": 2.5
},
{
"key": "maintained",
"name": "Maintained",
"detail": "30 commit(s) and 7 issue activity found in the last 90 days -- score normalized to 10",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "packaging",
"name": "Packaging",
"detail": "packaging workflow not detected",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 5
},
{
"key": "pinned_dependencies",
"name": "Pinned-Dependencies",
"detail": "dependency not pinned by hash detected -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "sast",
"name": "SAST",
"detail": "SAST tool is not run on all commits -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "security_policy",
"name": "Security-Policy",
"detail": "security policy file detected",
"points": 5,
"status": "met",
"details": [],
"max_points": 5
},
{
"key": "signed_releases",
"name": "Signed-Releases",
"detail": "no releases found",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 7.5
},
{
"key": "token_permissions",
"name": "Token-Permissions",
"detail": "detected GitHub workflow tokens with excessive permissions",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "vulnerabilities",
"name": "Vulnerabilities",
"detail": "17 existing vulnerabilities detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
}
]
},
{
"key": "dependency_advisories",
"band": "exceptional",
"name": "Dependency advisories",
"note": "Excluded from scoring (no data or not applicable): No advisories left outstanding. Remaining weights renormalized. Matched the npm:evalview@0.8.0 runtime dependency closure — what installing the published package pulls in — 60 packages. Reachability is not analyzed.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"no_advisories_left_outstanding"
]
}
},
{
"code": "weights_renormalized",
"params": {}
},
{
"code": "advisories_scope_published",
"params": {
"package": "npm:evalview@0.8.0",
"assessed": 60
}
},
{
"code": "advisories_reachability",
"params": {}
}
],
"value": 100,
"inputs": {
"source": "osv",
"advisories": 0,
"affected_packages": 0,
"assessed_packages": 60,
"unassessed_packages": 0,
"affected_by_severity": "none",
"direct_affected_packages": 0
},
"components": [
{
"key": "direct_dependencies_free_of_known_advisories",
"name": "Direct dependencies free of known advisories",
"detail": "no direct dependency carries a known advisory",
"points": 35,
"status": "met",
"details": [
{
"code": "no_direct_advisories",
"params": {}
}
],
"max_points": 35
},
{
"key": "indirect_dependencies_free_of_known_advisories",
"name": "Indirect dependencies free of known advisories",
"detail": "no indirect dependency carries a known advisory",
"points": 25,
"status": "met",
"details": [
{
"code": "no_indirect_advisories",
"params": {}
}
],
"max_points": 25
},
{
"key": "no_advisories_left_outstanding",
"name": "No advisories left outstanding",
"detail": "no advisory carries a publication date",
"points": 0,
"status": "excluded",
"details": [
{
"code": "advisories_no_publication_date",
"params": {}
}
],
"max_points": 40
}
]
},
{
"key": "malicious_dependencies",
"band": "exceptional",
"name": "Malicious dependencies",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"source": "osv",
"meaning": "reported as a malicious package by the OpenSSF corpus; the remedy is removal or moving off the compromised name, never an upgrade of the same artifact. Versions the registry has since pulled are listed but not scored",
"packages": [],
"red_flag": false,
"assessed_packages": 60,
"malicious_packages": 0,
"direct_malicious_packages": 0,
"withdrawn_malicious_packages": 0,
"installable_malicious_packages": 0
},
"components": [
{
"key": "no_dependency_reported_as_a_malicious_package",
"name": "No dependency reported as a malicious package",
"detail": "no dependency is reported as a malicious package",
"points": 100,
"status": "met",
"details": [
{
"code": "no_malicious_dependencies",
"params": {}
}
],
"max_points": 100
}
]
},
{
"key": "high_risk_jurisdiction_exposure",
"band": "exceptional",
"name": "High-Risk Jurisdiction Exposure",
"note": "Only high-confidence self-published location evidence affects this multiplier. Ambiguous matches are review-only; country evidence is not proof of nationality, citizenship, legal registration, malicious intent, or sanctions status.",
"notes": [
{
"code": "jurisdiction_evidence_limits",
"params": {}
}
],
"value": 100,
"inputs": {
"meaning": "self-published location evidence; not nationality or citizenship",
"red_flag": false,
"exposures": [],
"policy_countries": [
"Russia",
"Iran",
"North Korea"
],
"commit_weight_rule": {
"min_commits": 50,
"min_commit_share": 0.1
},
"review_only_matches": 0,
"below_threshold_exposures": [],
"assessed_self_published_locations": 6
},
"components": [
{
"key": "policy_exposure_multiplier",
"name": "Policy exposure multiplier",
"detail": "no confirmed policy-scope location match",
"points": 100,
"status": "met",
"details": [
{
"code": "jurisdiction_no_match",
"params": {}
}
],
"max_points": 100
}
]
}
],
"description": "Are visible security and supply-chain practices strong, with no malicious dependency and no unresolved high-risk jurisdiction exposure?"
},
{
"key": "ai_readiness",
"band": "excellent",
"name": "AI Readiness",
"value": 84,
"weight": 0.04,
"metrics": [
{
"key": "ai_agent_context",
"band": "exceptional",
"name": "Agent context & guidance",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"has_llms_txt": true,
"legible_history_share": 1,
"agent_instruction_files": [
"AGENTS.md",
"docs/AGENTS.md"
],
"agent_instruction_max_bytes": 10846
},
"components": [
{
"key": "agent_instructions",
"name": "Agent instructions",
"detail": "AGENTS.md, docs/AGENTS.md",
"points": 45,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "AGENTS.md, docs/AGENTS.md"
}
}
],
"max_points": 45
},
{
"key": "machine_readable_docs_llms_txt",
"name": "Machine-readable docs (llms.txt)",
"detail": "llms.txt present",
"points": 15,
"status": "met",
"details": [
{
"code": "llms_txt_present",
"params": {}
}
],
"max_points": 15
},
{
"key": "legible_commit_history",
"name": "Legible commit history",
"detail": "100 of 100 human commits state their intent (structured subject or explanatory body)",
"points": 40,
"status": "met",
"details": [
{
"code": "legible_history",
"params": {
"legible": 100,
"sampled": 100
}
}
],
"max_points": 40
}
]
},
{
"key": "ai_verify_loop",
"band": "excellent",
"name": "Verify loop (build / test / typecheck)",
"note": null,
"notes": [],
"value": 82,
"inputs": {
"has_nix": false,
"has_tests": true,
"lockfiles": [
"package-lock.json",
"uv.lock"
],
"has_dockerfile": true,
"typed_language": false,
"bootstrap_files": [
"Makefile"
],
"has_devcontainer": false,
"has_linter_config": true,
"typecheck_configs": [
"evalview/py.typed",
"sdks/node/tsconfig.json"
],
"agent_commit_share": 0.84,
"toolchain_manifests": [],
"dependency_bot_commit_share": 0
},
"components": [
{
"key": "one_command_bootstrap",
"name": "One-command bootstrap",
"detail": "Makefile",
"points": 18,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "Makefile"
}
}
],
"max_points": 18
},
{
"key": "automated_tests",
"name": "Automated tests",
"detail": null,
"points": 22,
"status": "met",
"details": [],
"max_points": 22
},
{
"key": "lint_format_config",
"name": "Lint / format config",
"detail": null,
"points": 11,
"status": "met",
"details": [],
"max_points": 11
},
{
"key": "static_type_checking",
"name": "Static type checking",
"detail": "evalview/py.typed, sdks/node/tsconfig.json",
"points": 11,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "evalview/py.typed, sdks/node/tsconfig.json"
}
}
],
"max_points": 11
},
{
"key": "reproducible_environment",
"name": "Reproducible environment",
"detail": "Dockerfile, lockfile",
"points": 10,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "Dockerfile, lockfile"
}
}
],
"max_points": 10
},
{
"key": "demonstrated_agent_practice",
"name": "Demonstrated agent practice",
"detail": "84 of the last 100 commits agent-authored or agent-credited",
"points": 10,
"status": "met",
"details": [
{
"code": "agent_authored_commits",
"params": {
"count": 84,
"sampled": 100
}
}
],
"max_points": 10
},
{
"key": "automated_maintenance",
"name": "Automated maintenance",
"detail": "no automated dependency updates observed",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_dependency_automation",
"params": {}
}
],
"max_points": 8
},
{
"key": "openssf_scorecard_pinned_dependencies",
"name": "OpenSSF Scorecard: Pinned-Dependencies",
"detail": "dependency not pinned by hash detected -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 10
}
]
},
{
"key": "ai_code_legibility",
"band": "excellent",
"name": "Code legibility for models",
"note": null,
"notes": [],
"value": 82,
"inputs": {
"primary_language": "Python",
"largest_source_bytes": 78287,
"source_files_sampled": 370,
"oversized_source_files": 1
},
"components": [
{
"key": "type_checkable_code",
"name": "Type-checkable code",
"detail": "Python with type-check config (evalview/py.typed, sdks/node/tsconfig.json)",
"points": 27,
"status": "partial",
"details": [
{
"code": "typecheck_config_language",
"params": {
"files": "evalview/py.typed, sdks/node/tsconfig.json",
"language": "Python"
}
}
],
"max_points": 45
},
{
"key": "manageable_file_sizes",
"name": "Manageable file sizes",
"detail": "1/370 source files over 60KB",
"points": 54.9,
"status": "partial",
"details": [
{
"code": "oversized_source_files",
"params": {
"kb": 60,
"sampled": 370,
"oversized": 1
}
}
],
"max_points": 55
}
]
},
{
"key": "ai_interfaces",
"band": "moderate",
"name": "Machine-readable interfaces",
"note": null,
"notes": [],
"value": 60,
"inputs": {
"example_dirs": [
"examples"
],
"has_mcp_signal": true,
"api_schema_files": []
},
"components": [
{
"key": "api_schema_openapi_graphql_proto",
"name": "API schema (OpenAPI/GraphQL/proto)",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 40
},
{
"key": "mcp_server",
"name": "MCP server",
"detail": null,
"points": 20,
"status": "met",
"details": [],
"max_points": 20
},
{
"key": "runnable_examples",
"name": "Runnable examples",
"detail": "examples",
"points": 40,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "examples"
}
}
],
"max_points": 40
}
]
}
],
"description": "How well is the repo equipped to be developed and maintained with AI coding agents? Carries a deliberately small weight: agent tooling is a real maintenance signal, but its absence must never gate the top of the scale (calibration saturates at raw 91, so 100/100 remains reachable with AI Readiness at zero)."
}
],
"classification": {
"top": [
"library",
"application"
],
"labels": [
"library",
"cli"
],
"scores": {
"cli": 10,
"library": 12,
"mcp-server": 3
},
"primary": "library",
"evidence": [
{
"tier": "distribution",
"label": "library",
"source": "registry:npm",
"weight": 6
},
{
"tier": "distribution",
"label": "library",
"source": "registry:pypi",
"weight": 6
},
{
"tier": "dependencies",
"label": "cli",
"source": "dep:click",
"weight": 4
},
{
"tier": "dependencies",
"label": "cli",
"source": "dep:commander",
"weight": 4
},
{
"tier": "structure",
"label": "mcp-server",
"source": "mcp_signal",
"weight": 3
},
{
"tier": "tags",
"label": "cli",
"source": "tag:cli",
"weight": 2
}
],
"artifacts": [],
"confidence": "medium",
"host_extension": false,
"runs_as_process": true,
"consumed_by_code": true
},
"metrics_version": "2.5.0"
},
"warnings": [
"Star history unavailable: GitHub GraphQL error: Resource not accessible by personal access token",
"GitHub dependency-graph SBOM unavailable (404); the dependency graph may be disabled for this repository"
],
"report_type": "repository",
"generated_at": "2026-07-26T20:35:41.543718Z",
"schema_version": "0.27.0",
"badge_url": "https://raw.githubusercontent.com/inspect-software/badges/main/v1/h/hidai25/eval-view.svg",
"full_name": "hidai25/eval-view",
"license_state": "standard",
"license_spdx": "Apache-2.0"
}