Informe JSON sin procesar legible por máquina
{
"data": {
"repo": {
"topics": [
"benchmarks",
"evaluation",
"gui-automation",
"openadapt",
"python"
],
"is_fork": false,
"size_kb": 105125,
"has_wiki": true,
"homepage": "https://pypi.org/project/openadapt-evals/",
"languages": {
"HTML": 30368,
"Shell": 21567,
"Python": 4288843,
"Batchfile": 1496,
"Dockerfile": 17002
},
"pushed_at": "2026-07-28T03:40:14Z",
"created_at": "2026-01-16T22:35:21Z",
"owner_type": "Organization",
"updated_at": "2026-07-28T03:40:30Z",
"description": "Evaluation infrastructure for GUI agent benchmarks",
"is_archived": false,
"is_disabled": false,
"license_spdx": "MIT",
"default_branch": "main",
"license_spdx_raw": "MIT",
"primary_language": "Python",
"significant_languages": [
"Python"
]
},
"owner": {
"blog": "https://openadapt.ai/",
"name": "OpenAdapt.AI",
"type": "Organization",
"login": "OpenAdaptAI",
"company": null,
"location": null,
"followers": 99,
"avatar_url": "https://avatars.githubusercontent.com/u/132681217?v=4",
"created_at": "2023-05-05T14:01:00Z",
"is_verified": null,
"public_repos": 54,
"account_age_days": 1179
},
"license": {
"state": "standard",
"spdx_id": "MIT",
"raw_spdx": "MIT",
"file_present": true,
"scorecard_found": true,
"profile_has_license": true
},
"activity": {
"releases": [
{
"tag": "v0.90.2",
"kind": "patch",
"published_at": "2026-07-28T03:40:15Z"
},
{
"tag": "v0.90.1",
"kind": "patch",
"published_at": "2026-07-28T02:34:48Z"
},
{
"tag": "v0.90.0",
"kind": "minor",
"published_at": "2026-07-27T06:44:59Z"
},
{
"tag": "v0.89.1",
"kind": "patch",
"published_at": "2026-07-16T00:22:05Z"
},
{
"tag": "v0.89.0",
"kind": "minor",
"published_at": "2026-07-14T14:10:55Z"
},
{
"tag": "v0.88.0",
"kind": "minor",
"published_at": "2026-07-14T04:25:05Z"
},
{
"tag": "v0.87.2",
"kind": "patch",
"published_at": "2026-07-10T19:34:59Z"
},
{
"tag": "v0.87.1",
"kind": "patch",
"published_at": "2026-06-13T00:19:04Z"
},
{
"tag": "v0.87.0",
"kind": "minor",
"published_at": "2026-04-01T15:29:14Z"
},
{
"tag": "v0.86.0",
"kind": "minor",
"published_at": "2026-04-01T15:26:12Z"
},
{
"tag": "v0.85.0",
"kind": "minor",
"published_at": "2026-03-31T22:47:44Z"
},
{
"tag": "v0.84.0",
"kind": "minor",
"published_at": "2026-03-31T22:39:02Z"
},
{
"tag": "v0.83.0",
"kind": "minor",
"published_at": "2026-03-31T22:15:40Z"
},
{
"tag": "v0.82.4",
"kind": "patch",
"published_at": "2026-03-30T16:58:14Z"
},
{
"tag": "v0.82.3",
"kind": "patch",
"published_at": "2026-03-29T23:48:00Z"
},
{
"tag": "v0.82.2",
"kind": "patch",
"published_at": "2026-03-29T23:38:13Z"
},
{
"tag": "v0.82.1",
"kind": "patch",
"published_at": "2026-03-29T23:25:45Z"
},
{
"tag": "v0.82.0",
"kind": "minor",
"published_at": "2026-03-29T23:05:10Z"
},
{
"tag": "v0.81.9",
"kind": "patch",
"published_at": "2026-03-29T22:30:14Z"
},
{
"tag": "v0.81.8",
"kind": "patch",
"published_at": "2026-03-29T22:07:11Z"
},
{
"tag": "v0.81.7",
"kind": "patch",
"published_at": "2026-03-29T21:41:04Z"
},
{
"tag": "v0.81.6",
"kind": "patch",
"published_at": "2026-03-29T21:24:17Z"
},
{
"tag": "v0.81.5",
"kind": "patch",
"published_at": "2026-03-29T20:47:20Z"
},
{
"tag": "v0.81.4",
"kind": "patch",
"published_at": "2026-03-29T20:44:23Z"
},
{
"tag": "v0.81.3",
"kind": "patch",
"published_at": "2026-03-29T20:16:24Z"
},
{
"tag": "v0.81.2",
"kind": "patch",
"published_at": "2026-03-29T19:09:22Z"
},
{
"tag": "v0.81.1",
"kind": "patch",
"published_at": "2026-03-29T18:22:43Z"
},
{
"tag": "v0.81.0",
"kind": "minor",
"published_at": "2026-03-29T18:17:30Z"
},
{
"tag": "v0.80.2",
"kind": "patch",
"published_at": "2026-03-29T17:22:24Z"
},
{
"tag": "v0.80.1",
"kind": "patch",
"published_at": "2026-03-29T17:06:24Z"
},
{
"tag": "v0.80.0",
"kind": "minor",
"published_at": "2026-03-29T16:37:46Z"
},
{
"tag": "v0.79.2",
"kind": "patch",
"published_at": "2026-03-29T16:14:23Z"
},
{
"tag": "v0.79.1",
"kind": "patch",
"published_at": "2026-03-29T16:07:57Z"
},
{
"tag": "v0.79.0",
"kind": "minor",
"published_at": "2026-03-29T15:57:49Z"
},
{
"tag": "v0.78.2",
"kind": "patch",
"published_at": "2026-03-29T15:49:45Z"
},
{
"tag": "v0.78.1",
"kind": "patch",
"published_at": "2026-03-29T15:30:55Z"
},
{
"tag": "v0.78.0",
"kind": "minor",
"published_at": "2026-03-29T15:16:12Z"
},
{
"tag": "v0.77.5",
"kind": "patch",
"published_at": "2026-03-29T14:39:23Z"
},
{
"tag": "v0.77.4",
"kind": "patch",
"published_at": "2026-03-29T14:03:15Z"
},
{
"tag": "v0.77.3",
"kind": "patch",
"published_at": "2026-03-29T13:55:05Z"
},
{
"tag": "v0.77.2",
"kind": "patch",
"published_at": "2026-03-29T05:03:11Z"
},
{
"tag": "v0.77.1",
"kind": "patch",
"published_at": "2026-03-29T04:53:20Z"
},
{
"tag": "v0.77.0",
"kind": "minor",
"published_at": "2026-03-29T04:31:39Z"
},
{
"tag": "v0.76.3",
"kind": "patch",
"published_at": "2026-03-29T04:16:43Z"
},
{
"tag": "v0.76.2",
"kind": "patch",
"published_at": "2026-03-28T23:02:48Z"
},
{
"tag": "v0.76.1",
"kind": "patch",
"published_at": "2026-03-28T22:54:02Z"
},
{
"tag": "v0.76.0",
"kind": "minor",
"published_at": "2026-03-28T22:34:58Z"
},
{
"tag": "v0.75.0",
"kind": "minor",
"published_at": "2026-03-28T21:38:14Z"
},
{
"tag": "v0.74.1",
"kind": "patch",
"published_at": "2026-03-28T21:29:42Z"
},
{
"tag": "v0.74.0",
"kind": "minor",
"published_at": "2026-03-28T21:20:22Z"
},
{
"tag": "v0.73.0",
"kind": "minor",
"published_at": "2026-03-28T20:33:03Z"
},
{
"tag": "v0.72.9",
"kind": "patch",
"published_at": "2026-03-28T20:22:04Z"
},
{
"tag": "v0.72.8",
"kind": "patch",
"published_at": "2026-03-28T20:18:52Z"
},
{
"tag": "v0.72.7",
"kind": "patch",
"published_at": "2026-03-28T20:06:58Z"
},
{
"tag": "v0.72.6",
"kind": "patch",
"published_at": "2026-03-28T19:55:18Z"
},
{
"tag": "v0.72.5",
"kind": "patch",
"published_at": "2026-03-28T19:28:54Z"
},
{
"tag": "v0.72.4",
"kind": "patch",
"published_at": "2026-03-28T19:20:08Z"
},
{
"tag": "v0.72.3",
"kind": "patch",
"published_at": "2026-03-28T19:07:07Z"
},
{
"tag": "v0.72.2",
"kind": "patch",
"published_at": "2026-03-28T18:52:00Z"
},
{
"tag": "v0.72.1",
"kind": "patch",
"published_at": "2026-03-28T18:47:32Z"
},
{
"tag": "v0.72.0",
"kind": "minor",
"published_at": "2026-03-28T16:44:50Z"
},
{
"tag": "v0.71.3",
"kind": "patch",
"published_at": "2026-03-28T16:33:58Z"
},
{
"tag": "v0.71.2",
"kind": "patch",
"published_at": "2026-03-28T04:32:53Z"
},
{
"tag": "v0.71.1",
"kind": "patch",
"published_at": "2026-03-27T17:32:55Z"
},
{
"tag": "v0.71.0",
"kind": "minor",
"published_at": "2026-03-26T21:07:53Z"
},
{
"tag": "v0.70.2",
"kind": "patch",
"published_at": "2026-03-26T16:32:45Z"
},
{
"tag": "v0.70.1",
"kind": "patch",
"published_at": "2026-03-26T16:26:15Z"
},
{
"tag": "v0.70.0",
"kind": "minor",
"published_at": "2026-03-24T21:31:44Z"
},
{
"tag": "v0.69.1",
"kind": "patch",
"published_at": "2026-03-24T04:08:21Z"
},
{
"tag": "v0.69.0",
"kind": "minor",
"published_at": "2026-03-24T02:28:41Z"
},
{
"tag": "v0.68.0",
"kind": "minor",
"published_at": "2026-03-23T20:33:01Z"
},
{
"tag": "v0.67.0",
"kind": "minor",
"published_at": "2026-03-23T20:04:12Z"
},
{
"tag": "v0.66.0",
"kind": "minor",
"published_at": "2026-03-23T19:58:57Z"
},
{
"tag": "v0.65.0",
"kind": "minor",
"published_at": "2026-03-23T19:45:29Z"
},
{
"tag": "v0.64.1",
"kind": "patch",
"published_at": "2026-03-23T15:49:51Z"
},
{
"tag": "v0.64.0",
"kind": "minor",
"published_at": "2026-03-23T00:56:22Z"
},
{
"tag": "v0.63.0",
"kind": "minor",
"published_at": "2026-03-22T22:45:14Z"
},
{
"tag": "v0.62.0",
"kind": "minor",
"published_at": "2026-03-22T19:09:09Z"
},
{
"tag": "v0.61.0",
"kind": "minor",
"published_at": "2026-03-22T17:21:29Z"
},
{
"tag": "v0.60.0",
"kind": "minor",
"published_at": "2026-03-22T16:53:40Z"
},
{
"tag": "v0.59.3",
"kind": "patch",
"published_at": "2026-03-22T06:10:09Z"
},
{
"tag": "v0.59.2",
"kind": "patch",
"published_at": "2026-03-22T04:46:02Z"
},
{
"tag": "v0.59.1",
"kind": "patch",
"published_at": "2026-03-22T03:56:25Z"
},
{
"tag": "v0.59.0",
"kind": "minor",
"published_at": "2026-03-21T22:27:54Z"
},
{
"tag": "v0.58.1",
"kind": "patch",
"published_at": "2026-03-21T20:27:03Z"
},
{
"tag": "v0.58.0",
"kind": "minor",
"published_at": "2026-03-21T19:22:20Z"
},
{
"tag": "v0.57.0",
"kind": "minor",
"published_at": "2026-03-21T19:00:15Z"
},
{
"tag": "v0.56.0",
"kind": "minor",
"published_at": "2026-03-21T18:48:44Z"
},
{
"tag": "v0.55.0",
"kind": "minor",
"published_at": "2026-03-21T18:40:34Z"
},
{
"tag": "v0.54.0",
"kind": "minor",
"published_at": "2026-03-21T18:04:03Z"
},
{
"tag": "v0.53.0",
"kind": "minor",
"published_at": "2026-03-21T17:47:53Z"
},
{
"tag": "v0.52.0",
"kind": "minor",
"published_at": "2026-03-21T17:19:36Z"
},
{
"tag": "v0.51.1",
"kind": "patch",
"published_at": "2026-03-21T16:29:23Z"
},
{
"tag": "v0.51.0",
"kind": "minor",
"published_at": "2026-03-21T01:07:18Z"
},
{
"tag": "v0.50.1",
"kind": "patch",
"published_at": "2026-03-20T18:34:08Z"
},
{
"tag": "v0.50.0",
"kind": "minor",
"published_at": "2026-03-20T17:43:09Z"
},
{
"tag": "v0.49.0",
"kind": "minor",
"published_at": "2026-03-20T17:15:18Z"
},
{
"tag": "v0.48.5",
"kind": "patch",
"published_at": "2026-03-20T16:11:28Z"
},
{
"tag": "v0.48.4",
"kind": "patch",
"published_at": "2026-03-20T15:54:47Z"
},
{
"tag": "v0.48.3",
"kind": "patch",
"published_at": "2026-03-20T01:37:08Z"
}
],
"recent_commits": [
{
"oid": "43ec260095ad60cb2c46527a72b9f857d56701bd",
"body": null,
"is_bot": false,
"headline": "chore: release 0.90.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-28T03:40:12Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "6a1518026e49212038b804e31d1a1aa4a8d22dca",
"body": "…uced a number (#277)\n\nA lint sweep exposed one failure class across this repo: a failure rendered as\na successful empty result. In an evaluation repo that does not crash anything\n-- it publishes a WRONG NUMBER. An unreachable backend that scores 0% and is\nreported as a legitimate 0% is the worst in\n[…]\nerror`,\n`exit 0`, or `set +e` anywhere. All four workflows fail loudly.\n\n\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\nCo-authored-by: Claude Opus 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix(evals): make \"could not measure\" representable everywhere it prod…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-28T03:38:45Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "4dc70d0544fa11d98c49a44b19460d0710b8b273",
"body": "`[tool.ruff]` has existed here for a long time with nothing running it. By the\ntime PR #275 looked, the tree had drifted to 478 findings against its own\ndeclared rule set -- among them three undefined names that no reviewer would\nhave let through if anything had been checking. #275 cleared all of th\n[…]\nassed!`\n\n🤖 Generated with [Claude Code](https://claude.com/claude-code)\n\n\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\nCo-authored-by: Claude Opus 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "ci: enforce the ruff config this repo already declares (#276)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-28T02:48:13Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "9268f4f87edd29f71b0272c7e7231246a6eeb646",
"body": null,
"is_bot": false,
"headline": "chore: release 0.90.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-28T02:34:45Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "c4b7e9bf3e7b6215da07ae6bbf1681c4f5026d59",
"body": "…ound (#275)\n\n`ruff>=0.1.0` was unbounded and `[tool.ruff]` set only `line-length`, so the\neffective rule set was whatever ruff the resolver happened to pick. ruff\n0.16.0 grew its DEFAULT set from 59 rules to 413; this repo silently went from\nits old implicit defaults to 1653 findings without a code\n[…]\nevery package module, and `ruff check .` clean under the locked 0.16.0.\n\n\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\nCo-authored-by: Claude Opus 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: bound ruff, declare an explicit lint rule set, and fix what it f…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-28T02:33:13Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "8bd4afe31615423efc2929d34d89834b7d21bdb3",
"body": "`.private/` is the workspace-wide convention for material that must never be\npublished. It was not ignored here, so a directory created inside this checkout\nwas one stray `git add` from being committed. Ignore it mechanically rather\nthan relying on that never happening.\n\n\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\nCo-authored-by: Claude Opus 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "chore: gitignore .private/ (#274)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-28T00:03:52Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "18010273c834bebc78b3204511069959b7d74bc6",
"body": "…#273)\n\n* docs(evidence): correct the 1.16.1 vs 1.24.0 apples-to-apples claim\n\n`REPRODUCE.md` stated that the two measurements \"differ only in the engine\nunder test\". That is false. MockMed ships *inside* the Flow wheel\n(`openadapt_flow/mockmed/`), so repinning the wheel repins the engine AND the\nap\n[…]\n>\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\n* fix(evidence): bind reproduction to measured Flow release\n\n---------\n\nCo-authored-by: Claude Opus 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs(evidence): correct the 1.16.1 vs 1.24.0 apples-to-apples claim (…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-27T18:30:45Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "5b02c24de01a1f3dd23ce1d6692621c4d68a6927",
"body": null,
"is_bot": false,
"headline": "chore: release 0.90.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-27T06:44:55Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "3fc35e9dfb4840de1201d387d4b796aa141a1785",
"body": "…st drift (#272)\n\nThe published comparison was measured against Flow v1.16.1 and was still the\ncurrent public evidence eight minor releases later. Re-run it against the exact\npublished 1.24.0 wheel with the byte-identical runner, add coverage for the new\ntransaction outcome taxonomy, and add a guard\n[…]\nidence set had to be\nforce-added and one could silently go unpublished.\n\n\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\nCo-authored-by: Claude Opus 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat(evals): rebind published evidence to Flow 1.24.0 and guard again…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-27T06:43:10Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "1fec68532e70dd05d6ad8668870c376c7fd8d996",
"body": "Reframe the README around what this repo actually is: evaluation and\nbenchmarking infrastructure that produces evidence for OpenAdapt's\ngoverned-demonstration-compiler claims, not an end-user product.\n\n- Keep and clean the research-status banner (no em dashes anywhere)\n- Add the openadapt-flow evalu\n[…]\ns out of scope\n- Cross-link docs.openadapt.ai and the OpenAdaptAI org\n\n\nClaude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM\n\nCo-authored-by: Claude Opus 4.8 <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs: refresh README as evidence-generating eval infra (#271)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-21T04:45:48Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "8b00d3c8036eca9dbde81d124ee1b66675605dfc",
"body": null,
"is_bot": false,
"headline": "docs(evals): bind current Flow release performance (#270)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-19T02:47:54Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7629ba5a2447c919e499e88e9b4eedbc80c8f3ab",
"body": "* docs(evals): publish bounded OpenAdapt performance evidence\n\n* fix(evals): bind performance report provenance",
"is_bot": false,
"headline": "docs(evals): publish bounded OpenAdapt performance evidence (#269)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-18T02:01:58Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "24a3108dc4a2c301895881d06172a2d280518dfc",
"body": "Adds the standard lifecycle banner used across the org (matching the\nbanner wave on openadapt-capture, openadapt-grounding, etc.), derived\nfrom this repo's classification in the repository lifecycle registry\n(OpenAdaptAI/.github repository-lifecycle.yml, reviewed 2026-07-15):\nResearch.\n\nExisting README content is unchanged below the banner.\n\nCo-authored-by: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs: add lifecycle status banner (#268)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-17T22:04:26Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "d98254d5721620e46b1251d2c2a19a7d4918b170",
"body": null,
"is_bot": false,
"headline": "chore: release 0.89.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-16T00:22:01Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "a08917a017945cf374e5fb2460454ebe39ace136",
"body": "Pin release tooling, enforce the no-sources lock in CI, and build the reviewed lock state during semantic release.",
"is_bot": false,
"headline": "fix: keep release lock metadata consistent",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-16T00:20:38Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "2059316876f116d3123995fb87245418bf221f14",
"body": null,
"is_bot": false,
"headline": "chore: release 0.89.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-14T14:10:52Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ae9a8c7ff428d8c74f6f1dd526599dae6827ff7e",
"body": "…metrics) (#266)\n\nConsolidate the three existing openadapt-evals abstractions -- BenchmarkAdapter\n(adapters/base.py), TaskVerifierRegistry (evaluation/verifier_registry.py), and\nthe flow-side EffectVerifier -- behind ONE runtime_checkable Environment\nprotocol whose verify() folds all three scoring p\n[…]\nError from the external stubs. No live\nAzure/VM/paid infra exercised.\n\n\nClaude-Session: https://claude.ai/code/session_01CKrVJJy5jWVCkXAqgUqtqZ\n\nCo-authored-by: Claude Opus 4.8 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: lightweight meta-benchmark harness (unify Environment/verify + …",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-14T14:09:39Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "ed008f48967896b06195995b7832d424cee6806e",
"body": null,
"is_bot": false,
"headline": "chore: release 0.88.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-14T04:25:03Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "b95e27da398662ef7fca5af5a9e13638fa2cbeca",
"body": "…d-as-agent) with cost-guarded dry-run (#265)\n\n* feat: evaluate openadapt-flow on WAA (demonstrate-then-replay + hybrid-as-agent) with cost-guarded dry-run\n\nWire openadapt-flow (the demonstration compiler) into the WAA benchmark\nharness with two eval modes, both scored by WAA's own task verifier:\n\n-\n[…]\not.\n\nCo-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>\nClaude-Session: https://claude.ai/code/session_01CKrVJJy5jWVCkXAqgUqtqZ\n\n---------\n\nCo-authored-by: Claude Opus 4.8 <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: evaluate openadapt-flow on WAA (demonstrate-then-replay + hybri…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-14T04:23:49Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "d93254abb86ee96a70d9a817e7b015c0e3f172ba",
"body": "…als cycle (#264)\n\n* refactor: source Benchmark* types from openadapt-types; break ml<->evals cycle\n\nPhase 1 of the evals->ml refactor.\n\nFork A: BenchmarkTask/Observation/Action (adapters/base.py) and\nBenchmarkAgent (agents/base.py) are now imported from openadapt-types\n(the canonical schema package\n[…]\ne.ai/code/session_01CKrVJJy5jWVCkXAqgUqtqZ\n\n* chore: pin openadapt-types>=0.3.0 (published w/ benchmark), drop local editable source\n\n---------\n\nCo-authored-by: Claude Opus 4.8 <noreply@anthropic.com>",
"is_bot": false,
"headline": "refactor: source Benchmark* types from openadapt-types; break ml<->ev…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-13T16:40:41Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "3843e883c84e7d33759b926064a40c80d3a4b641",
"body": null,
"is_bot": false,
"headline": "chore: release 0.87.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-07-10T19:34:56Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "dca4ff75203752b1d8a1b6de5f94ed24fa18f721",
"body": "…ts (#263)\n\nThe oa-vm console script imports openadapt_evals.benchmarks.vm_cli, which\ntriggered the package __init__ modules' eager `from openadapt_evals.agents\nimport ...`. That cascaded into transformers/peft — dead weight for VM\nlifecycle management, and a hard crash under a NumPy 2 / stale-trans\n[…]\ns a check that\nthe lazy re-exports still resolve to the real objects.\n\n\nClaude-Session: https://claude.ai/code/session_01CKrVJJy5jWVCkXAqgUqtqZ\n\nCo-authored-by: Claude Opus 4.8 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: decouple oa-vm from the ML training stack via lazy package impor…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-07-10T19:33:37Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "671eb419e517ed76c5781b1d868ceeeef7027428",
"body": null,
"is_bot": false,
"headline": "chore: release 0.87.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-06-13T00:19:02Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7c05b76432bd4bc600e31de584fda7891ef1a905",
"body": "…ease alerting (#262)\n\nEcosystem rollout of the #999 guards (see openadapt-ml#64,\nOpenAdapt#1002):\n\n- tests/test_import_integrity.py: AST-based phantom-import and\n phantom-kwarg detection across the whole package, including imports\n inside function bodies\n- The new guard immediately found one live\n[…]\npend a GitHub issue when the release workflow\n fails, so PyPI cannot silently go stale (openadapt-ml's releases\n failed silently Mar-Jun 2026)\n\nCo-authored-by: Claude Fable 5 <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: implement missing cmd_tasks; add import-integrity guards and rel…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-06-13T00:17:49Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "3defa5b489df8f9d39ad82fea1ac23ad89aee038",
"body": null,
"is_bot": false,
"headline": "chore: release 0.87.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-04-01T15:29:12Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "dd01054723fc7b88416c2010691324247dd551cf",
"body": "Add two scripts for populating GroundingTarget data on demo click steps:\n\n- enrich_demo_targets.py: Enriches each click step with GroundingTarget\n metadata (target_type, crop_bbox, click_offset, nearby_text) using OCR\n when real screenshots are available, or description-derived heuristics\n when t\n[…]\npts use fire for CLI, handle the existing demo JSON format, and\nintegrate with grounding.py GroundingTarget.to_dict()/from_dict().\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: demo enrichment pipeline for GroundingTarget data (#261)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-04-01T15:27:45Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "4683b178ffc89ade57faf805ca7e4e40c803d4be",
"body": null,
"is_bot": false,
"headline": "chore: release 0.86.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-04-01T15:26:10Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "c66468a07fc98643936572fe75af8cc265a13c95",
"body": "run_ocr() now tries backends in order:\n1. GLM-OCR (VLM-based, pip install glmocr, better accuracy on complex UIs)\n2. pytesseract (traditional OCR, requires system Tesseract binary)\n3. Empty list (graceful degradation)\n\nAdded [ocr] optional dependency group for glmocr.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add GLM-OCR as primary OCR backend, pytesseract as fallback (#260)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-04-01T15:24:51Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "6a3dc5a08988647893930fd08b03bc4081515c1b",
"body": null,
"is_bot": false,
"headline": "chore: release 0.85.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-31T22:47:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "aa797dd2c01ce6803cf81b7f1a16464dcb05cad7",
"body": "Add Phase 5 text anchoring on top of Phase 4 state narrowing:\n\n- grounding.py: run_ocr() with pytesseract (optional dep, graceful\n fallback), ground_by_text() with tiered scoring (exact/case-insensitive/\n substring/fuzzy) and nearby-text proximity boost, plus helper functions\n _char_overlap_ratio\n[…]\ns,\n proximity boost, edge cases, and graceful pytesseract fallback.\n All tests use mocked OCR results (no pytesseract required).\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: OCR text anchoring (Tier 1.5a) for grounding cascade (#259)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-31T22:46:27Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "a5ebabb2877d622b27b659e652fd4748c3f2739e",
"body": null,
"is_bot": false,
"headline": "chore: release 0.84.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-31T22:39:00Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "e22b404cd6ebd618a06c5a9748315d2a6ecd910e",
"body": "…de (#257)\n\nPhase 4 of the grounding cascade — detect \"wrong screen\" before\ngrounding and verify state changes after clicking.\n\nAdded to grounding.py:\n- check_state_preconditions(): verifies window title, nearby text,\n and surrounding labels match expectations before grounding a click.\n Skips grac\n[…]\nrance/disappearance, window title change, modal skip,\n combined scenarios)\n- 4 tests for GroundingTarget round-trip serialization\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: state narrowing and transition verification for grounding casca…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-31T22:37:37Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "68e4b2ad6f5294929e9313c2e090ed698b4aa184",
"body": null,
"is_bot": false,
"headline": "chore: release 0.83.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-31T22:15:39Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "e912b653942424513b4762fdbae8051c0b71c3b3",
"body": "…hitecture (#256)\n\nPhase 3 of the grounding cascade design (v3):\n\n- grounding.py: GroundingTarget (rich target per click step — description,\n crop, nearby text, window title, structured transition expectations)\n and GroundingCandidate (normalized output from each grounding tier)\n- demo_library.py:\n[…]\nscade. Every\ndownstream tier (OCR, CLIP, UI-Venus, GPT-5.4) operates on the same\nrich signal instead of a weak description string.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: GroundingTarget + GroundingCandidate data model for cascade arc…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-31T22:14:26Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "2393c537e4e3b9ca9d0876a605a269acd15ee359",
"body": null,
"is_bot": false,
"headline": "chore: release 0.82.4",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-30T16:58:12Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "321dceac303d61a416b25ca9624dedc1b3a90da9",
"body": "…eprecation (#255)\n\nThree changes based on client training results (grad_norm=101, 0.00 eval delta):\n\n1. Add max_grad_norm to TrainingConfig (was hardcoded to 1.0). When\n grad_norm >> max_grad_norm, gradients are clipped to a near-random\n direction — training makes no progress despite non-zero l\n[…]\nn't support multimodal VLMs (issue #5120).\n The standalone trainer is the production training path until TRL\n PR #5323 merges.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: configurable max_grad_norm, lower default lr, remove premature d…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-30T16:56:50Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "a5f290dce555a2c7ab358ee0465e63729e6db3f2",
"body": null,
"is_bot": false,
"headline": "chore: release 0.82.3",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T23:47:58Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "96120198f5cd413675b37c6752fe0f2c79dae78e",
"body": "forward() patch handles training logprob recomputation, but TRL also\ncalls model.generate(input_ids=...) without pixel_values. HF's\ngenerate() uses prepare_inputs_for_generation() which builds a fresh\nkwargs dict — cached pixel_values in forward() aren't enough because\ngenerate() needs them at the top level to pass them through.\n\nNow patches BOTH forward() and generate() on the model instance.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: also patch model.generate() to inject cached pixel_values (#254)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T23:46:36Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "362c976fbb3903ed4e91a747fc573cb5bcce3632",
"body": null,
"is_bot": false,
"headline": "chore: release 0.82.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T23:38:11Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "0f381b1c6ff0de3c2ff2e9f89cde91542df71e2f",
"body": "TRL unwraps models via Accelerate, stripping wrapper classes. The fix:\npatch forward() on the model instance itself. This survives unwrapping.\n\n- patch_model_for_trl(model) → returns cache_fn\n- cache_fn(inputs) caches pixel_values from processor output\n- Patched forward() injects cached pixel_values\n[…]\ncovers all call paths)\n- trl_wrapper passes original model to TRL (not a wrapper)\n- cache_vision_fn passed through to rollout_func\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: patch model.forward() directly instead of wrapper class (#253)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T23:36:51Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "e27dafc263e0ce1fa70087a244ff9f0677d9dc4e",
"body": null,
"is_bot": false,
"headline": "chore: release 0.82.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T23:25:43Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7879deef1bc1a50ae3292b3f9861a6a9c5910ac6",
"body": "… (#252)\n\nTRL's validate_quantization_for_training() uses isinstance(model,\nPeftModel) to check for adapters. The wrapper hid the PeftModel,\ncausing: \"You cannot perform fine-tuning on purely quantized models.\"\n\nFix: dynamically create a combined class inheriting from both\nVLMModelWrapper and the wr\n[…]\nwrapper_passes_peft_validation (e2e): full TRL validation sim\n- test_wrapper_preserves_trainable_parameters (e2e): optimizer setup\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: VLMModelWrapper PEFT isinstance compatibility for TRL validation…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T23:24:23Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "13ad5116f4f6326c7863730fd63aae003330d4f6",
"body": null,
"is_bot": false,
"headline": "chore: release 0.82.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T23:05:08Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "fa26d553d05e6bcae1f41191d52375331888095e",
"body": "* feat: VLMModelWrapper — multimodal compatibility layer for TRL\n\nTRL's GRPOTrainer calls model.forward(input_ids=...) during training\nwithout pixel_values. VLMs need pixel_values to produce meaningful\nlogits. Without them, the model is blind and generates garbage.\n\nVLMModelWrapper caches vision ten\n[…]\nodal TRL failures before they reach\nthe customer.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: VLMModelWrapper — multimodal compatibility layer for TRL (#251)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T23:03:46Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "93fa3954fc780dad503f60a739e583fad864f613",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.9",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T22:30:12Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "c1d35883851ad111f049167051e71e9a44e7819d",
"body": "Stripping <think> from rendered text was insufficient — TRL or the\nprocessor may re-apply the template, re-inserting the tags. The fix:\npatch processor.chat_template and processor.tokenizer.chat_template\non first rollout call, removing <think>/<think> from the Jinja\ntemplate itself. This ensures no code path can re-insert thinking mode.\n\nAlso strips </think> (was missed in #249).\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: patch chat_template to remove <think> tags at the source (#250)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T22:28:52Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "bd3acaff2934e2a80610c1a18d77fd5db2b448c5",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.8",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T22:07:10Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "5a2bf7f7d6dc262608fba994b02cfbce50eaa811",
"body": "Root cause of persistent garbage output: Qwen3.5-9B's chat template\ninserts <think> which activates internal reasoning mode. The model\nproduces opaque thinking tokens (# # # # #) instead of DSL actions.\n\nFix: pass enable_thinking=False to apply_chat_template. Falls back to\nstripping <think> from rendered text if the kwarg is not supported.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: disable Qwen3.5 thinking mode in TRL generation (#249)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T22:05:52Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "dd0dd45151339fda0b53afd44c7d236f2ed4f27e",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.7",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T21:41:02Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8e3bc45097231e35060c30e0064753c4dea527d1",
"body": "…248)\n\nAdds detailed one-time logging to help debug the persistent garbage\noutput issue:\n\n1. Raw messages (role, content types, text preview) before chat template\n2. Full rendered text_input (2000 chars, not 300)\n3. Image metadata (mode, size, format)\n4. Generation config (max_new_tokens, temperatur\n[…]\nvalues is MISSING, the\nmodel isn't seeing the screenshot — which would explain degenerate output\nregardless of prompt correctness.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: comprehensive prompt diagnostics for debugging garbage output (#…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T21:39:41Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "4e49c80d571f3c71731e1baee3c4beffd8c03717",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.6",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T21:24:16Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "1d948996a117b8d9a20430e28ed631fb8361b8cc",
"body": "… (#247)\n\nTwo critical fixes:\n\n1. Garbage output root cause: TRL constructed user messages differently\n from the standalone trainer. Standalone wraps instruction with\n \"Goal:\" prefix, format guidance, and {\"type\": \"image\"} placeholder.\n TRL passed raw instruction text. Now imports build_agent_\n[…]\nprompt\n per step with num_gen rollouts. No dataset padding needed.\n\nAlso adds one-time prompt logging for operator verification.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: use build_agent_messages for TRL prompt + fix 4x over-generation…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T21:22:56Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "682b581b2e5dd405fcc903b3c266d9a07f8f0c3c",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.5",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T20:47:18Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "fc40bf40784482a20a49600dd95b151b1342d6b7",
"body": "… rollout_func (#243)\n\n* fix: add truncation warning to TRL generate paths\n\nAdd a truncation check after both generation paths (Outlines constrained\nand HF unconstrained) in generate_fn. When the output length reaches\nmax_new_tokens - 1, a warning is logged suggesting to increase\nmax_new_tokens or e\n[…]\nacks from HookBridge (keep only on_step_complete)\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: wire on_before_collect and on_rollout_complete callbacks through…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T20:45:47Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "36ac839ba0812f50c16f3fc4eb576dbdf3d1aeb9",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.4",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T20:44:22Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "e71ed9fe17168524963b564aa050bb4d4d4d305e",
"body": "Add a truncation check after both generation paths (Outlines constrained\nand HF unconstrained) in generate_fn. When the output length reaches\nmax_new_tokens - 1, a warning is logged suggesting to increase\nmax_new_tokens or enable constrained_decoding. This helps diagnose\ncases where the model genera\n[…]\nts that exercise the\nactual generate_fn code path by calling it through the rollout function\nwith mocked torch and model.generate.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: add truncation warning to TRL generate paths (#242)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T20:42:42Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "6a38956f3da2776701b0b92b94134609e83f4d4d",
"body": "Adds tests/test_trl_parity.py with 25 test cases covering the 10 areas\nidentified in docs/STANDALONE_VS_TRL_COMPARISON.md as needed before the\nstandalone GRPO trainer can be deprecated:\n\n1. Constrained decoding — Outlines generator build + ACTION_REGEX\n2. Constrained decoding ImportError — returns N\n[…]\nJSON schema, roundtrip\n\nAll tests are light (no torch/transformers/trl imports), use unittest.mock,\nand pass with [dev] deps only.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: add 10 TRL parity tests for deprecation readiness (#241)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T20:42:40Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "114ad0e8bdc33c35a966ba820ad958fba4269550",
"body": "… eval (#246)\n\nReverts the evaluate_dense reordering from #245 (local-first was too\naggressive — skipped binary eval entirely, losing the signal when 5050\nIS available).\n\nThe actual fix: set evaluate_timeout=15s and evaluate_retries=1 on the\nWAALiveAdapter in the TRL wrapper. The evaluate_dense logi\n[…]\n\n- Benchmarking: 180s timeout, 3 retries (thorough, one-shot)\n- Training: 15s timeout, 1 retry (fast feedback, thousands of evals)\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: use training-appropriate evaluate timeouts instead of reordering…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T20:41:47Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "0922b0a6da435bd659290449aa87b2580faf2962",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.3",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T20:16:23Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "3b8c1c2b6317a693fec2e97cf8aa459205f1be4d",
"body": "…(#245)\n\n51% of TRL training time wasted on 5050 evaluate timeouts (180s × 3\nretries = 9 min per evaluation). The local evaluation via\nevaluate_checks_local takes ~5s.\n\nFix: when task config has checks defined, try local eval FIRST. Only\nfall through to the slow /evaluate endpoint when no local chec\n[…]\n define\ntheir own checks.\n\nBefore: evaluate() [9 min] → if 0.0 → local [5s]\nAfter: local [5s] → if no checks → evaluate() [9 min]\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: try local eval before slow /evaluate endpoint in evaluate_dense …",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T20:15:04Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "d8c6187fb6411314c092aadb76e23ef8a80905dc",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T19:09:20Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d6e1b5bff59d672e5ec74126d35302f852ffe09a",
"body": "…eeded (#244)\n\nTRL requires generation_batch_size % num_generations == 0. With\nbatch_size=1 and num_generations=4, TRL rejects it. Fix:\n\n1. Set per_device_train_batch_size = num_generations (minimum valid)\n2. Pad dataset by repeating tasks if len(dataset) < batch_size\n\nWith 1 task and num_generations=4: dataset padded to 4 rows,\nbatch_size=4, generation_batch_size=4, 4 % 4 == 0 ✓\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: batch_size must be multiple of num_generations, pad dataset if n…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T19:07:52Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "345f1a95d589166c228ced41abd723edc3183fe1",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T18:22:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "048796c020a474293758ff8a95ed6ef520f41fbf",
"body": "* fix: set per_device_train_batch_size to match dataset size\n\nTRL's default per_device_train_batch_size=8, but with 1-3 tasks the\ndataset is too small to form a single batch. TRL computes 0 steps and\nexits with \"There seems not to be a single sample in your epoch_iterator\".\n\nFix: set batch_size=n_ta\n[…]\nations\nrollouts, so learning signal is preserved.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: set per_device_train_batch_size to match dataset size (#240)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T18:21:22Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "8dab865a812e3b7e93101e8455f4e8c85a88b3e5",
"body": null,
"is_bot": false,
"headline": "chore: release 0.81.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T18:17:28Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d7896d562135366fce600bba9e5b342699b3ee09",
"body": "- Add DiagnosticsCallback to trl_callbacks.py: logs loss, |loss|,\n grad_norm, reward in scientific notation (matches standalone trainer\n diagnostic output)\n- Register DiagnosticsCallback in trl_wrapper.py alongside TelemetryCallback\n- Add test_trl_robustness.py: 19 tests covering health check, corrupt\n screenshot retry, stuck detection, truncation warning, diagnostics\n callback, and empty rollout result shape\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add DiagnosticsCallback and TRL robustness tests (#238)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T18:15:31Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "fb7e87f5a728bb0a05226765d5440a273de6f7b2",
"body": "- Add openadapt-types>=0.1.0 to core dependencies (canonical action\n schema for the OpenAdapt ecosystem — Pydantic v2, lightweight)\n- Add _AgentOutput Pydantic model for future Outlines JSON schema\n constrained decoding (currently unused — default is DSL regex)\n- Does NOT change the system prompt \n[…]\n model enables switching to outlines.json(model, schema)\nonce models are SFT'd on JSON format. For now, DSL regex remains default.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add openadapt-types dependency and _AgentOutput schema (#239)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T18:14:57Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "83d549b19ab9b079777c5db17295a514e0192fa6",
"body": null,
"is_bot": false,
"headline": "chore: release 0.80.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T17:22:23Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "de515b8cde49ffd0e08a0097f9857c042d4c017a",
"body": "…parsing (#236)\n\n* fix: critical TRL trainer bugs — wrong prompt, ignored task_ids, DSL parsing\n\nThree bugs reported from client testing the TRL path:\n\n1. Garbage output: TRL used a JSON system prompt but the model was SFT'd\n on DSL format (Thought/Action). Now imports SYSTEM_PROMPT from the\n st\n[…]\nest mocks to accept **kwargs for stuck_threshold.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: critical TRL trainer bugs — wrong prompt, ignored task_ids, DSL …",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T17:21:00Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "3e7debc273602272b5ab31f68b4b045923306583",
"body": null,
"is_bot": false,
"headline": "chore: release 0.80.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T17:06:23Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7a202c9d93341ff29258c03c00c6d29e9786fbc5",
"body": "* fix: add triple-layer CI protection against heavy import failures\n\n- Add pytest markers (heavy, gpu, vm) to pyproject.toml\n- Guard test_vision_loss.py with importorskip(\"torch\") + @heavy marker\n- Guard test_api_agent_ml.py with importorskip(\"openadapt_ml\") + @heavy marker\n- Add CI lint step that f\n[…]\n openadapt-types instead of creating new schemas.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: add triple-layer CI protection against heavy import failures (#235)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T17:05:08Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "65f0980ab583552e9d71296a72df8d351ebb25cc",
"body": null,
"is_bot": false,
"headline": "chore: release 0.80.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T16:37:44Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "b403c2aea21703b1bae9ab02b49263dcd8aef4b2",
"body": "* docs: pyproject.toml telemetry for enterprises\n\n* feat: add TRL, Unsloth, datasets to [training] extra\n\npip install openadapt-evals[training] now includes everything needed\nfor GRPO training: TRL, Unsloth, datasets, outlines.\n\nClear error if use_unsloth=True but unsloth somehow not installed.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: add TRL + Unsloth to [training] extra (#234)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T16:36:25Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "ad9844bcfdd86c3ebab55181ccacc9bf9382f9a2",
"body": null,
"is_bot": false,
"headline": "docs: pyproject.toml telemetry for enterprises (#233)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T16:31:04Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "a4c131687a2f6c658fda93c19bbed3e0cafacea1",
"body": "* fix: TelemetryCallback __bases__ crash + 12 TRL integration tests\n\nThe dynamic __bases__ assignment to inject TrainerCallback as a base\nclass fails in Python: \"deallocator differs from object\". Fixed by\ncreating a proper subclass at definition time instead.\n\n12 new tests:\n- Mock rollout_func: corr\n[…]\ncy scrubbing, CI behavior, and source code links.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>\n\n---------\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "docs: telemetry guide (disable with DO_NOT_TRACK=1) (#232)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T16:17:59Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "82179b2343d0c1a5e81da0b3df7e1c03dd6ea7ee",
"body": null,
"is_bot": false,
"headline": "chore: release 0.79.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T16:14:22Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ac2df2f9d9dece530648a97101e31d0039c1c805",
"body": "The dynamic __bases__ assignment to inject TrainerCallback as a base\nclass fails in Python: \"deallocator differs from object\". Fixed by\ncreating a proper subclass at definition time instead.\n\n12 new tests:\n- Mock rollout_func: correct keys, count, reward variance\n- Config separation: TrainingConfig \n[…]\ntrl_config\n- Wrapper construction: all callback combinations, trl_config passthrough\n- TelemetryCallback: importable, fires events\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: TelemetryCallback __bases__ crash + 12 TRL integration tests (#231)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T16:13:06Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "0acd4111a403a3b622a5817621fdec5a5d4c99e6",
"body": null,
"is_bot": false,
"headline": "chore: release 0.79.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T16:07:56Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "1c23f0b0082040580ea37f1e56e82e53fe1cc766",
"body": "…ion (#230)\n\nTrainingConfig owns OpenAdapt concerns: model, task_dir, server_url,\nconstrained_decoding, max_new_tokens, use_unsloth, weave_project.\n\nTRL's GRPOConfig owns training concerns: loss_type, learning_rate,\nbatch_size, gradient_accumulation, vLLM, bf16, W&B reporting.\n\nThe wrapper accepts b\n[…]\nerations=4),\n on_step_complete=my_logger,\n )\n\nIf trl_config is omitted, sensible defaults are built from TrainingConfig.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: clean config separation — our config + TRL's config, no duplicat…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T16:06:39Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "9759cdce5faa7d798609f0cf72b66124ba474ae3",
"body": null,
"is_bot": false,
"headline": "chore: release 0.79.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T15:57:47Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "f7d840c1089e403a01256de8db18e1b2af32de87",
"body": "TRL integration:\n- Outlines constrained decoding ported to rollout_func\n- TelemetryCallback maps to our telemetry events\n- train_trl_grpo.py: --constrained-decoding, --weave-project, --no-telemetry\n- README: TRL training section with 4 usage examples\n\nDrop-in Python wrapper (trl_wrapper.py):\n- Same \n[…]\nlete=my_logger)\n trainer.train()\n\nStandalone trainer:\n- Deprecated with warning (not removed)\n- Falls back if TRL not installed\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: TRL GRPOTrainer migration with drop-in Python wrapper (#229)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T15:56:33Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "50fdc33ae23a627997ca85664689ad32020126dd",
"body": null,
"is_bot": false,
"headline": "chore: release 0.78.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T15:49:44Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d000f5d78af487ef2ea0a2e6d168c39fb06c92a0",
"body": "These were committed before PR #227 and missed in the first cleanup.\nwaa_recordings/ contains WAA experiment screenshots (PNGs).\n.beads/ contains a SQLite database for local tooling.\n\nBoth are already in .gitignore from the prior cleanup commit.\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: remove remaining 125 tracked data files (waa_recordings, .beads)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T15:48:23Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "b736b7905045be572f5abe3d47adf79821b85606",
"body": null,
"is_bot": false,
"headline": "chore: release 0.78.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T15:30:53Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ee1ceb9bd20b37d8f742c7b9a3657f8a59a8155f",
"body": "PR #227 accidentally committed local experiment data via git add -A:\n- flywheel_results/ (224 screenshots + JSON)\n- .claude/worktrees/ (31 agent gitlinks)\n- annotated_demos/ (16 files)\n- eval_results/ (11 screenshots)\n- grpo_output/ (1 file)\n- demos/*/synthetic_correction/ (placeholder PNGs)\n- .bead\n[…]\n)\n\nAll removed from tracking. .gitignore updated to prevent reoccurrence.\nNo sensitive data was exposed (confirmed via tidy scan).\n\nCo-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: remove 307 accidentally committed data files, update .gitignore",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T15:29:35Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "0824a3648e8aa6e270bb9fa88d9fd6562d81a52c",
"body": null,
"is_bot": false,
"headline": "chore: release 0.78.0",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T15:16:10Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "6d9fcb704e0ea0d4f1e7d2beff927a64e2676d48",
"body": "Weave auto-patches OpenAI and Anthropic clients after weave.init(),\ngiving automatic tracing of every VLM call with prompts, responses,\ncosts, and latency in hierarchical trace trees.\n\nIntegration points:\n- vlm_call() — @weave_op: all planner/grounder/evaluator calls traced\n- vlm_judge() — @weave_op\n[…]\neave is not installed, all decorators are zero-cost passthrough.\nweave>=0.50.0 added to [wandb] optional extra.\n\n76/76 tests pass.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "feat: Weave (W&B) integration for LLM/agent tracing (#228)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T15:14:55Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "8481198d179f2434397710a3be61e6760736d997",
"body": null,
"is_bot": false,
"headline": "chore: release 0.77.5",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T14:39:21Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "74fd6466bcd22a203e7e1ff1ba7a6a787355f75f",
"body": "…ges) (#227)\n\nloss=0.0000 was misleading: %.4f truncation + symmetric advantages\ncanceling. Now logs loss in scientific notation, absolute loss per\nrollout, gradient norm, and per-rollout advantages.\n\n13 vision loss tests (was 12). New test verifies loss_abs > 0 and\nadvantages are symmetric with reward variance.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: diagnostic logging (loss scientific notation, grad_norm, advanta…",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T14:37:55Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "2355d53ae7de4e1823b742377122550e498ec9f2",
"body": "VisionMergeModel mimics Qwen2.5/3.5-VL: replaces placeholder tokens\nwith N visual features, changing sequence length. 4 new tests:\n\n- test_manual_concat_crashes: OLD approach → IndexError (mask mismatch)\n- test_unified_processor_works: NEW approach → correct post-merge shape\n- test_no_vision_no_merg\n[…]\nmerge → mask safe\n- test_exclude_strips_vision: exclude mode → no pixel_values → safe\n\nArchitecture-agnostic. 12/12 pass in 0.05s.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: synthetic vision-merge model proves fix correctness (#226)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T14:29:40Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "b488794027dffd04d24a3dc352d5ad24a3cc61e9",
"body": "Would have caught the Qwen3 vision merge crash before shipping.\n8/8 pass in 0.07s, no GPU, uses real tiny nn.Module.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: vision loss computation tests (8 tests) (#225)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T14:19:24Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "b634e8fbf947333a07e872be51577f0368447649",
"body": null,
"is_bot": false,
"headline": "chore: release 0.77.4",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T14:03:14Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "5413864342a71f3cead106d640e5ba6adc5fae95",
"body": "Root cause: manually concatenating action_ids onto prompt input_ids\ncreated inconsistent input (pixel_values sized for prompt, input_ids\nincludes action tokens). Qwen3's vision merge changes internal\nsequence length, crashing with attention mask mismatches.\n\nFix: process prompt_text + action_text as\n[…]\naces the silent fallback from PR #223 with a proper solution that\ngives correct vision-aware gradients for ALL steps in ALL modes.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: proper vision-safe loss — process full text as one unit (#224)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T14:01:56Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "5e42bd6b3f4011969cd885f4ab6a60485537760c",
"body": null,
"is_bot": false,
"headline": "chore: release 0.77.3",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T13:55:03Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d348f1b6bca44cc4d94ccff0695351b9f90c6ed1",
"body": "Qwen3's vision-language merge changes internal sequence length\nunpredictably. Both include and checkpoint modes crash intermittently\nwith attention mask mismatches (mask too large OR too small depending\non generated sequence length).\n\nFix: catch IndexError/RuntimeError from the vision forward pass a\n[…]\nning.\n\nThis is the pragmatic fix. The proper fix (capturing logits during\ngeneration to avoid re-forward entirely) is future work.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: vision loss forward pass falls back to exclude on crash (#223)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T13:53:49Z",
"body_truncated": true,
"is_coding_agent": true
},
{
"oid": "02e8216b97d8956ba6b9d5f3fa7505cb96bee431",
"body": null,
"is_bot": false,
"headline": "chore: release 0.77.2",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T05:03:10Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8e6f685e94a61425df3e563623ece79690c7e44a",
"body": "12 tests covering the full DemoExecutor pipeline:\n- Keyboard-only demo: 3 steps execute in order, all Tier 1\n- Mixed demo: click uses grounder, keyboard bypasses it\n- Evaluation: dense score with milestones, binary without\n- Telemetry: start/completed events with tier counts\n- Edge cases: empty demo, missing values, unknown action types\n\nAll tests use MockEnv (no WAA server, no HTTP, no API keys).\n12/12 pass in 0.05s.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "test: DemoExecutor e2e tests with mock WAA environment (#222)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T05:01:47Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "6a0374a25c4ea4e0b3808ac9cc0780ee89a7f087",
"body": "test_workflow_models.py and workflow pipeline import numpy directly.\nWas transitive via openadapt-ml, now needed as dev dep.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: add numpy to dev dependencies for CI (#221)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T05:00:37Z",
"body_truncated": false,
"is_coding_agent": true
},
{
"oid": "5a219ced392647603dc1f2109efdc638188eb47e",
"body": null,
"is_bot": false,
"headline": "chore: release 0.77.1",
"author_name": "semantic-release",
"author_login": null,
"committed_at": "2026-03-29T04:53:18Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "0d0616e71699897f2d83c744e8c17c6337529695",
"body": "PyYAML was a transitive dependency via openadapt-ml. Now that ml is\noptional (PR #218), yaml import fails in CI. TaskConfig uses yaml\ndirectly — it must be a core dep.\n\nCo-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>",
"is_bot": false,
"headline": "fix: add PyYAML to core dependencies (#220)",
"author_name": "Richard Abrich",
"author_login": "abrichr",
"committed_at": "2026-03-29T04:52:01Z",
"body_truncated": false,
"is_coding_agent": true
}
],
"releases_count": 100,
"commits_last_year": 490,
"latest_release_at": "2026-07-28T03:40:15Z",
"latest_release_tag": "v0.90.2",
"releases_from_tags": false,
"days_since_last_push": 0,
"active_weeks_last_year": 16,
"days_since_latest_release": 0,
"mean_days_between_releases": 13.1
},
"community": {
"has_readme": true,
"has_license": true,
"has_description": true,
"has_contributing": false,
"health_percentage": 50,
"has_issue_template": false,
"has_code_of_conduct": false,
"has_pull_request_template": false
},
"ecosystem": {
"packages": [
{
"name": "openadapt-evals",
"exists": true,
"license": "MIT",
"keywords": [
"agent",
"ai",
"automation",
"benchmark",
"evaluation",
"gui",
"Development Status :: 3 - Alpha",
"Intended Audience :: Developers",
"Intended Audience :: Science/Research",
"License :: OSI Approved :: MIT License",
"Operating System :: OS Independent",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.10",
"Programming Language :: Python :: 3.11",
"Programming Language :: Python :: 3.12",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Topic :: Software Development :: Testing"
],
"ecosystem": "pypi",
"matches_repo": true,
"registry_url": "https://pypi.org/project/openadapt-evals/",
"is_deprecated": false,
"latest_version": "0.90.2",
"repository_url": "https://github.com/OpenAdaptAI/openadapt-evals",
"versions_count": 181,
"total_downloads": null,
"dependents_count": null,
"deprecation_note": null,
"maintainers_count": null,
"monthly_downloads": 2941,
"first_published_at": "2026-01-17T00:04:04.056122Z",
"latest_published_at": "2026-07-28T03:40:30.327929Z",
"latest_version_yanked": null,
"days_since_latest_publish": 0
}
]
},
"popularity": {
"forks": 3,
"stars": 2,
"watchers": 0,
"fork_history": {
"days": [
{
"date": "2026-03-03",
"count": 1
},
{
"date": "2026-04-03",
"count": 1
},
{
"date": "2026-05-18",
"count": 1
}
],
"complete": true,
"collected": 3,
"total_forks": 3
},
"star_history": null,
"open_issues_and_prs": 4
},
"ai_readiness": {
"has_nix": false,
"example_dirs": [
"demos",
"examples"
],
"has_llms_txt": false,
"has_dockerfile": true,
"has_mcp_signal": false,
"bootstrap_files": [],
"api_schema_files": [],
"has_devcontainer": false,
"typecheck_configs": [],
"toolchain_manifests": [],
"largest_source_bytes": 304758,
"source_files_sampled": 269,
"oversized_source_files": 6,
"agent_instruction_files": [
"CLAUDE.md"
],
"agent_instruction_max_bytes": 39650
},
"dependencies": {
"manifests": [
"pyproject.toml"
],
"advisories": {
"error": null,
"scope": null,
"source": null,
"findings": [],
"collected": false,
"malicious": [],
"truncated": false,
"by_severity": {},
"advisory_count": 0,
"affected_count": 0,
"assessed_count": 0,
"malicious_count": 0,
"assessed_package": null,
"unassessed_count": 0,
"direct_affected_count": 0
},
"ecosystems": [
"pypi"
],
"dependencies": [
{
"name": "pillow",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=10.0.0"
},
{
"name": "pydantic-settings",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=2.0.0"
},
{
"name": "python-dotenv",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=1.2.1"
},
{
"name": "tenacity",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=8.2.0"
},
{
"name": "requests",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=2.28.0"
},
{
"name": "httpx",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.25.0"
},
{
"name": "openai",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=1.0.0"
},
{
"name": "anthropic",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.76.0"
},
{
"name": "pyyaml",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=6.0"
},
{
"name": "openadapt-consilium",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.3.2"
},
{
"name": "openadapt-telemetry",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.2.0"
},
{
"name": "openadapt-types",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.3.0"
}
],
"all_dependencies": {
"error": "GitHub dependency-graph SBOM unavailable (404); the dependency graph may be disabled for this repository",
"source": null,
"packages": [],
"collected": false,
"truncated": false,
"total_count": null,
"direct_count": null,
"indirect_count": null
}
},
"maintainership": {
"issues": {
"open_prs": 0,
"merged_prs": 250,
"open_issues": 4,
"closed_ratio": 0.667,
"closed_issues": 8,
"closed_unmerged_prs": 14
},
"bus_factor": 1,
"bot_contributors": 0,
"top_contributors": [
{
"type": "User",
"login": "abrichr",
"commits": 314,
"avatar_url": "https://avatars.githubusercontent.com/u/774615?v=4"
}
],
"contributors_sampled": 1,
"top_contributor_share": 1
},
"quality_signals": {
"has_ci": true,
"has_tests": true,
"ci_workflows": [
"evidence-freshness.yml",
"notify-docs.yml",
"release.yml",
"test.yml"
],
"has_docs_dir": true,
"linter_configs": [],
"has_editorconfig": false,
"has_linter_config": false,
"has_precommit_config": false
},
"security_signals": {
"lockfiles": [
"uv.lock"
],
"scorecard": {
"checks": [
{
"name": "Binary-Artifacts",
"score": 10,
"reason": "no binaries found in the repo",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#binary-artifacts"
},
{
"name": "Branch-Protection",
"score": null,
"reason": "internal error: error during branchesHandler.setup: internal error: some github tokens can't read classic branch protection rules: https://github.com/ossf/scorecard-action/blob/main/docs/authentication/fine-grained-auth-token.md",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#branch-protection"
},
{
"name": "CI-Tests",
"score": 10,
"reason": "19 out of 19 merged PRs checked by a CI test -- score normalized to 10",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#ci-tests"
},
{
"name": "CII-Best-Practices",
"score": 0,
"reason": "no effort to earn an OpenSSF best practices badge detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#cii-best-practices"
},
{
"name": "Code-Review",
"score": 0,
"reason": "Found 0/30 approved changesets -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#code-review"
},
{
"name": "Contributors",
"score": 6,
"reason": "project has 2 contributing companies or organizations -- score normalized to 6",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#contributors"
},
{
"name": "Dangerous-Workflow",
"score": 10,
"reason": "no dangerous workflow patterns detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#dangerous-workflow"
},
{
"name": "Dependency-Update-Tool",
"score": 0,
"reason": "no update tool detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#dependency-update-tool"
},
{
"name": "Fuzzing",
"score": 0,
"reason": "project is not fuzzed",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#fuzzing"
},
{
"name": "License",
"score": 10,
"reason": "license file detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#license"
},
{
"name": "Maintained",
"score": 10,
"reason": "24 commit(s) and 0 issue activity found in the last 90 days -- score normalized to 10",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#maintained"
},
{
"name": "Packaging",
"score": 10,
"reason": "packaging workflow detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#packaging"
},
{
"name": "Pinned-Dependencies",
"score": 3,
"reason": "dependency not pinned by hash detected -- score normalized to 3",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#pinned-dependencies"
},
{
"name": "SAST",
"score": 0,
"reason": "SAST tool is not run on all commits -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#sast"
},
{
"name": "Security-Policy",
"score": 0,
"reason": "security policy file not detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#security-policy"
},
{
"name": "Signed-Releases",
"score": 0,
"reason": "Project has not signed or included provenance with any releases.",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#signed-releases"
},
{
"name": "Token-Permissions",
"score": 0,
"reason": "detected GitHub workflow tokens with excessive permissions",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#token-permissions"
},
{
"name": "Vulnerabilities",
"score": 0,
"reason": "88 existing vulnerabilities detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#vulnerabilities"
}
],
"commit": "43ec260095ad60cb2c46527a72b9f857d56701bd",
"ran_at": "2026-07-28T03:43:34Z",
"aggregate_score": 3.9,
"scorecard_version": "v5.5.0"
},
"has_codeql_workflow": false,
"has_security_policy": false,
"has_dependabot_config": false
},
"contribution_flow": {
"collected": true,
"ci_last_run_at": "2026-07-28T03:41:32Z",
"oldest_open_prs": [],
"last_merged_pr_at": "2026-07-28T03:38:45Z",
"ci_last_conclusion": "SUCCESS",
"oldest_open_issues": [
{
"number": 76,
"created_at": "2026-03-03T00:20:55Z",
"last_comment_at": null,
"last_comment_author": null
},
{
"number": 77,
"created_at": "2026-03-03T00:20:59Z",
"last_comment_at": null,
"last_comment_author": null
},
{
"number": 78,
"created_at": "2026-03-03T00:21:03Z",
"last_comment_at": null,
"last_comment_author": null
},
{
"number": 85,
"created_at": "2026-03-03T03:39:59Z",
"last_comment_at": null,
"last_comment_author": null
}
]
}
},
"config": {
"disabled_metrics": [],
"disabled_categories": [],
"disabled_components": {}
},
"source": {
"url": "https://github.com/OpenAdaptAI/openadapt-evals",
"host": "github.com",
"name": "openadapt-evals",
"owner": "OpenAdaptAI"
},
"metrics": {
"overall": {
"key": "overall",
"band": "moderate",
"name": "Overall health",
"note": null,
"notes": [],
"value": 60,
"inputs": {
"security": 39,
"vitality": 81,
"community": 33,
"governance": 58,
"engineering": 81
},
"components": []
},
"categories": [
{
"key": "vitality",
"band": "good",
"name": "Vitality",
"value": 81,
"weight": 0.22,
"metrics": [
{
"key": "development_activity",
"band": "good",
"name": "Development activity",
"note": null,
"notes": [],
"value": 75,
"inputs": {
"commits_last_year": 490,
"human_commit_share": 1,
"days_since_last_push": 0,
"active_weeks_last_year": 16
},
"components": [
{
"key": "push_recency",
"name": "Push recency",
"detail": "last push 0 days ago",
"points": 36,
"status": "met",
"details": [
{
"code": "push_recency",
"params": {
"days": 0
}
}
],
"max_points": 36
},
{
"key": "commit_cadence",
"name": "Commit cadence",
"detail": "16/52 weeks with commits",
"points": 11.1,
"status": "partial",
"details": [
{
"code": "commit_cadence_weeks",
"params": {
"weeks": 16
}
}
],
"max_points": 36
},
{
"key": "commit_volume",
"name": "Commit volume",
"detail": "490 commits in the last year",
"points": 18,
"status": "met",
"details": [
{
"code": "commits_last_year",
"params": {
"count": 490
}
}
],
"max_points": 18
},
{
"key": "openssf_scorecard_maintained",
"name": "OpenSSF Scorecard: Maintained",
"detail": "24 commit(s) and 0 issue activity found in the last 90 days -- score normalized to 10",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
}
]
},
{
"key": "release_discipline",
"band": "excellent",
"name": "Release discipline",
"note": null,
"notes": [],
"value": 90,
"inputs": {
"releases_count": 100,
"latest_release_tag": "v0.90.2",
"releases_from_tags": false,
"days_since_latest_release": 0,
"mean_days_between_releases": 13.1
},
"components": [
{
"key": "ships_releases",
"name": "Ships releases",
"detail": "100 releases published",
"points": 27,
"status": "met",
"details": [
{
"code": "releases_published",
"params": {
"count": 100
}
}
],
"max_points": 27
},
{
"key": "release_recency",
"name": "Release recency",
"detail": "latest release 0 days ago",
"points": 36,
"status": "met",
"details": [
{
"code": "release_recency",
"params": {
"days": 0
}
}
],
"max_points": 36
},
{
"key": "release_cadence",
"name": "Release cadence",
"detail": "a release every ~13.1 days",
"points": 27,
"status": "met",
"details": [
{
"code": "release_cadence",
"params": {
"gap": 13.1
}
}
],
"max_points": 27
},
{
"key": "openssf_scorecard_signed_releases",
"name": "OpenSSF Scorecard: Signed-Releases",
"detail": "Project has not signed or included provenance with any releases.",
"points": 0,
"status": "missed",
"details": [],
"max_points": 10
}
]
},
{
"key": "abandonment",
"band": "excellent",
"name": "Abandonment",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"cap": null,
"state": "maintained",
"guards": [],
"signals": [],
"red_flag": false,
"multiplier_pct": 100,
"declared_reason": null,
"unverified_reason": null,
"unanswered_open_prs": null,
"unanswered_open_issues": null,
"days_since_last_merged_pr": null,
"days_since_last_human_commit": 0,
"days_since_last_human_commit_is_floor": false
},
"components": [
{
"key": "project_is_still_maintained",
"name": "Project is still maintained",
"detail": "last human commit 0 days ago",
"points": 100,
"status": "met",
"details": [
{
"code": "abandonment_maintained",
"params": {
"days": 0
}
}
],
"max_points": 100
}
]
}
],
"description": "Is the project alive — is code being written and are releases shipping?"
},
{
"key": "community",
"band": "at_risk",
"name": "Community & Adoption",
"value": 33,
"weight": 0.18,
"metrics": [
{
"key": "popularity",
"band": "critical",
"name": "Popularity & adoption",
"note": null,
"notes": [],
"value": 2,
"inputs": {
"forks": 3,
"stars": 2,
"watchers": 0,
"growth_state": "unverified",
"growth_factor_pct": 100,
"growth_unverified_reason": "no_history"
},
"components": [
{
"key": "stars",
"name": "Stars",
"detail": "2 stars",
"points": 0,
"status": "missed",
"details": [
{
"code": "stars",
"params": {
"count": 2
}
}
],
"max_points": 60
},
{
"key": "forks",
"name": "Forks",
"detail": "3 forks",
"points": 2.5,
"status": "partial",
"details": [
{
"code": "forks",
"params": {
"count": 3
}
}
],
"max_points": 25
},
{
"key": "watchers",
"name": "Watchers",
"detail": "0 watchers",
"points": 0,
"status": "missed",
"details": [
{
"code": "watchers",
"params": {
"count": 0
}
}
],
"max_points": 15
}
]
},
{
"key": "community_health",
"band": "moderate",
"name": "Community health",
"note": null,
"notes": [],
"value": 50,
"inputs": {
"has_readme": true,
"has_license": true,
"has_contributing": false,
"has_issue_template": false,
"has_code_of_conduct": false,
"has_pull_request_template": false
},
"components": [
{
"key": "readme",
"name": "README",
"detail": null,
"points": 22.5,
"status": "met",
"details": [],
"max_points": 22.5
},
{
"key": "license",
"name": "License",
"detail": "recognized license (MIT)",
"points": 22.5,
"status": "met",
"details": [
{
"code": "license_standard",
"params": {}
},
{
"code": "license_spdx",
"params": {
"spdx": "MIT"
}
}
],
"max_points": 22.5
},
{
"key": "contributing_guide",
"name": "CONTRIBUTING guide",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 18
},
{
"key": "code_of_conduct",
"name": "Code of conduct",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 13.5
},
{
"key": "issue_template",
"name": "Issue template",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.2
},
{
"key": "pr_template",
"name": "PR template",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 6.3
}
]
},
{
"key": "ecosystem_adoption",
"band": "moderate",
"name": "Ecosystem adoption (downloads)",
"note": "Excluded from scoring (no data or not applicable): Registry dependents. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"registry_dependents"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 58,
"inputs": {
"packages": [
"openadapt-evals"
],
"dependents": null,
"ecosystems": "pypi",
"total_downloads": null,
"monthly_downloads": 2941
},
"components": [
{
"key": "monthly_downloads",
"name": "Monthly downloads",
"detail": "2,941 downloads/month across pypi",
"points": 46.2,
"status": "partial",
"details": [
{
"code": "downloads_monthly",
"params": {
"count": 2941,
"ecosystems": "pypi"
}
}
],
"max_points": 80
},
{
"key": "registry_dependents",
"name": "Registry dependents",
"detail": "not reported by this ecosystem",
"points": 0,
"status": "excluded",
"details": [
{
"code": "not_reported_by_this_ecosystem",
"params": {}
}
],
"max_points": 20
}
]
}
],
"description": "Does the project have users, downloads, attention, and a welcoming setup for contributors?"
},
{
"key": "governance",
"band": "moderate",
"name": "Sustainability & Governance",
"value": 58,
"weight": 0.24,
"metrics": [
{
"key": "maintainer_resilience",
"band": "critical",
"name": "Maintainer resilience (bus factor)",
"note": null,
"notes": [],
"value": 16,
"inputs": {
"bus_factor": 1,
"contributors_sampled": 1,
"top_contributor_share": 1
},
"components": [
{
"key": "bus_factor",
"name": "Bus factor",
"detail": "1 contributor(s) cover half of all commits",
"points": 9,
"status": "partial",
"details": [
{
"code": "bus_factor",
"params": {
"count": 1
}
}
],
"max_points": 54
},
{
"key": "commit_distribution",
"name": "Commit distribution",
"detail": "top contributor authored 100% of commits",
"points": 0,
"status": "missed",
"details": [
{
"code": "top_contributor_share",
"params": {
"share": 100
}
}
],
"max_points": 22.5
},
{
"key": "contributor_breadth",
"name": "Contributor breadth",
"detail": "1 contributors",
"points": 1.4,
"status": "partial",
"details": [
{
"code": "contributors_sampled",
"params": {
"count": 1
}
}
],
"max_points": 13.5
},
{
"key": "openssf_scorecard_contributors",
"name": "OpenSSF Scorecard: Contributors",
"detail": "project has 2 contributing companies or organizations -- score normalized to 6",
"points": 6,
"status": "partial",
"details": [],
"max_points": 10
}
]
},
{
"key": "responsiveness",
"band": "moderate",
"name": "Issue & PR responsiveness",
"note": null,
"notes": [],
"value": 67,
"inputs": {
"merged_prs": 250,
"open_issues": 4,
"closed_issues": 8,
"issue_closed_ratio": 0.667,
"closed_unmerged_prs": 14
},
"components": [
{
"key": "issue_resolution",
"name": "Issue resolution",
"detail": "67% of issues closed",
"points": 31.2,
"status": "partial",
"details": [
{
"code": "issues_closed_share",
"params": {
"share": 67
}
}
],
"max_points": 46.75
},
{
"key": "pr_acceptance",
"name": "PR acceptance",
"detail": "250/264 decided PRs merged",
"points": 36.2,
"status": "partial",
"details": [
{
"code": "decided_prs_merged",
"params": {
"merged": 250,
"decided": 264
}
}
],
"max_points": 38.25
},
{
"key": "openssf_scorecard_code_review",
"name": "OpenSSF Scorecard: Code-Review",
"detail": "Found 0/30 approved changesets -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 15
}
]
},
{
"key": "stewardship",
"band": "moderate",
"name": "Ownership & stewardship",
"note": null,
"notes": [],
"value": 64,
"inputs": {
"followers": 99,
"owner_type": "Organization",
"is_verified": null,
"owner_login": "OpenAdaptAI",
"public_repos": 54,
"account_age_days": 1179
},
"components": [
{
"key": "ownership_backing",
"name": "Ownership backing",
"detail": "organization-owned",
"points": 30,
"status": "met",
"details": [
{
"code": "owner_organization",
"params": {}
}
],
"max_points": 30
},
{
"key": "verified_domain",
"name": "Verified domain",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 20
},
{
"key": "owner_reach",
"name": "Owner reach",
"detail": "99 followers of OpenAdaptAI",
"points": 14.4,
"status": "partial",
"details": [
{
"code": "owner_followers",
"params": {
"count": 99,
"login": "OpenAdaptAI"
}
}
],
"max_points": 25
},
{
"key": "track_record",
"name": "Track record",
"detail": "54 public repos, account ~3 yr old",
"points": 19.1,
"status": "partial",
"details": [
{
"code": "public_repos",
"params": {
"count": 54
}
},
{
"code": "account_age_years",
"params": {
"years": 3
}
}
],
"max_points": 25
}
]
},
{
"key": "package_maintenance",
"band": "excellent",
"name": "Package maintenance",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"packages": [
"openadapt-evals"
],
"ecosystems": "pypi",
"any_deprecated": false,
"min_days_since_publish": 0
},
"components": [
{
"key": "published_resolvable",
"name": "Published & resolvable",
"detail": "1 package(s) on pypi",
"points": 25,
"status": "met",
"details": [
{
"code": "packages_published",
"params": {
"count": 1,
"ecosystems": "pypi"
}
}
],
"max_points": 25
},
{
"key": "publish_recency",
"name": "Publish recency",
"detail": "latest publish 0 days ago",
"points": 35,
"status": "met",
"details": [
{
"code": "publish_recency",
"params": {
"days": 0
}
}
],
"max_points": 35
},
{
"key": "version_history",
"name": "Version history",
"detail": "181 published versions",
"points": 20,
"status": "met",
"details": [
{
"code": "published_versions",
"params": {
"count": 181
}
}
],
"max_points": 20
},
{
"key": "not_deprecated",
"name": "Not deprecated",
"detail": "active, not deprecated or yanked",
"points": 20,
"status": "met",
"details": [
{
"code": "package_not_deprecated",
"params": {}
}
],
"max_points": 20
}
]
}
],
"description": "Will the project survive its people — bus factor, responsiveness, who backs it, and package upkeep?"
},
{
"key": "engineering",
"band": "good",
"name": "Engineering Quality",
"value": 81,
"weight": 0.2,
"metrics": [
{
"key": "engineering_practices",
"band": "moderate",
"name": "Engineering practices",
"note": null,
"notes": [],
"value": 68,
"inputs": {
"has_ci": true,
"has_tests": true,
"has_editorconfig": false,
"has_linter_config": false,
"has_precommit_config": false
},
"components": [
{
"key": "ci_workflows",
"name": "CI workflows",
"detail": "4 workflow(s)",
"points": 24,
"status": "met",
"details": [
{
"code": "ci_workflows",
"params": {
"count": 4
}
}
],
"max_points": 24
},
{
"key": "tests_present",
"name": "Tests present",
"detail": null,
"points": 24,
"status": "met",
"details": [],
"max_points": 24
},
{
"key": "linter_config",
"name": "Linter config",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 16
},
{
"key": "pre_commit_hooks",
"name": "Pre-commit hooks",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 9.6
},
{
"key": "editorconfig",
"name": ".editorconfig",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 6.4
},
{
"key": "openssf_scorecard_ci_tests",
"name": "OpenSSF Scorecard: CI-Tests",
"detail": "19 out of 19 merged PRs checked by a CI test -- score normalized to 10",
"points": 20,
"status": "met",
"details": [],
"max_points": 20
}
]
},
{
"key": "documentation",
"band": "excellent",
"name": "Documentation",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"topics": [
"benchmarks",
"evaluation",
"gui-automation",
"openadapt",
"python"
],
"has_wiki": true,
"homepage": "https://pypi.org/project/openadapt-evals/",
"has_readme": true,
"has_docs_dir": true,
"has_description": true
},
"components": [
{
"key": "readme",
"name": "README",
"detail": null,
"points": 30,
"status": "met",
"details": [],
"max_points": 30
},
{
"key": "documentation_directory",
"name": "Documentation directory",
"detail": null,
"points": 25,
"status": "met",
"details": [],
"max_points": 25
},
{
"key": "documentation_homepage_site",
"name": "Documentation / homepage site",
"detail": "https://pypi.org/project/openadapt-evals/",
"points": 15,
"status": "met",
"details": [],
"max_points": 15
},
{
"key": "repository_description",
"name": "Repository description",
"detail": null,
"points": 10,
"status": "met",
"details": [],
"max_points": 10
},
{
"key": "topics",
"name": "Topics",
"detail": "5 topics",
"points": 10,
"status": "met",
"details": [
{
"code": "topics_count",
"params": {
"count": 5
}
}
],
"max_points": 10
},
{
"key": "wiki",
"name": "Wiki",
"detail": null,
"points": 10,
"status": "met",
"details": [],
"max_points": 10
}
]
}
],
"description": "Are baseline engineering and documentation practices in place?"
},
{
"key": "security",
"band": "at_risk",
"name": "Security",
"value": 39,
"weight": 0.16,
"metrics": [
{
"key": "security_posture",
"band": "at_risk",
"name": "Security posture",
"note": "Excluded from scoring (no data or not applicable): Branch-Protection. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"branch_protection"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 39,
"inputs": {
"source": "openssf_scorecard",
"checks_evaluated": 17,
"scorecard_version": "v5.5.0",
"checks_inconclusive": 1,
"scorecard_aggregate": 3.9
},
"components": [
{
"key": "binary_artifacts",
"name": "Binary-Artifacts",
"detail": "no binaries found in the repo",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "branch_protection",
"name": "Branch-Protection",
"detail": "internal error: error during branchesHandler.setup: internal error: some github tokens can't read classic branch protection rules: https://github.com/ossf/scorecard-action/blob/main/docs/authentication/fine-grained-auth-token.md",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 7.5
},
{
"key": "ci_tests",
"name": "CI-Tests",
"detail": "19 out of 19 merged PRs checked by a CI test -- score normalized to 10",
"points": 2.5,
"status": "met",
"details": [],
"max_points": 2.5
},
{
"key": "cii_best_practices",
"name": "CII-Best-Practices",
"detail": "no effort to earn an OpenSSF best practices badge detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 2.5
},
{
"key": "code_review",
"name": "Code-Review",
"detail": "Found 0/30 approved changesets -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "contributors",
"name": "Contributors",
"detail": "project has 2 contributing companies or organizations -- score normalized to 6",
"points": 1.5,
"status": "partial",
"details": [],
"max_points": 2.5
},
{
"key": "dangerous_workflow",
"name": "Dangerous-Workflow",
"detail": "no dangerous workflow patterns detected",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
},
{
"key": "dependency_update_tool",
"name": "Dependency-Update-Tool",
"detail": "no update tool detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "fuzzing",
"name": "Fuzzing",
"detail": "project is not fuzzed",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "license",
"name": "License",
"detail": "license file detected",
"points": 2.5,
"status": "met",
"details": [],
"max_points": 2.5
},
{
"key": "maintained",
"name": "Maintained",
"detail": "24 commit(s) and 0 issue activity found in the last 90 days -- score normalized to 10",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "packaging",
"name": "Packaging",
"detail": "packaging workflow detected",
"points": 5,
"status": "met",
"details": [],
"max_points": 5
},
{
"key": "pinned_dependencies",
"name": "Pinned-Dependencies",
"detail": "dependency not pinned by hash detected -- score normalized to 3",
"points": 1.5,
"status": "partial",
"details": [],
"max_points": 5
},
{
"key": "sast",
"name": "SAST",
"detail": "SAST tool is not run on all commits -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "security_policy",
"name": "Security-Policy",
"detail": "security policy file not detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "signed_releases",
"name": "Signed-Releases",
"detail": "Project has not signed or included provenance with any releases.",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "token_permissions",
"name": "Token-Permissions",
"detail": "detected GitHub workflow tokens with excessive permissions",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "vulnerabilities",
"name": "Vulnerabilities",
"detail": "88 existing vulnerabilities detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
}
]
}
],
"description": "Are visible security and supply-chain practices strong, with no malicious dependency and no unresolved high-risk jurisdiction exposure?"
},
{
"key": "ai_readiness",
"band": "moderate",
"name": "AI Readiness",
"value": 58,
"weight": 0,
"metrics": [
{
"key": "ai_agent_context",
"band": "excellent",
"name": "Agent context & guidance",
"note": null,
"notes": [],
"value": 85,
"inputs": {
"has_llms_txt": false,
"legible_history_share": 1,
"agent_instruction_files": [
"CLAUDE.md"
],
"agent_instruction_max_bytes": 39650
},
"components": [
{
"key": "agent_instructions",
"name": "Agent instructions",
"detail": "CLAUDE.md",
"points": 45,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "CLAUDE.md"
}
}
],
"max_points": 45
},
{
"key": "machine_readable_docs_llms_txt",
"name": "Machine-readable docs (llms.txt)",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 15
},
{
"key": "legible_commit_history",
"name": "Legible commit history",
"detail": "100 of 100 human commits state their intent (structured subject or explanatory body)",
"points": 40,
"status": "met",
"details": [
{
"code": "legible_history",
"params": {
"legible": 100,
"sampled": 100
}
}
],
"max_points": 40
}
]
},
{
"key": "ai_verify_loop",
"band": "at_risk",
"name": "Verify loop (build / test / typecheck)",
"note": null,
"notes": [],
"value": 45,
"inputs": {
"has_nix": false,
"has_tests": true,
"lockfiles": [
"uv.lock"
],
"has_dockerfile": true,
"typed_language": false,
"bootstrap_files": [],
"has_devcontainer": false,
"has_linter_config": false,
"typecheck_configs": [],
"agent_commit_share": 0.54,
"toolchain_manifests": [],
"dependency_bot_commit_share": 0
},
"components": [
{
"key": "one_command_bootstrap",
"name": "One-command bootstrap",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 18
},
{
"key": "automated_tests",
"name": "Automated tests",
"detail": null,
"points": 22,
"status": "met",
"details": [],
"max_points": 22
},
{
"key": "lint_format_config",
"name": "Lint / format config",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 11
},
{
"key": "static_type_checking",
"name": "Static type checking",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 11
},
{
"key": "reproducible_environment",
"name": "Reproducible environment",
"detail": "Dockerfile, lockfile",
"points": 10,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "Dockerfile, lockfile"
}
}
],
"max_points": 10
},
{
"key": "demonstrated_agent_practice",
"name": "Demonstrated agent practice",
"detail": "54 of the last 100 commits agent-authored or agent-credited",
"points": 10,
"status": "met",
"details": [
{
"code": "agent_authored_commits",
"params": {
"count": 54,
"sampled": 100
}
}
],
"max_points": 10
},
{
"key": "automated_maintenance",
"name": "Automated maintenance",
"detail": "no automated dependency updates observed",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_dependency_automation",
"params": {}
}
],
"max_points": 8
},
{
"key": "openssf_scorecard_pinned_dependencies",
"name": "OpenSSF Scorecard: Pinned-Dependencies",
"detail": "dependency not pinned by hash detected -- score normalized to 3",
"points": 3,
"status": "partial",
"details": [],
"max_points": 10
}
]
},
{
"key": "ai_code_legibility",
"band": "moderate",
"name": "Code legibility for models",
"note": null,
"notes": [],
"value": 54,
"inputs": {
"primary_language": "Python",
"largest_source_bytes": 304758,
"source_files_sampled": 269,
"oversized_source_files": 6
},
"components": [
{
"key": "type_checkable_code",
"name": "Type-checkable code",
"detail": "Python without a type-check config",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_typecheck_config_language",
"params": {
"language": "Python"
}
}
],
"max_points": 45
},
{
"key": "manageable_file_sizes",
"name": "Manageable file sizes",
"detail": "6/269 source files over 60KB",
"points": 53.8,
"status": "partial",
"details": [
{
"code": "oversized_source_files",
"params": {
"kb": 60,
"sampled": 269,
"oversized": 6
}
}
],
"max_points": 55
}
]
},
{
"key": "ai_interfaces",
"band": "at_risk",
"name": "Machine-readable interfaces",
"note": null,
"notes": [],
"value": 40,
"inputs": {
"example_dirs": [
"demos",
"examples"
],
"has_mcp_signal": false,
"api_schema_files": []
},
"components": [
{
"key": "api_schema_openapi_graphql_proto",
"name": "API schema (OpenAPI/GraphQL/proto)",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 40
},
{
"key": "mcp_server",
"name": "MCP server",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 20
},
{
"key": "runnable_examples",
"name": "Runnable examples",
"detail": "demos, examples",
"points": 40,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "demos, examples"
}
}
],
"max_points": 40
}
]
}
],
"description": "How well is the repo equipped to be developed and maintained with AI coding agents? An independent, experimental badge — weight 0.0, so it is surfaced on its own and does not affect the overall health score."
}
],
"metrics_version": "1.13.0"
},
"warnings": [
"Star history unavailable: GitHub GraphQL error: Resource not accessible by personal access token",
"GitHub dependency-graph SBOM unavailable (404); the dependency graph may be disabled for this repository",
"deps.dev does not index pypi:openadapt-evals@0.90.2; advisories assessed against the repository dependency graph instead"
],
"report_type": "repository",
"generated_at": "2026-07-28T03:43:52.319394Z",
"schema_version": "0.27.0",
"badge_url": "https://raw.githubusercontent.com/inspect-software/badges/main/v1/o/OpenAdaptAI/openadapt-evals.svg",
"full_name": "OpenAdaptAI/openadapt-evals",
"license_state": "standard",
"license_spdx": "MIT"
}