JSON-Rohbericht maschinenlesbar
{
"data": {
"repo": {
"topics": [
"llm",
"nlp",
"tokenization",
"tokenizer"
],
"is_fork": false,
"size_kb": 3193,
"has_wiki": true,
"homepage": null,
"languages": {
"Rust": 1038876,
"Shell": 8388,
"Python": 523918
},
"pushed_at": "2026-07-21T22:59:12Z",
"created_at": "2025-11-10T02:52:34Z",
"owner_type": "User",
"updated_at": "2026-07-22T02:07:24Z",
"description": "Language model tokenization at GB/s",
"is_archived": false,
"is_disabled": false,
"license_spdx": "MIT",
"default_branch": "main",
"license_spdx_raw": "MIT",
"primary_language": "Rust",
"significant_languages": [
"Rust",
"Python"
]
},
"owner": {
"blog": null,
"name": "Marcel Rød",
"type": "User",
"login": "marcelroed",
"company": "Stanford",
"location": "Stanford, CA",
"followers": 151,
"avatar_url": "https://avatars.githubusercontent.com/u/13435604?v=4",
"created_at": "2015-07-21T13:29:01Z",
"is_verified": null,
"public_repos": 91,
"account_age_days": 4018
},
"license": {
"state": "standard",
"spdx_id": "MIT",
"raw_spdx": "MIT",
"file_present": true,
"scorecard_found": true,
"profile_has_license": true
},
"activity": {
"releases": [],
"recent_commits": [
{
"oid": "542367a3efed134883fb4f1140b49c04e6fad3a3",
"body": "Claude-Session: https://claude.ai/code/session_01ERiw4X6uv7TmVP1wUBf4WB",
"is_bot": false,
"headline": "Bump minor version to 0.9.0",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T22:48:55Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "b66a612e49b3733ec054133fef11f71bef78d5cd",
"body": "Claude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "Benchmark details wording: OWT motivation, SP note",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T22:45:49Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "06729ea31c049fa6b7c0c53b07166aefcd37b511",
"body": "Add Linux benchmark results (Ryzen 7 9800X3D + EPYC 9565)",
"is_bot": false,
"headline": "Merge pull request #24 from marcelroed/bench-fred",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T17:30:49Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "c9a4fae96db0de1e40666ede54bf142d7a099b28",
"body": "Reworded by hand; the renderer's strings are updated to match so\nre-rendering reproduces the same text.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "README wording: tagline, SP note, coverage intro",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T17:16:35Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8df1e1b2dffd7ad1fcb1d48b4fe9143e95f9ef4c",
"body": "The methodology, SentencePiece note, and per-row coverage list were\nrepeated (or cross-referenced) inside every CPU table; they now live in\none 'Benchmark details' spoiler beneath the tables. EPYC moves to the\ntop of CPU_ORDER, and CPU_DISPLAY relabels its heading to 'x 2 sockets\n(144 cores)' since cpu_label() counts SMT threads.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "Deduplicate benchmark prose into a shared details block",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T17:11:22Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "2a30c8f6e0533dc1ab8a1588b73fe78e39685b35",
"body": "Same sweep on a dual-socket AMD EPYC 9565 (288 threads): gigatoken\n15.5-24.5 GB/s on the BPE families and 2.5-4.8 GB/s on SentencePiece.\nTables now follow an explicit CPU_ORDER list (new machines appended\nunderneath) instead of sorting by peak speed, and the SP footnote no\nlonger names a row position since sort order differs per machine.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "Add EPYC 9565 (blackwell1) benchmark results as a third README table",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T17:10:28Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "fa7b808475e15cdff8838fd14e75c24f4239831f",
"body": "Same sweep protocol run on a 16-thread AMD Ryzen 7 9800X3D running\nUbuntu; results merged into benchmarks/results.json under its CPU key.\nrender now orders tables fastest-machine-first (results.json sorts CPU\nkeys alphabetically) and emits the per-row coverage footnotes only in\nthe first table, with later tables pointing to it.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "Add Linux benchmark results (Ryzen 7 9800X3D) as a second README table",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T17:10:28Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "b1b297a5bd37642312e051aa54466593b1b05f66",
"body": null,
"is_bot": false,
"headline": "Update EPYC numbers",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T17:05:06Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "28066b8605512dab5158209e4cece734f320b14f",
"body": null,
"is_bot": false,
"headline": "Known issues + some benchmark clarification",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-21T16:23:35Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "abe82e3dae86d9362d56d6555226e107beac9feb",
"body": null,
"is_bot": false,
"headline": "Add tiktoken compatibility mode",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T23:48:46Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "aa50c540b8dbef2434bdbb4820b47e6b094964a0",
"body": null,
"is_bot": false,
"headline": "Clarify subsetting of validation data for HF",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T23:17:27Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "c8ade5bc91b4ce70f05dc6f0e7a62421fbdde325",
"body": null,
"is_bot": false,
"headline": "Bump minor version, release, windows build clang shim",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T23:08:58Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "007f6b4f7435b964ce8d49b12acc4eb5afd5392a",
"body": "bench CLI: in-memory by default, --doc-separator, 100MB comparison cap",
"is_bot": false,
"headline": "Merge pull request #23 from marcelroed/bench-cli-defaults",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T22:42:41Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7afbc717fe4e70664d41735650f67d805de73f8f",
"body": "--in-memory is now the default; --stream-from-disk opts back into timing\ndisk IO, and in that mode the hf comparison reads, splits, and decodes\ninside its timed region too (lazily, capped at the comparison subset) so\nneither side gets a free preload. --comparison-limit defaults to 100MB\n('none' lift\n[…]\nor, and the\ndefault in-memory mode warns with a --stream-from-disk hint when the\nrun looks unlikely to fit in available memory.\n\nClaude-Session: https://claude.ai/code/session_01PtWzxW6nbbZgv5Eu4kwXQB",
"is_bot": false,
"headline": "bench CLI: in-memory by default, --doc-separator, 100MB comparison cap",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T22:27:42Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "7deef4959ba2dfd032746265e5bad9cf6e8ad153",
"body": null,
"is_bot": false,
"headline": "\"Keep in mind\" -> Other connectors",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T21:52:17Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "51b1c6f588fae1c80e2fb9bdd6478f05f74ddd1b",
"body": "Add cross-library throughput benchmarks and README results table",
"is_bot": false,
"headline": "Merge pull request #22 from marcelroed/bench-results",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T21:48:52Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "980b25237b33f043f6da47732afd3ca5ee84688e",
"body": "The README table names one representative per tokenizer, which hid\nnon-obvious members (MiMo under Qwen, SmolLM3 under Llama 3.3,\nGLM-4.7-Flash under GLM 5). Coverage footnotes are back, as hand-curated\nfamily lists per row: automatic suffix-stripping of the scanned repo names\nkept producing artifac\n[…]\no the COVERS map in results.py is written by hand from the\nverified families.json scan data and should be updated alongside it.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "List the model families covered by each benchmark row",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T21:46:29Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "9d042ea3c0a587048619f881c0dd5f7d7c75a195",
"body": "benchmarks/compare/ holds the pipeline: measure.py times one library\n(gigatoken/HF tokenizers/tiktoken) per fresh process on one file, named by\nHF repo id for every library so measurements line up; tiktoken runs only\nrepos its own model registry resolves (gpt2, gpt-oss), never converted\nvocabs. swee\n[…]\n(sp-parallel branch); all others on main at 34c1e25. M4 Max, 11.9 GB\nOpenWebText: BPE tokenizers 5.5-8.8 GB/s, SP 1.4-2.0 GB/s.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "Add cross-library throughput benchmarks and README results table",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T21:36:48Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "67c82c68e3455d945e562e13369527a5452f2abd",
"body": "Parallelize SentencePiece encoding within oversized documents",
"is_bot": false,
"headline": "Merge pull request #21 from marcelroed/sp-parallel",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T21:35:48Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d5a4d6f156985fbbb9ad7f39ea56e87719ebfa90",
"body": "A single huge document through an SP-backed tokenizer previously encoded\nserially (~154 MB/s on a 290 MB doc); only multi-document batches\nparallelized. Oversized SP documents are now fragmented at provably safe\nboundaries and encoded across the rayon pool, mirroring the BPE path's\ninternal chunking\n[…]\ns 154 MB/s serial).\n\nbenches/encode.rs no longer pre-splits SP input: both backends now\nreceive the whole file as one document.\n\nClaude-Session: https://claude.ai/code/session_013cwmfWS1kEXh3C9S2st3WU",
"is_bot": false,
"headline": "Parallelize SentencePiece encoding within oversized documents",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T21:26:28Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "7b84bb5c01fb12821915744af197a6f0dab94252",
"body": null,
"is_bot": false,
"headline": "Add bibtex for citation",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T19:08:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "34c1e256cc0737836d8d29a6ae5551439abe9ec9",
"body": "Refuse unsupported tokenizer model families by name",
"is_bot": false,
"headline": "Merge pull request #19 from marcelroed/model-type-errors",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T17:06:29Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "022ee7a3a9f0ddd93126098c5e22e9cc531f5a4f",
"body": "WordPiece and Unigram tokenizer.json files are valid JSON with a\ndifferent model shape, so deserializing them into the BPE schema failed\nwith misleading errors (\"missing field `merges`\", \"invalid type:\nsequence, expected a map at column 9096716\") that read as file\ncorruption. Probe model.type before\n[…]\noberta)\nand `max_input_chars_per_word` (WordPiece: bert-base-uncased).\n`continuing_subword_prefix` cannot serve as the WordPiece marker — BPE\nserializes it too (the original gpt2 upload has it as \"\").",
"is_bot": false,
"headline": "Refuse unsupported tokenizer model families by name",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T17:00:25Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "832f2f5c68c7ff605469f4d052fd344ccb9bcf94",
"body": "Support fairseq-ordered vocabs, added-token lstrip/rstrip, and ByteLevel add_prefix_space",
"is_bot": false,
"headline": "Merge pull request #18 from marcelroed/parity-fixes",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T16:37:13Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8152d8ddf71bbc75ef844d617b4ddcc6784eccff",
"body": "- Restore the doc comment and #[inline(never)] that the outlined ranked\n miss path had displaced off encode_pretoken_miss (the insertion landed\n between the attribute and the function, silently un-outlining the hot\n miss path and double-attributing the ranked one).\n- Collapse the private AddedTok\n[…]\nken_overwrites/Piece doc comment (pre-existing).\n\nNet -103 lines. 182/182 tests pass; parity re-validation unchanged\n(20/21 fixed, 11/11 sanity exact); interleaved gpt2 A/B within the A/A\nnoise floor.",
"is_bot": false,
"headline": "Simplify the parity-fix additions after multi-agent review",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T16:17:16Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "6912ead72d4c9320eeb38c04ddcf7101843e622d",
"body": "…vel add_prefix_space\n\nThree ByteLevel-BPE parity gaps found by the top-1000 HF model sweep\n(19 tokenizer groups / 42 models move from mismatch to exact vs HF):\n\n- Fairseq-heritage vocabs (RoBERTa/OPT/DeBERTa/BLIP2/CLAP): vocab IDs are\n frequency-ordered, so merge priority cannot be the merged toke\n[…]\nPerf: interleaved A/B vs main (min-of-N, ST gpt2/qwen2.5/phi-4 cold+warm,\nMT gpt2/phi-4) shows all deltas inside the A/A noise floor, with token\nstreams byte-identical to main on all three ST configs.",
"is_bot": false,
"headline": "Support fairseq-ordered vocabs, added-token lstrip/rstrip, and ByteLe…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-20T15:49:25Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "f3e44aabc9b02b40292a3bb6db71cab1a6edba32",
"body": "Config-driven tokenizer dispatch; generalize the Kimi loader",
"is_bot": false,
"headline": "Merge pull request #17 from marcelroed/config-dispatch",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-19T22:45:47Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "a650a8f5532bd20a7f0e13e9ab33fb2fa8a6b435",
"body": "Loading from a repo id, directory, or vocab-file path now dispatches on\ntokenizer_config.json as the prime source: its tokenizer_class/auto_map\nidentify registered tokenizer lines (repos shipping no tokenizer.json),\nand anything else routes to the tokenizer.json path — replacing the\nexception-driven\n[…]\n become\nload_tiktoken_model/from_tiktoken_model taking a named pretokenizer\nscheme (PretokenizerType::from_name), since only the scheme — defined\nsolely in each line's remote code — was Kimi-specific.",
"is_bot": false,
"headline": "Config-driven tokenizer dispatch; generalize the Kimi loader",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-19T22:44:22Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "2175f37e15d8a80d4389c0716d932f0e78b2050b",
"body": "Support the moonshotai Kimi tokenizer line",
"is_bot": false,
"headline": "Merge pull request #16 from marcelroed/kimi-support",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-19T17:09:14Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "37a3520fc4c78798657e33605ecdb5aec154d9cf",
"body": "…, VL, Moonlight)\n\nThe Kimi repos ship no tokenizer.json: a shared tiktoken.model rank file\n(byte-identical across all 10 repos) plus per-repo special tokens in\ntokenizer_config.json, with the pretokenizer regex living in remote code.\nThe pattern is the o200k scheme with a leading [\\p{Han}]+ alterna\n[…]\n/code cases, specials, 5079/5079 OWT docs,\n89/89 repo source files, 20k adversarial fuzz strings, zh-wikipedia.\nMask-vs-scalar matches on full ~12 GB OWT (2.37B tokens). MT throughput\n5.4 GB/s on OWT.",
"is_bot": false,
"headline": "Support the moonshotai Kimi tokenizer line (K2/K2.5/K2.6/K2.7, Linear…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-19T01:40:34Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "3973fdf20c2bc054b8f680685fa17bfb7c081a4f",
"body": "Fix 10-100x slow encode for gemma-3/4: keep unit splitting despite boundary-crossing vocab pieces",
"is_bot": false,
"headline": "Merge pull request #15 from marcelroed/gemma-encode-perf",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-19T00:24:18Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "50f58bbea351070e719a9bf7ca030e6c765c28e8",
"body": "gemma-3/4 encoded 10-100x slower than other SentencePiece configs\nbecause a single vocab piece with an interior metaspace mark (the HTML\nfragment \">▁</\") failed the all-or-nothing merge-safety proof and\ndemoted the whole tokenizer to WordSplit::None: whole-chunk merges with\nno pretoken cache. gemma-\n[…]\n)\nToken ids are byte-identical to the old path on 20 MB of OWT for all\nthree (with >▁</ forming 56 times), and a new HF-parity test covers\ncrossing pieces incl. the empty-post (trailing-▁ piece) case.",
"is_bot": false,
"headline": "Keep unit splitting when vocab pieces cross word boundaries (gemma-3/4)",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-19T00:23:21Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "5cc4453cd27bcab26d06f88a6c9da74f7b3a8009",
"body": "…s to profile more models.",
"is_bot": false,
"headline": "Load from HF directly from Rust instead of relying on Python. Use thi…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T23:36:53Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "12a3f6183602720dafd4de52430747d53e148fd9",
"body": null,
"is_bot": false,
"headline": "Bump version and release",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T22:26:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ab3584ebca0d2c14e58a6f8c6e6f327af8e5872e",
"body": null,
"is_bot": false,
"headline": "Add colors to the CLI output",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T22:25:58Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "0612ddd879fc111edf677d6bc090301e32d9f7ca",
"body": null,
"is_bot": false,
"headline": "Fix annoying code block format",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T20:15:05Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "029d79c2924b127a4478a3699a1d4d0d866ba5ad",
"body": null,
"is_bot": false,
"headline": "Add fun fact about tokenization speed",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:56:28Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "9a8a4732dc4ee382ccdab3b7c33a9495dad2c88a",
"body": null,
"is_bot": false,
"headline": "Clean up CLI output in README",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:47:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "cf7f7443ad0e65de485027efd4e4f4e2f22a4dc8",
"body": null,
"is_bot": false,
"headline": "Remove (by MB/s) in CLI bench output",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:47:23Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8595c733c5afbf1154c53c640140cee3350e8f3d",
"body": null,
"is_bot": false,
"headline": "Bump version, update README",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:46:05Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "e03efc8278efc00138c6fa8cbc7ac03f84e88c49",
"body": null,
"is_bot": false,
"headline": "Show CPU in cli bench",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:37:18Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "0f12aa08ea1e6b1cd887d193ebeed715077123a0",
"body": "Zen 5 round 2: vpcompressb flatten_bits, merge-scan rank prefetch, qwen2-family classifier cuts",
"is_bot": false,
"headline": "Merge pull request #14 from marcelroed/worktree-encode-profiling",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:15:55Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "6ba2b24395c63c4f3980492b7a30fdeebe4d53b2",
"body": "…en2-family classifier cuts\n\nProfile-guided follow-up to the §7 campaign (profiling/x86_port_plan.md\n§8 has the full round log). Full-dataset (11.9 GB OWT) encode_st:\ngpt2 939 -> 1007 MB/s (+7.2%), Qwen3-8B 807 -> 862 MB/s (+7.0%);\ntoken counts bit-identical throughout.\n\n- vpcompressb flatten_bits (\n[…]\n\nRebased onto a4bcfcb (x86-tier fill monomorphization); §8's rebase note\nrecords the re-expression. Numbers predate the rebase.\n\nClaude-Session: https://claude.ai/code/session_01Gp1DAKshDULeWydpMXBVXc",
"is_bot": false,
"headline": "Zen 5 round 2: vpcompressb flatten_bits, merge-scan rank prefetch, qw…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T18:12:47Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "b4103f8f57a4387b7c98c37b3a27a2147ec219bd",
"body": "The state-machine, combinator, portable-SIMD, and AVX-512 prototype\npretokenizers are benchmark baselines and test oracles only — nothing\nin the encode path uses them. Group them under reference/ with a\nmodule doc saying so, and drop the AVX-512 prototype's short-lived\nruntime gating: as a baseline it stays compile-time gated (build with\ntarget-cpu=native to include it), unlike the production scanners,\nwhich runtime-dispatch on any build.",
"is_bot": false,
"headline": "Move legacy pretokenizers into pretokenize::reference",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T17:53:00Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "a4bcfcbbaba9b1ee3ef78623eb0e1a71cff76167",
"body": "Hoist the AVX-512/AVX2 choice from a branch+call per 64-byte batch to\nonce per fill: fill_spans_two_phase dispatches into target_feature\nwrappers monomorphized on (CRC, tier), so the batch classifiers inline\nand the fill loop's codegen matches a target-cpu=native build. The\nper-family avx512/avx2 cl\n[…]\ns collapse into one\ninline(always) const-generic body each; next_span keeps runtime\ndispatch through thin per-tier wrappers (the shared body regresses\n25-30% if instantiated outside a feature region).",
"is_bot": false,
"headline": "Monomorphize the x86 SIMD tier into the fill loop",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T17:53:00Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "6afe050be2bed8dd2e9848f6c15663de595b37fb",
"body": null,
"is_bot": false,
"headline": "Use a wider range for speedup claim",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-18T17:16:46Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "b092ad72072f10c654122913e2a17ce11b8d32ba",
"body": null,
"is_bot": false,
"headline": "Make tests portable without having the bespoke data/ dir",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T22:58:43Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "cab5fa082798e99f39992fb5da486c5e53afe53f",
"body": null,
"is_bot": false,
"headline": "Bump version and publish",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T22:30:54Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "83321e5b2915b2599a478c7be202ead38c8e97c0",
"body": "Serve HuggingFace files from the standard HF cache without heavyweight imports",
"is_bot": false,
"headline": "Merge pull request #12 from marcelroed/hf-standard-cache",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T21:11:02Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "869d3972ab1f6ae6a0197f1038c6570a772e4803",
"body": "…t imports\n\nTest fixtures and benches no longer copy HF downloads into repo data/;\nthey resolve paths straight from the standard HF cache. Cached files are\nfound with a pure-filesystem lookup (new stdlib-only hf_hub_cache_dir /\ncached_hub_file in gigatoken._load.hub, also used as a fast path by\ndown\n[…]\nload. The r50k tiktoken\nfile (non-HF) moves to ~/.cache/gigatoken-tests/. The suite passes with\nno repo data/ directory at all.\n\nClaude-Session: https://claude.ai/code/session_011AyPaJoBHetCaU534nw7Bj",
"is_bot": false,
"headline": "Serve HuggingFace files from the standard HF cache without heavyweigh…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T21:05:54Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "164a48f33e2695a94460ba10b7f2b2bbeea48ef3",
"body": "Add ParquetFileSource",
"is_bot": false,
"headline": "Merge pull request #11 from marcelroed/parquet-source",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T18:02:03Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "43ec81137d72c736ba0d22b3fe2245a27c5de28d",
"body": "ParquetFileSource(paths, column=\"text\") reads one document per row from a\nstring or binary column. Rows are materialized as owned buffers via the\narrow-rs parquet reader (parallel over row groups, row order preserved,\nnulls become empty documents to keep results row-aligned), then encoded\nthrough th\n[…]\nrquet paths default to column\n\"text\" for both encode_files and train_bpe.\n\nThe optional polars-based training path and its 'parquet' cargo feature\nare removed; parquet support is now always available.",
"is_bot": false,
"headline": "Add ParquetFileSource; replace polars training path with built-in reader",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T18:00:08Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "c308fb31067b3bd106a35da943b4bb84c2ee4401",
"body": "Replace Rust examples with Python examples; drop unusable BPETokenizer() constructor",
"is_bot": false,
"headline": "Merge pull request #10 from marcelroed/worktree-examples-py",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T17:34:18Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "091b024372e6fb47398f428b1a6ca196cafdf554",
"body": "It loaded a hardcoded ~/data/tokenizers/r50k_base.tiktoken, a dev\nleftover that fails anywhere that file doesn't exist. Construct through\ngigatoken.Tokenizer or the from_tiktoken/from_hf staticmethods instead.",
"is_bot": false,
"headline": "Remove the no-arg BPETokenizer constructor",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T17:27:04Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "f886ca15ab46c37f9e41f45e34eb91d93fdbd857",
"body": "Four self-contained scripts covering the common use cases: the gigatoken\nAPI (quickstart), file/buffer sources (encode_files), and the HuggingFace\nand tiktoken drop-in wrappers. The old Rust examples were dev leftovers.",
"is_bot": false,
"headline": "Replace Rust examples with minimal Python examples",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T17:20:20Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "2593eae04afd18b1aecd5c4bea48b25d0d992965",
"body": null,
"is_bot": false,
"headline": "Add download for the example dataset used to test tokenization speed",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-17T01:52:41Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "e6564f3ad3364e58ff02b2b251a96b518b6eaaa2",
"body": null,
"is_bot": false,
"headline": "Minor clarification",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T17:12:54Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "01e3577446310f42259cda489944eadfce52f32b",
"body": "Claude-Session: https://claude.ai/code/session_01K1RV4CHUnaP5E2gZiG5smm",
"is_bot": false,
"headline": "Bump version to 0.5.0",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T17:09:12Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "f080dbd78a5869da7b84501143a173ef59c879bc",
"body": null,
"is_bot": false,
"headline": "Relock",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T17:07:11Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "49f9d30c8fc48f5e0dc19bcd1efd1964932cf048",
"body": null,
"is_bot": false,
"headline": "Update benchmark using BytesSource, relock",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T16:52:43Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "2d2e38e4c8ec719931841b2873305d712857d40f",
"body": "Add BytesSource: separator-splitting inside the encode for in-memory batches",
"is_bot": false,
"headline": "Merge pull request #9 from marcelroed/bytes-source",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T16:43:38Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "4fe1d5aab4a414265275be76d2e004bf7e6dc647",
"body": null,
"is_bot": false,
"headline": "Slightly less bold headline",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T16:41:08Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "97678e8d1dd2426035909235e03bbafeeefe6cc1",
"body": "…batches\n\nencode_batch (all variants) now accepts a BytesSource — borrowed byte\nbuffers plus an optional document separator — and splits documents on the\nseparator inside Rust during the parallel encode, via the same lazy\ntwo-level region machinery encode_files uses (chunk boundaries probed at\n~targ\n[…]\n conversion.\n- The CLI passes separators through (BytesSource with --in-memory,\n TextFileSource otherwise) instead of splitting in Python; byte counts\n are now whole-file bytes, separators included.",
"is_bot": false,
"headline": "Add BytesSource: separator-splitting inside the encode for in-memory …",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T16:41:00Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "f60d9cea9934f252504639c0adf286c5fedd5c2e",
"body": null,
"is_bot": false,
"headline": "Some rewording, fix typos",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T15:26:20Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "cdfd36b7a3d1ddcc583d7cd7ed8b215ec8b19b82",
"body": null,
"is_bot": false,
"headline": "Small comment on macOS benchmarks",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T04:34:08Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "bd46d1c874eb1b6640fa3f8cf55a3bb325129d83",
"body": null,
"is_bot": false,
"headline": "Bump patch version, update README",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T04:26:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "39b88cf80c34971bc4d3333b661fd87ab4abc1f1",
"body": null,
"is_bot": false,
"headline": "Add small FAQ section",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T04:11:39Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ed4609ca22ad633967e94753d497c39c2a74d486",
"body": null,
"is_bot": false,
"headline": "Bump minor version",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T03:49:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7a08d376a3ce40d20082c6405fe26177854642bc",
"body": null,
"is_bot": false,
"headline": "Add CLI for benchmarking",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T03:35:52Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7ce2cea510e8739a15db192edbac99ded5a0387d",
"body": null,
"is_bot": false,
"headline": "Fix README gigatoken API example, clarify plot text",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-16T03:35:43Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "94974e32f6360da4c9ca8c20e65c41006b07cef3",
"body": null,
"is_bot": false,
"headline": "README wording",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T21:46:59Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "79d369cd195c30a7f6c782785f60dcac6832432a",
"body": "…ython.",
"is_bot": false,
"headline": "Handle more of the compat implementation from Rust rather than from P…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T21:46:52Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "4cb24232adaf22142ec7c91969167a7d96019b39",
"body": null,
"is_bot": false,
"headline": "Add some clarification on plot",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T21:35:18Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d0fc756e5d6a61b4fce875ee00297b151c89eff7",
"body": null,
"is_bot": false,
"headline": "README wording",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T21:20:49Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "4311e389074a45871c34c694d3fc4ddb06d48b32",
"body": null,
"is_bot": false,
"headline": "Complete README sentence",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T21:15:13Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "ce7e5fa9a35048450e91e60d42d364e84e5a6f07",
"body": null,
"is_bot": false,
"headline": "Update README to include plot and a more complete description",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T20:58:51Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "a9987c05fa50ed93349f2a2551dadc984543db19",
"body": null,
"is_bot": false,
"headline": "Add back winnow to avoid build issues",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T18:00:19Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "d89abd5dc1ccf172f9d2d4a1a0c8d312cf580104",
"body": null,
"is_bot": false,
"headline": "Remove unused dependencies, move profiling-specific deps",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T17:46:24Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "34b71205ad3f644f6065de3280f8f69153419393",
"body": "Add o200k and Nemotron-3 fast pretokenizers, gemma-3/4 SP support",
"is_bot": false,
"headline": "Merge pull request #7 from marcelroed/o200k-support",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T16:13:34Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "8b2a352d3456905c8a564b25688b8aba3630b2ac",
"body": "Factor the duplicated ASCII CaseState initializer, drop an unused\npub(crate), merge duplicate test scaffolding, and document that the mark\nbad-smear's stragglers are covered by MaskState's resume masking (with a\npadded case pinning the 4-byte-follower behavior).\n\nClaude-Session: https://claude.ai/code/session_0176ZSLsz7QhxfirEC2RDxio",
"is_bot": false,
"headline": "Simplify o200k_family after review pass",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T05:57:20Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "35aeb27535e03e6642571048a0da25e86f31ca33",
"body": "- o200k_family: shared scalar walker + mask-scanner boundary algebra for\n the o200k_base regex (gpt-oss, GPT-4o) and the Nemotron-3 variant\n (const-generic over CONTRACTIONS/DIGITS3). Cased letter runs, suffix\n contractions, and [\\r\\n/]* punct tails; O200kCharClass packed table.\n- SP loader: acce\n[…]\ndigits (the pd\n seed cannot see the straddling char); found by the o200k port's\n differential fuzz, regression test included.\n\nClaude-Session: https://claude.ai/code/session_0176ZSLsz7QhxfirEC2RDxio",
"is_bot": false,
"headline": "Add o200k and Nemotron-3 fast pretokenizers, gemma-3/4 SP loader support",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T05:21:41Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "64578ec42e1d17097cb824c979135387fc498b52",
"body": null,
"is_bot": false,
"headline": "Add GLM 5.2 test, fix ignore_merges for weird vocab entries",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-15T01:00:30Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "30af5bc9bcb3bbd58b5aa1fee2df035d9509c288",
"body": "Encode performance: SIMD pretokenization, memoized encode pipeline, tokenizer coverage, and the encode-opt campaign",
"is_bot": false,
"headline": "Merge pull request #6 from marcelroed/encode-perf",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T22:49:04Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "f03a41198cc7825e1612b1c9bc32358e960c3f88",
"body": "Encode optimization campaign: cold/warm encode pins, x86-64 port, MT overlap, sequential batch paths",
"is_bot": false,
"headline": "Merge pull request #5 from marcelroed/encode-opt-main",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T22:23:21Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "4a9b4736b143cc66864dc95b3ec3183d02c08f24",
"body": "encode-opt-x86 madvises each parallel chunk's token output before the\nencode's stores fault it in; the sequential BPE paths added on this\nbranch reserve the same kind of buffer at whole-batch scale (and the\nparallel path's Committer reservation already gets the hint), so give\nthem the same treatment. The SP serial paths grow from empty like the\nSP parallel chunk encoders, so there is nothing to hint there.\n\nClaude-Session: https://claude.ai/code/session_01MCLuW9pUhvDjuFeGxsFWEn",
"is_bot": false,
"headline": "Extend the merged hugepage hint to the serial encode output buffers",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T21:52:42Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "7794a4bfcf7ddc1d1252e1eac05d1365298f3fa6",
"body": null,
"is_bot": false,
"headline": "Merge remote-tracking branch 'origin/encode-opt-x86' into seq-mp-threads",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T21:49:23Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "c63fb14ffc1debf1a9446d2ee0689a1189ecf46a",
"body": "…s, emit-loop mask fold, wider dense rank grid\n\nFull-dataset (11.9 GB OWT): encode_st 763 -> 948 MB/s (+24%),\n16-thread encode 5.15 -> 6.1 GB/s (+18%), token counts bit-identical.\nDriven by the perf deep-profile in profiling/zen5_st_profile.md; the\ncampaign log lives in profiling/x86_port_plan.md §7\n[…]\nral\nscalar folds, or x86-only paths; NEON dispatch and table shapes are\nunchanged except dense_log2 (arch-neutral, A/B'd here).\n\nClaude-Session: https://claude.ai/code/session_015V7Rv4HMfFSesXchG5hiWP",
"is_bot": false,
"headline": "Zen 5 profile-guided encode optimization: THP ordering/alignment fixe…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T20:33:55Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "bbbf2ed8d81c6bf30bf98fbf2bc221e51a71cd0c",
"body": "Batch encodes fan out on rayon's process-global pool, which composes\nbadly with Python multiprocessing: every worker sizes a pool to all\ncores, and a pool built before os.fork has no threads in the child, so\na parallel encode there waits forever (verified empirically).\n\nRust: every batch entry point\n[…]\nparallel/sequential output identity on both backends, and the fork-\nafter-warmup regression that deadlocked before this change.\n\nClaude-Session: https://claude.ai/code/session_01MCLuW9pUhvDjuFeGxsFWEn",
"is_bot": false,
"headline": "Sequential batch paths + auto-detection for multiprocessing workers",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T20:00:56Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "e806e612c4d2377eff36ead62a9544c907523206",
"body": "…rt-merge SIMD port measured negative\n\nEverything runs and verifies on x86_64 (Ryzen 7 9800X3D, AVX-512, baseline\nruntime-dispatch build): full suite plus the heavy OWT differentials (50 MB\nmemoized, 1 GB public + join, 200 MB multi-tokenizer, 1 GB parallel ragged\nin both LPT modes, family/r50k mask\n[…]\ntial-covered by short_merges_match_vec_merge_loop on\nany x86 host (both tiers forced explicitly, not just the dispatcher pick).\n\nClaude-Session: https://claude.ai/code/session_015V7Rv4HMfFSesXchG5hiWP",
"is_bot": false,
"headline": "Zen 5 session: first x86 metal validation of the encode-opt port; sho…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T18:05:27Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "8d704a3464a6b2ea8b61d4a3fd8309670ed8d523",
"body": "…ied neutral)\n\n- encode_with_added_tokens{,_flat}: replace the two 6-arm concrete-\n pretokenizer matches with PretokenizerType::pretokenize enum dispatch —\n superseded since 20bd089, when FastPretokenizerDispatch's PretokenSpans\n impl moved scheme dispatch to once per 256-pretoken chunk fill; the\n[…]\n03 vs 1002 MB/s (8 pairs, +0.1%) — neutral. Hot-symbol\ninstruction histograms match control modulo regalloc/address immediates.\n\nClaude-Session: https://claude.ai/code/session_0148GT36xf76W9YNHtp7AmJ4",
"is_bot": false,
"headline": "Simplify dispatch, fill, prefetch, and Committer structure (A/B-verif…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T17:52:22Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "ff7c821a9525fd2f1b8f956b2264611788683715",
"body": "…ant docs\n\nReviewed simplifications with no hot-path codegen impact:\n- tiktoken.rs: shared from_tables construction tail; apply_added_token_overwrites\n is now the single body of the cache-seed sync invariant (fork_sized +\n set_added_tokens); forward vocab-seed iteration (order provably irrelevant\n\n[…]\nuite green (65 passed). Campaign-history numbers live in\nprofiling/campaign_report.md; code comments now state invariants only.\n\nClaude-Session: https://claude.ai/code/session_0148GT36xf76W9YNHtp7AmJ4",
"is_bot": false,
"headline": "Simplify post-campaign accretions: shared helpers, test dedup, invari…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T17:32:57Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "9b79130694695cf495d84b3118d47544cdeae4ab",
"body": "… math\n\nReview findings F1/F2 on the x86 port (opt/x86-port @ 4b352fe):\n\nF1: the SSE4.2 CRC hash arm was compile-time gated\n(cfg(target_feature = \"sse4.2\")), which ships in NO artifact: wheels/CI\nset no target-cpu/target-feature flags and distributed wheels must stay\nbaseline x86-64, so only local -\n[…]\nkens (exact);\ncargo check --release --target x86_64-apple-darwin clean both baseline\nand with RUSTFLAGS='-C target-cpu=znver2'.\n\nClaude-Session: https://claude.ai/code/session_013C91S1p8LohRQtdHKW8qfW",
"is_bot": false,
"headline": "Runtime-select the CRC32C pretoken hash on x86_64; fix plan-doc clamp…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T13:44:36Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "2d46c6ae05c36b51be6d4868582bc5b318506a63",
"body": null,
"is_bot": false,
"headline": "Merge branch 'opt/x86-port' into encode-opt-main",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T13:24:51Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "4b352fe8888903c8798f5408e6dd63b423cd9d0c",
"body": "…, SSE4.2 CRC hash\n\nTwo implement-now items, decided from first principles + znver2/x86-64-v3\ncodegen inspection (no timing on this machine; see the plan doc):\n\n- ProbeView::probe_pair: the mask-arithmetic x86 fallback re-folds under\n LLVM into a cmov of the slot ADDRESS feeding a dependent load (v\n[…]\nntical (cfg-gated): rebuilt, ENCODE_MB=100 encode_st =\n22834020 tokens, encode_doc gpt2 = 22723342, cargo test --release green.\n\nClaude-Session: https://claude.ai/code/session_013C91S1p8LohRQtdHKW8qfW",
"is_bot": false,
"headline": "x86-64 port of the encode-opt aarch64 pins: probe_pair cmov-of-values…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T13:04:41Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "279ab896daa7ca267f0505ab9f44adbb3fb968de",
"body": "…onsolidation, agent ledger\n\n- §10: opt/split-table (288fa8c) rejected with interleaved A/B numbers\n (MT -22%, ST -32%); correct but ALU/spill costs beat the footprint win\n on M4; noted as a low-priority EPYC retry starting from the branch.\n- §11: final fresh-eyes review summary (no critical/major\n[…]\naudit-verified).\n- §13: campaign ledger — 30 Fable subagents by role, plus the in-flight\n x86-port-prep agent on opt/x86-port.\n\nClaude-Session: https://claude.ai/code/session_013C91S1p8LohRQtdHKW8qfW",
"is_bot": false,
"headline": "Campaign report addendum: split-table rejection, final review, test c…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T12:46:26Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "dd28809a01359cbbd8baed0a38749eee93481042",
"body": "The pre-move base pointer is now used only during the shared commit\nphase (Committer unmoved); finish() re-derives the destination from the\nVec after destructuring, so no write or set_len goes through a pointer\nthat predates a Vec move (strict-aliasing retag hazard flagged by the\nfinal fix review). SyncPtr::at carries its bounds contract as unsafe fn.\n\nClaude-Session: https://claude.ai/code/session_013C91S1p8LohRQtdHKW8qfW",
"is_bot": false,
"headline": "Harden Committer: post-move pointer re-derivation, unsafe fn at()",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T12:15:20Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "a13c125aa94531ad7ef0e7737285b0417589ceb2",
"body": "…pt campaign\n\nThe five optimization rounds merged test modules from several agent\nbranches into src/bpe/tiktoken.rs; this deduplicates and tidies them\nwithout changing any kept test's logic:\n\n- Drop verify_heavy's boundary-fuzz / edge-length tests and their helpers\n (check_partition, check_scheme_e\n[…]\ndata/owt_train.txt and data/*.json\npass when run individually; the three ~12 GB full-OWT tests compile.\nNo new clippy warnings.\n\nClaude-Session: https://claude.ai/code/session_013C91S1p8LohRQtdHKW8qfW",
"is_bot": false,
"headline": "Consolidate accreted test modules in tiktoken.rs/batch.rs after the o…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T12:12:49Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "dd116b8d602fd794114c880438bda24149da1b1d",
"body": "… final numbers\n\n- Round 5 concluded: opt/mt5 merged (+4.4% MT, 8680 vs 8311 mean —\n overlapped prefix-commit gather, single-chunk no-copy); opt/walker5\n rejected at +-0, with its dependency analysis recorded as the proof\n that the walker is at its floor (phase B issue-bound at 25 inst/span,\n ph\n[…]\nranch inventory and open-items updated; stale in-flight/round-5\n statements and the unqualified bit-identical claim corrected.\n\nClaude-Session: https://claude.ai/code/session_013C91S1p8LohRQtdHKW8qfW",
"is_bot": false,
"headline": "Update campaign report: round-5 results, verification and fix rounds,…",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T12:02:55Z",
"body_truncated": true,
"is_coding_agent": false
},
{
"oid": "47726828354cd1cc6c789b738e3949ca13bee3cf",
"body": "# Conflicts:\n#\tsrc/batch.rs",
"is_bot": false,
"headline": "Merge branch 'opt/mt5' into encode-opt-main",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T11:52:44Z",
"body_truncated": false,
"is_coding_agent": false
},
{
"oid": "437d4c3022c50fcbe22d771f46388131019a7c2c",
"body": "# Conflicts:\n#\tsrc/bpe/tiktoken.rs",
"is_bot": false,
"headline": "Merge branch 'opt/fix-walker-edge' into encode-opt-main",
"author_name": "Marcel Rød",
"author_login": "marcelroed",
"committed_at": "2026-07-14T11:47:00Z",
"body_truncated": false,
"is_coding_agent": false
}
],
"releases_count": 0,
"commits_last_year": 355,
"latest_release_at": null,
"latest_release_tag": null,
"releases_from_tags": false,
"days_since_last_push": 0,
"active_weeks_last_year": 18,
"days_since_latest_release": null,
"mean_days_between_releases": null
},
"community": {
"has_readme": true,
"has_license": true,
"has_description": true,
"has_contributing": false,
"health_percentage": 42,
"has_issue_template": false,
"has_code_of_conduct": false,
"has_pull_request_template": false
},
"ecosystem": {
"packages": [
{
"name": "gigatoken",
"exists": true,
"license": "MIT",
"keywords": [
"tokenizer",
"tokenizers",
"tokenization",
"bpe",
"nlp",
"sentencepiece",
"tiktoken",
"Development Status :: 4 - Beta",
"Intended Audience :: Developers",
"Intended Audience :: Science/Research",
"Operating System :: OS Independent",
"Programming Language :: Python :: 3 :: Only",
"Programming Language :: Python :: 3.10",
"Programming Language :: Python :: 3.11",
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.13",
"Programming Language :: Python :: 3.14",
"Programming Language :: Python :: Implementation :: CPython",
"Programming Language :: Rust",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Topic :: Text Processing :: Linguistic"
],
"ecosystem": "pypi",
"matches_repo": true,
"registry_url": "https://pypi.org/project/gigatoken/",
"is_deprecated": false,
"latest_version": "0.9.0",
"repository_url": "https://github.com/marcelroed/gigatoken",
"versions_count": 10,
"total_downloads": null,
"dependents_count": null,
"deprecation_note": null,
"maintainers_count": null,
"monthly_downloads": 2741,
"first_published_at": "2026-07-12T23:22:04.102600Z",
"latest_published_at": "2026-07-21T22:59:37.441143Z",
"latest_version_yanked": null,
"days_since_latest_publish": 0
}
]
},
"popularity": {
"forks": 5,
"stars": 115,
"watchers": 0,
"fork_history": {
"days": [
{
"date": "2026-07-16",
"count": 1
},
{
"date": "2026-07-21",
"count": 3
},
{
"date": "2026-07-22",
"count": 1
}
],
"complete": true,
"collected": 5,
"total_forks": 5
},
"star_history": {
"days": [
{
"date": "2026-07-16",
"count": 7
},
{
"date": "2026-07-18",
"count": 2
},
{
"date": "2026-07-21",
"count": 45
},
{
"date": "2026-07-22",
"count": 61
}
],
"complete": true,
"collected": 115,
"total_stars": 115
},
"open_issues_and_prs": 0
},
"ai_readiness": {
"has_nix": false,
"example_dirs": [
"examples"
],
"has_llms_txt": false,
"has_dockerfile": false,
"has_mcp_signal": false,
"bootstrap_files": [],
"api_schema_files": [],
"has_devcontainer": false,
"typecheck_configs": [
"gigatoken/py.typed"
],
"toolchain_manifests": [
"Cargo.toml"
],
"largest_source_bytes": 113509,
"source_files_sampled": 123,
"oversized_source_files": 6,
"agent_instruction_files": [],
"agent_instruction_max_bytes": null
},
"dependencies": {
"manifests": [
"Cargo.toml",
"pyproject.toml"
],
"advisories": {
"error": null,
"scope": null,
"source": null,
"findings": [],
"collected": false,
"malicious": [],
"truncated": false,
"by_severity": {},
"advisory_count": 0,
"affected_count": 0,
"assessed_count": 0,
"malicious_count": 0,
"assessed_package": null,
"unassessed_count": 0,
"direct_affected_count": 0
},
"ecosystems": [
"crates",
"pypi"
],
"dependencies": [
{
"name": "dashmap",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "6.1"
},
{
"name": "indicatif",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.18"
},
{
"name": "itertools",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.15"
},
{
"name": "priority-queue",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "2.0"
},
{
"name": "pyo3",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.29"
},
{
"name": "numpy",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.29"
},
{
"name": "rayon",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "1.11"
},
{
"name": "rustc-hash",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "2.1"
},
{
"name": "memmap2",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.9"
},
{
"name": "base64",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.22"
},
{
"name": "icu",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "2.2"
},
{
"name": "eyre",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.6.12"
},
{
"name": "sonic-rs",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.5.8"
},
{
"name": "memchr",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "2.8.0"
},
{
"name": "aho-corasick",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "1.1"
},
{
"name": "serde",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "1.0.228"
},
{
"name": "flate2",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "1.1"
},
{
"name": "zstd",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.13"
},
{
"name": "spm_precompiled",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.1.4"
},
{
"name": "simdutf",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "0.7.0"
},
{
"name": "winnow",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "1"
},
{
"name": "parquet",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "59.1.0"
},
{
"name": "arrow-array",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "59"
},
{
"name": "arrow-schema",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "59"
},
{
"name": "ureq",
"manifest": "Cargo.toml",
"ecosystem": "crates",
"version_constraint": "3.3.0"
},
{
"name": "awkward",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=2.6.3"
},
{
"name": "numpy",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=1.24.4"
},
{
"name": "typer",
"manifest": "pyproject.toml",
"ecosystem": "pypi",
"version_constraint": ">=0.12"
}
],
"all_dependencies": {
"error": "GitHub dependency-graph SBOM unavailable (404); the dependency graph may be disabled for this repository",
"source": null,
"packages": [],
"collected": false,
"truncated": false,
"total_count": null,
"direct_count": null,
"indirect_count": null
}
},
"maintainership": {
"issues": {
"open_prs": 0,
"merged_prs": 17,
"open_issues": 0,
"closed_ratio": null,
"closed_issues": 0,
"closed_unmerged_prs": 7
},
"bus_factor": 1,
"bot_contributors": 0,
"top_contributors": [
{
"type": "User",
"login": "marcelroed",
"commits": 355,
"avatar_url": "https://avatars.githubusercontent.com/u/13435604?v=4"
}
],
"contributors_sampled": 1,
"top_contributor_share": 1
},
"quality_signals": {
"has_ci": true,
"has_tests": true,
"ci_workflows": [
"CI.yml",
"wheels.yml"
],
"has_docs_dir": false,
"linter_configs": [],
"has_editorconfig": false,
"has_linter_config": false,
"has_precommit_config": false
},
"security_signals": {
"lockfiles": [
"Cargo.lock",
"uv.lock"
],
"scorecard": {
"checks": [
{
"name": "Binary-Artifacts",
"score": 10,
"reason": "no binaries found in the repo",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#binary-artifacts"
},
{
"name": "Branch-Protection",
"score": 0,
"reason": "branch protection not enabled on development/release branches",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#branch-protection"
},
{
"name": "CI-Tests",
"score": 0,
"reason": "0 out of 8 merged PRs checked by a CI test -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#ci-tests"
},
{
"name": "CII-Best-Practices",
"score": 0,
"reason": "no effort to earn an OpenSSF best practices badge detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#cii-best-practices"
},
{
"name": "Code-Review",
"score": 0,
"reason": "Found 0/16 approved changesets -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#code-review"
},
{
"name": "Contributors",
"score": 10,
"reason": "project has 3 contributing companies or organizations -- score normalized to 10",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#contributors"
},
{
"name": "Dangerous-Workflow",
"score": 10,
"reason": "no dangerous workflow patterns detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#dangerous-workflow"
},
{
"name": "Dependency-Update-Tool",
"score": 0,
"reason": "no update tool detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#dependency-update-tool"
},
{
"name": "Fuzzing",
"score": 0,
"reason": "project is not fuzzed",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#fuzzing"
},
{
"name": "License",
"score": 10,
"reason": "license file detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#license"
},
{
"name": "Maintained",
"score": 10,
"reason": "30 commit(s) and 0 issue activity found in the last 90 days -- score normalized to 10",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#maintained"
},
{
"name": "Packaging",
"score": null,
"reason": "packaging workflow not detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#packaging"
},
{
"name": "Pinned-Dependencies",
"score": 1,
"reason": "dependency not pinned by hash detected -- score normalized to 1",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#pinned-dependencies"
},
{
"name": "SAST",
"score": 0,
"reason": "SAST tool is not run on all commits -- score normalized to 0",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#sast"
},
{
"name": "Security-Policy",
"score": 0,
"reason": "security policy file not detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#security-policy"
},
{
"name": "Signed-Releases",
"score": null,
"reason": "no releases found",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#signed-releases"
},
{
"name": "Token-Permissions",
"score": 10,
"reason": "GitHub workflow tokens follow principle of least privilege",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#token-permissions"
},
{
"name": "Vulnerabilities",
"score": 9,
"reason": "1 existing vulnerabilities detected",
"documentation_url": "https://github.com/ossf/scorecard/blob/c395761df6afe1a69e476bc60a013a94bcbc153f/docs/checks.md#vulnerabilities"
}
],
"commit": "542367a3efed134883fb4f1140b49c04e6fad3a3",
"ran_at": "2026-07-22T02:08:49Z",
"aggregate_score": 4.8,
"scorecard_version": "v5.5.0"
},
"has_codeql_workflow": false,
"has_security_policy": false,
"has_dependabot_config": false
},
"contribution_flow": {
"collected": true,
"ci_last_run_at": "2026-07-21T22:59:13Z",
"oldest_open_prs": [],
"last_merged_pr_at": "2026-07-21T17:30:49Z",
"ci_last_conclusion": null,
"oldest_open_issues": []
}
},
"config": {
"disabled_metrics": [],
"disabled_categories": [],
"disabled_components": {}
},
"source": {
"url": "https://github.com/marcelroed/gigatoken",
"host": "github.com",
"name": "gigatoken",
"owner": "marcelroed"
},
"metrics": {
"overall": {
"key": "overall",
"band": "moderate",
"name": "Overall health",
"note": null,
"notes": [],
"value": 50,
"inputs": {
"security": 48,
"vitality": 46,
"community": 47,
"governance": 54,
"engineering": 53
},
"components": []
},
"categories": [
{
"key": "vitality",
"band": "at_risk",
"name": "Vitality",
"value": 46,
"weight": 0.22,
"metrics": [
{
"key": "development_activity",
"band": "good",
"name": "Development activity",
"note": null,
"notes": [],
"value": 76,
"inputs": {
"commits_last_year": 355,
"human_commit_share": 1,
"days_since_last_push": 0,
"active_weeks_last_year": 18
},
"components": [
{
"key": "push_recency",
"name": "Push recency",
"detail": "last push 0 days ago",
"points": 36,
"status": "met",
"details": [
{
"code": "push_recency",
"params": {
"days": 0
}
}
],
"max_points": 36
},
{
"key": "commit_cadence",
"name": "Commit cadence",
"detail": "18/52 weeks with commits",
"points": 12.5,
"status": "partial",
"details": [
{
"code": "commit_cadence_weeks",
"params": {
"weeks": 18
}
}
],
"max_points": 36
},
{
"key": "commit_volume",
"name": "Commit volume",
"detail": "355 commits in the last year",
"points": 18,
"status": "met",
"details": [
{
"code": "commits_last_year",
"params": {
"count": 355
}
}
],
"max_points": 18
},
{
"key": "openssf_scorecard_maintained",
"name": "OpenSSF Scorecard: Maintained",
"detail": "30 commit(s) and 0 issue activity found in the last 90 days -- score normalized to 10",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
}
]
},
{
"key": "release_discipline",
"band": "critical",
"name": "Release discipline",
"note": "Excluded from scoring (no data or not applicable): OpenSSF Scorecard: Signed-Releases. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"openssf_scorecard_signed_releases"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 1,
"inputs": {
"releases_count": 0
},
"components": [
{
"key": "ships_releases",
"name": "Ships releases",
"detail": "no releases published",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_releases_published",
"params": {}
}
],
"max_points": 27
},
{
"key": "release_recency",
"name": "Release recency",
"detail": "no releases",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_releases",
"params": {}
}
],
"max_points": 36
},
{
"key": "release_cadence",
"name": "Release cadence",
"detail": "no releases",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_releases",
"params": {}
}
],
"max_points": 27
},
{
"key": "openssf_scorecard_signed_releases",
"name": "OpenSSF Scorecard: Signed-Releases",
"detail": "no releases found",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 10
}
]
},
{
"key": "abandonment",
"band": "excellent",
"name": "Abandonment",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"cap": null,
"state": "maintained",
"guards": [],
"signals": [],
"red_flag": false,
"multiplier_pct": 100,
"declared_reason": null,
"unverified_reason": null,
"unanswered_open_prs": null,
"unanswered_open_issues": null,
"days_since_last_merged_pr": null,
"days_since_last_human_commit": 0,
"days_since_last_human_commit_is_floor": false
},
"components": [
{
"key": "project_is_still_maintained",
"name": "Project is still maintained",
"detail": "last human commit 0 days ago",
"points": 100,
"status": "met",
"details": [
{
"code": "abandonment_maintained",
"params": {
"days": 0
}
}
],
"max_points": 100
}
]
}
],
"description": "Is the project alive — is code being written and are releases shipping?"
},
{
"key": "community",
"band": "at_risk",
"name": "Community & Adoption",
"value": 47,
"weight": 0.18,
"metrics": [
{
"key": "popularity",
"band": "at_risk",
"name": "Popularity & adoption",
"note": null,
"notes": [],
"value": 38,
"inputs": {
"forks": 5,
"stars": 115,
"watchers": 0,
"growth_state": "unverified",
"growth_factor_pct": 100,
"growth_unverified_reason": "window_too_short"
},
"components": [
{
"key": "stars",
"name": "Stars",
"detail": "115 stars",
"points": 33.4,
"status": "partial",
"details": [
{
"code": "stars",
"params": {
"count": 115
}
}
],
"max_points": 60
},
{
"key": "forks",
"name": "Forks",
"detail": "5 forks",
"points": 5,
"status": "partial",
"details": [
{
"code": "forks",
"params": {
"count": 5
}
}
],
"max_points": 25
},
{
"key": "watchers",
"name": "Watchers",
"detail": "0 watchers",
"points": 0,
"status": "missed",
"details": [
{
"code": "watchers",
"params": {
"count": 0
}
}
],
"max_points": 15
}
]
},
{
"key": "community_health",
"band": "moderate",
"name": "Community health",
"note": null,
"notes": [],
"value": 50,
"inputs": {
"has_readme": true,
"has_license": true,
"has_contributing": false,
"has_issue_template": false,
"has_code_of_conduct": false,
"has_pull_request_template": false
},
"components": [
{
"key": "readme",
"name": "README",
"detail": null,
"points": 22.5,
"status": "met",
"details": [],
"max_points": 22.5
},
{
"key": "license",
"name": "License",
"detail": "recognized license (MIT)",
"points": 22.5,
"status": "met",
"details": [
{
"code": "license_standard",
"params": {}
},
{
"code": "license_spdx",
"params": {
"spdx": "MIT"
}
}
],
"max_points": 22.5
},
{
"key": "contributing_guide",
"name": "CONTRIBUTING guide",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 18
},
{
"key": "code_of_conduct",
"name": "Code of conduct",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 13.5
},
{
"key": "issue_template",
"name": "Issue template",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.2
},
{
"key": "pr_template",
"name": "PR template",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 6.3
}
]
},
{
"key": "ecosystem_adoption",
"band": "moderate",
"name": "Ecosystem adoption (downloads)",
"note": "Excluded from scoring (no data or not applicable): Registry dependents. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"registry_dependents"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 57,
"inputs": {
"packages": [
"gigatoken"
],
"dependents": null,
"ecosystems": "pypi",
"total_downloads": null,
"monthly_downloads": 2741
},
"components": [
{
"key": "monthly_downloads",
"name": "Monthly downloads",
"detail": "2,741 downloads/month across pypi",
"points": 45.8,
"status": "partial",
"details": [
{
"code": "downloads_monthly",
"params": {
"count": 2741,
"ecosystems": "pypi"
}
}
],
"max_points": 80
},
{
"key": "registry_dependents",
"name": "Registry dependents",
"detail": "not reported by this ecosystem",
"points": 0,
"status": "excluded",
"details": [
{
"code": "not_reported_by_this_ecosystem",
"params": {}
}
],
"max_points": 20
}
]
}
],
"description": "Does the project have users, downloads, attention, and a welcoming setup for contributors?"
},
{
"key": "governance",
"band": "moderate",
"name": "Sustainability & Governance",
"value": 54,
"weight": 0.24,
"metrics": [
{
"key": "maintainer_resilience",
"band": "critical",
"name": "Maintainer resilience (bus factor)",
"note": null,
"notes": [],
"value": 20,
"inputs": {
"bus_factor": 1,
"contributors_sampled": 1,
"top_contributor_share": 1
},
"components": [
{
"key": "bus_factor",
"name": "Bus factor",
"detail": "1 contributor(s) cover half of all commits",
"points": 9,
"status": "partial",
"details": [
{
"code": "bus_factor",
"params": {
"count": 1
}
}
],
"max_points": 54
},
{
"key": "commit_distribution",
"name": "Commit distribution",
"detail": "top contributor authored 100% of commits",
"points": 0,
"status": "missed",
"details": [
{
"code": "top_contributor_share",
"params": {
"share": 100
}
}
],
"max_points": 22.5
},
{
"key": "contributor_breadth",
"name": "Contributor breadth",
"detail": "1 contributors",
"points": 1.4,
"status": "partial",
"details": [
{
"code": "contributors_sampled",
"params": {
"count": 1
}
}
],
"max_points": 13.5
},
{
"key": "openssf_scorecard_contributors",
"name": "OpenSSF Scorecard: Contributors",
"detail": "project has 3 contributing companies or organizations -- score normalized to 10",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
}
]
},
{
"key": "responsiveness",
"band": "moderate",
"name": "Issue & PR responsiveness",
"note": "Excluded from scoring (no data or not applicable): Issue resolution. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"issue_resolution"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 51,
"inputs": {
"merged_prs": 17,
"open_issues": 0,
"closed_issues": 0,
"issue_closed_ratio": null,
"closed_unmerged_prs": 7
},
"components": [
{
"key": "issue_resolution",
"name": "Issue resolution",
"detail": "no issues or no data",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_issues_or_data",
"params": {}
}
],
"max_points": 46.75
},
{
"key": "pr_acceptance",
"name": "PR acceptance",
"detail": "17/24 decided PRs merged",
"points": 27.1,
"status": "partial",
"details": [
{
"code": "decided_prs_merged",
"params": {
"merged": 17,
"decided": 24
}
}
],
"max_points": 38.25
},
{
"key": "openssf_scorecard_code_review",
"name": "OpenSSF Scorecard: Code-Review",
"detail": "Found 0/16 approved changesets -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 15
}
]
},
{
"key": "stewardship",
"band": "moderate",
"name": "Ownership & stewardship",
"note": "Excluded from scoring (no data or not applicable): Verified domain. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"verified_domain"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 63,
"inputs": {
"followers": 151,
"owner_type": "User",
"is_verified": null,
"owner_login": "marcelroed",
"public_repos": 91,
"account_age_days": 4018
},
"components": [
{
"key": "ownership_backing",
"name": "Ownership backing",
"detail": "personal (user) account",
"points": 10,
"status": "partial",
"details": [
{
"code": "owner_personal",
"params": {}
}
],
"max_points": 30
},
{
"key": "verified_domain",
"name": "Verified domain",
"detail": "not applicable to user accounts",
"points": 0,
"status": "excluded",
"details": [
{
"code": "not_applicable_to_user_accounts",
"params": {}
}
],
"max_points": 20
},
{
"key": "owner_reach",
"name": "Owner reach",
"detail": "151 followers of marcelroed",
"points": 15.7,
"status": "partial",
"details": [
{
"code": "owner_followers",
"params": {
"count": 151,
"login": "marcelroed"
}
}
],
"max_points": 25
},
{
"key": "track_record",
"name": "Track record",
"detail": "91 public repos, account ~11 yr old",
"points": 25,
"status": "met",
"details": [
{
"code": "public_repos",
"params": {
"count": 91
}
},
{
"code": "account_age_years",
"params": {
"years": 11
}
}
],
"max_points": 25
}
]
},
{
"key": "package_maintenance",
"band": "excellent",
"name": "Package maintenance",
"note": null,
"notes": [],
"value": 100,
"inputs": {
"packages": [
"gigatoken"
],
"ecosystems": "pypi",
"any_deprecated": false,
"min_days_since_publish": 0
},
"components": [
{
"key": "published_resolvable",
"name": "Published & resolvable",
"detail": "1 package(s) on pypi",
"points": 25,
"status": "met",
"details": [
{
"code": "packages_published",
"params": {
"count": 1,
"ecosystems": "pypi"
}
}
],
"max_points": 25
},
{
"key": "publish_recency",
"name": "Publish recency",
"detail": "latest publish 0 days ago",
"points": 35,
"status": "met",
"details": [
{
"code": "publish_recency",
"params": {
"days": 0
}
}
],
"max_points": 35
},
{
"key": "version_history",
"name": "Version history",
"detail": "10 published versions",
"points": 20,
"status": "met",
"details": [
{
"code": "published_versions",
"params": {
"count": 10
}
}
],
"max_points": 20
},
{
"key": "not_deprecated",
"name": "Not deprecated",
"detail": "active, not deprecated or yanked",
"points": 20,
"status": "met",
"details": [
{
"code": "package_not_deprecated",
"params": {}
}
],
"max_points": 20
}
]
}
],
"description": "Will the project survive its people — bus factor, responsiveness, who backs it, and package upkeep?"
},
{
"key": "engineering",
"band": "moderate",
"name": "Engineering Quality",
"value": 53,
"weight": 0.2,
"metrics": [
{
"key": "engineering_practices",
"band": "at_risk",
"name": "Engineering practices",
"note": null,
"notes": [],
"value": 48,
"inputs": {
"has_ci": true,
"has_tests": true,
"has_editorconfig": false,
"has_linter_config": false,
"has_precommit_config": false
},
"components": [
{
"key": "ci_workflows",
"name": "CI workflows",
"detail": "2 workflow(s)",
"points": 24,
"status": "met",
"details": [
{
"code": "ci_workflows",
"params": {
"count": 2
}
}
],
"max_points": 24
},
{
"key": "tests_present",
"name": "Tests present",
"detail": null,
"points": 24,
"status": "met",
"details": [],
"max_points": 24
},
{
"key": "linter_config",
"name": "Linter config",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 16
},
{
"key": "pre_commit_hooks",
"name": "Pre-commit hooks",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 9.6
},
{
"key": "editorconfig",
"name": ".editorconfig",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 6.4
},
{
"key": "openssf_scorecard_ci_tests",
"name": "OpenSSF Scorecard: CI-Tests",
"detail": "0 out of 8 merged PRs checked by a CI test -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 20
}
]
},
{
"key": "documentation",
"band": "moderate",
"name": "Documentation",
"note": null,
"notes": [],
"value": 60,
"inputs": {
"topics": [
"llm",
"nlp",
"tokenization",
"tokenizer"
],
"has_wiki": true,
"homepage": null,
"has_readme": true,
"has_docs_dir": false,
"has_description": true
},
"components": [
{
"key": "readme",
"name": "README",
"detail": null,
"points": 30,
"status": "met",
"details": [],
"max_points": 30
},
{
"key": "documentation_directory",
"name": "Documentation directory",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 25
},
{
"key": "documentation_homepage_site",
"name": "Documentation / homepage site",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 15
},
{
"key": "repository_description",
"name": "Repository description",
"detail": null,
"points": 10,
"status": "met",
"details": [],
"max_points": 10
},
{
"key": "topics",
"name": "Topics",
"detail": "4 topics",
"points": 10,
"status": "met",
"details": [
{
"code": "topics_count",
"params": {
"count": 4
}
}
],
"max_points": 10
},
{
"key": "wiki",
"name": "Wiki",
"detail": null,
"points": 10,
"status": "met",
"details": [],
"max_points": 10
}
]
}
],
"description": "Are baseline engineering and documentation practices in place?"
},
{
"key": "security",
"band": "at_risk",
"name": "Security",
"value": 48,
"weight": 0.16,
"metrics": [
{
"key": "security_posture",
"band": "at_risk",
"name": "Security posture",
"note": "Excluded from scoring (no data or not applicable): Packaging, Signed-Releases. Remaining weights renormalized.",
"notes": [
{
"code": "excluded_no_data",
"params": {
"components": [
"packaging",
"signed_releases"
]
}
},
{
"code": "weights_renormalized",
"params": {}
}
],
"value": 48,
"inputs": {
"source": "openssf_scorecard",
"checks_evaluated": 16,
"scorecard_version": "v5.5.0",
"checks_inconclusive": 2,
"scorecard_aggregate": 4.8
},
"components": [
{
"key": "binary_artifacts",
"name": "Binary-Artifacts",
"detail": "no binaries found in the repo",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "branch_protection",
"name": "Branch-Protection",
"detail": "branch protection not enabled on development/release branches",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "ci_tests",
"name": "CI-Tests",
"detail": "0 out of 8 merged PRs checked by a CI test -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 2.5
},
{
"key": "cii_best_practices",
"name": "CII-Best-Practices",
"detail": "no effort to earn an OpenSSF best practices badge detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 2.5
},
{
"key": "code_review",
"name": "Code-Review",
"detail": "Found 0/16 approved changesets -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "contributors",
"name": "Contributors",
"detail": "project has 3 contributing companies or organizations -- score normalized to 10",
"points": 2.5,
"status": "met",
"details": [],
"max_points": 2.5
},
{
"key": "dangerous_workflow",
"name": "Dangerous-Workflow",
"detail": "no dangerous workflow patterns detected",
"points": 10,
"status": "met",
"details": [],
"max_points": 10
},
{
"key": "dependency_update_tool",
"name": "Dependency-Update-Tool",
"detail": "no update tool detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 7.5
},
{
"key": "fuzzing",
"name": "Fuzzing",
"detail": "project is not fuzzed",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "license",
"name": "License",
"detail": "license file detected",
"points": 2.5,
"status": "met",
"details": [],
"max_points": 2.5
},
{
"key": "maintained",
"name": "Maintained",
"detail": "30 commit(s) and 0 issue activity found in the last 90 days -- score normalized to 10",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "packaging",
"name": "Packaging",
"detail": "packaging workflow not detected",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 5
},
{
"key": "pinned_dependencies",
"name": "Pinned-Dependencies",
"detail": "dependency not pinned by hash detected -- score normalized to 1",
"points": 0.5,
"status": "partial",
"details": [],
"max_points": 5
},
{
"key": "sast",
"name": "SAST",
"detail": "SAST tool is not run on all commits -- score normalized to 0",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "security_policy",
"name": "Security-Policy",
"detail": "security policy file not detected",
"points": 0,
"status": "missed",
"details": [],
"max_points": 5
},
{
"key": "signed_releases",
"name": "Signed-Releases",
"detail": "no releases found",
"points": 0,
"status": "excluded",
"details": [
{
"code": "no_data",
"params": {}
}
],
"max_points": 7.5
},
{
"key": "token_permissions",
"name": "Token-Permissions",
"detail": "GitHub workflow tokens follow principle of least privilege",
"points": 7.5,
"status": "met",
"details": [],
"max_points": 7.5
},
{
"key": "vulnerabilities",
"name": "Vulnerabilities",
"detail": "1 existing vulnerabilities detected",
"points": 6.8,
"status": "partial",
"details": [],
"max_points": 7.5
}
]
},
{
"key": "high_risk_jurisdiction_exposure",
"band": "excellent",
"name": "High-Risk Jurisdiction Exposure",
"note": "Only high-confidence self-published location evidence affects this multiplier. Ambiguous matches are review-only; country evidence is not proof of nationality, citizenship, legal registration, malicious intent, or sanctions status.",
"notes": [
{
"code": "jurisdiction_evidence_limits",
"params": {}
}
],
"value": 100,
"inputs": {
"meaning": "self-published location evidence; not nationality or citizenship",
"red_flag": false,
"exposures": [],
"policy_countries": [
"Russia",
"Iran",
"North Korea"
],
"review_only_matches": 0,
"assessed_self_published_locations": 2
},
"components": [
{
"key": "policy_exposure_multiplier",
"name": "Policy exposure multiplier",
"detail": "no confirmed policy-scope location match",
"points": 100,
"status": "met",
"details": [
{
"code": "jurisdiction_no_match",
"params": {}
}
],
"max_points": 100
}
]
}
],
"description": "Are visible security and supply-chain practices strong, with no malicious dependency and no unresolved high-risk jurisdiction exposure?"
},
{
"key": "ai_readiness",
"band": "moderate",
"name": "AI Readiness",
"value": 52,
"weight": 0,
"metrics": [
{
"key": "ai_agent_context",
"band": "critical",
"name": "Agent context & guidance",
"note": null,
"notes": [],
"value": 28,
"inputs": {
"has_llms_txt": false,
"legible_history_share": 0.53,
"agent_instruction_files": [],
"agent_instruction_max_bytes": null
},
"components": [
{
"key": "agent_instructions",
"name": "Agent instructions",
"detail": "no CLAUDE.md / AGENTS.md / editor rules",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_agent_instructions",
"params": {}
}
],
"max_points": 45
},
{
"key": "machine_readable_docs_llms_txt",
"name": "Machine-readable docs (llms.txt)",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 15
},
{
"key": "legible_commit_history",
"name": "Legible commit history",
"detail": "53 of 100 human commits state their intent (structured subject or explanatory body)",
"points": 28.3,
"status": "partial",
"details": [
{
"code": "legible_history",
"params": {
"legible": 53,
"sampled": 100
}
}
],
"max_points": 40
}
]
},
{
"key": "ai_verify_loop",
"band": "moderate",
"name": "Verify loop (build / test / typecheck)",
"note": null,
"notes": [],
"value": 57,
"inputs": {
"has_nix": false,
"has_tests": true,
"lockfiles": [
"Cargo.lock",
"uv.lock"
],
"has_dockerfile": false,
"typed_language": true,
"bootstrap_files": [],
"has_devcontainer": false,
"has_linter_config": false,
"typecheck_configs": [
"gigatoken/py.typed"
],
"agent_commit_share": 0,
"toolchain_manifests": [
"Cargo.toml"
],
"dependency_bot_commit_share": 0
},
"components": [
{
"key": "one_command_bootstrap",
"name": "One-command bootstrap",
"detail": "Cargo.toml (toolchain convention, no task runner)",
"points": 12.6,
"status": "partial",
"details": [
{
"code": "toolchain_convention",
"params": {
"files": "Cargo.toml"
}
}
],
"max_points": 18
},
{
"key": "automated_tests",
"name": "Automated tests",
"detail": null,
"points": 22,
"status": "met",
"details": [],
"max_points": 22
},
{
"key": "lint_format_config",
"name": "Lint / format config",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 11
},
{
"key": "static_type_checking",
"name": "Static type checking",
"detail": "gigatoken/py.typed",
"points": 11,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "gigatoken/py.typed"
}
}
],
"max_points": 11
},
{
"key": "reproducible_environment",
"name": "Reproducible environment",
"detail": "lockfile",
"points": 10,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "lockfile"
}
}
],
"max_points": 10
},
{
"key": "demonstrated_agent_practice",
"name": "Demonstrated agent practice",
"detail": "no agent-authored commits among the last 100",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_agent_authored_commits",
"params": {
"sampled": 100
}
}
],
"max_points": 10
},
{
"key": "automated_maintenance",
"name": "Automated maintenance",
"detail": "no automated dependency updates observed",
"points": 0,
"status": "missed",
"details": [
{
"code": "no_dependency_automation",
"params": {}
}
],
"max_points": 8
},
{
"key": "openssf_scorecard_pinned_dependencies",
"name": "OpenSSF Scorecard: Pinned-Dependencies",
"detail": "dependency not pinned by hash detected -- score normalized to 1",
"points": 1,
"status": "partial",
"details": [],
"max_points": 10
}
]
},
{
"key": "ai_code_legibility",
"band": "excellent",
"name": "Code legibility for models",
"note": null,
"notes": [],
"value": 97,
"inputs": {
"primary_language": "Rust",
"largest_source_bytes": 113509,
"source_files_sampled": 123,
"oversized_source_files": 6
},
"components": [
{
"key": "type_checkable_code",
"name": "Type-checkable code",
"detail": "Rust (statically typed)",
"points": 45,
"status": "met",
"details": [
{
"code": "statically_typed_language",
"params": {
"language": "Rust"
}
}
],
"max_points": 45
},
{
"key": "manageable_file_sizes",
"name": "Manageable file sizes",
"detail": "6/123 source files over 60KB",
"points": 52.3,
"status": "partial",
"details": [
{
"code": "oversized_source_files",
"params": {
"kb": 60,
"sampled": 123,
"oversized": 6
}
}
],
"max_points": 55
}
]
},
{
"key": "ai_interfaces",
"band": "at_risk",
"name": "Machine-readable interfaces",
"note": null,
"notes": [],
"value": 40,
"inputs": {
"example_dirs": [
"examples"
],
"has_mcp_signal": false,
"api_schema_files": []
},
"components": [
{
"key": "api_schema_openapi_graphql_proto",
"name": "API schema (OpenAPI/GraphQL/proto)",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 40
},
{
"key": "mcp_server",
"name": "MCP server",
"detail": null,
"points": 0,
"status": "missed",
"details": [],
"max_points": 20
},
{
"key": "runnable_examples",
"name": "Runnable examples",
"detail": "examples",
"points": 40,
"status": "met",
"details": [
{
"code": "file_list",
"params": {
"files": "examples"
}
}
],
"max_points": 40
}
]
}
],
"description": "How well is the repo equipped to be developed and maintained with AI coding agents? An independent, experimental badge — weight 0.0, so it is surfaced on its own and does not affect the overall health score."
}
],
"metrics_version": "1.13.0"
},
"warnings": [
"Could not fetch crates package 'gigatoken' from its registry",
"GitHub dependency-graph SBOM unavailable (404); the dependency graph may be disabled for this repository",
"deps.dev does not index pypi:gigatoken@0.9.0; advisories assessed against the repository dependency graph instead"
],
"report_type": "repository",
"generated_at": "2026-07-22T02:09:00.483329Z",
"schema_version": "0.26.0",
"badge_url": "https://raw.githubusercontent.com/inspect-software/badges/main/v1/m/marcelroed/gigatoken.svg",
"full_name": "marcelroed/gigatoken",
"license_state": "standard",
"license_spdx": "MIT"
}