fabryka-english-base-250m-e01 / evaluation /instruct-base-e01.json
kacperwikiel's picture
Publish verified original E01 Instruct Bench baseline
301baff verified
Raw History Blame Contribute Delete
4.2 kB
{
"status": "verified",
"model": "SlayerLab/fabryka-english-base-250m-e01",
"model_weights_sha256": "7e3d9655bde8b62b72fece4d715ce51c99ab09c57be9aff322e187fa46e62ae9",
"benchmark": "BananaMind Instruct Bench 1.1",
"dataset_id": "BananaMind/BananaMind-Instruct-Bench-1.1",
"dataset_revision": "40494cb4a9224bfd78722968efd2bff440e08186",
"dataset_sha256": "2369407245d2d440b0be991009d0d1f28a97fd3df5c8fba8e806cbe9f31d4eb1",
"runner_sha256": "87cae0182a190da421c796cddd484ecfef630446b43b6b285bf178e4abf5d8ab",
"private_report_sha256": "4e16e8be49c9c3dce86bfce44dbd24e32937c025db7f3425714203c8f53c825f",
"device": "cuda:0",
"dtype": "bfloat16",
"prompt_format": "alpaca_instruction_fallback",
"generation_settings": {
"do_sample": false,
"repetition_penalty": 1.1,
"use_cache": true,
"seed": 42
},
"summary": {
"cases": 300,
"passed": 1,
"pass_rate": 0.0033333333333333335,
"weighted_points": 2.0250000000000004,
"possible_weighted_points": 573.5625,
"weighted_score": 0.003530565544295522,
"official_complete_run": true,
"overall_elo": 163,
"overall_elo_unrounded": 162.9586801157697,
"categories": {
"general": {
"cases": 120,
"passed": 0,
"pass_rate": 0.0,
"weighted_points": 0.0,
"possible_weighted_points": 190.0,
"weighted_score": 0.0,
"elo": 129,
"elo_unrounded": 129.29686854727566
},
"multi_turn": {
"cases": 75,
"passed": 0,
"pass_rate": 0.0,
"weighted_points": 0.0,
"possible_weighted_points": 148.4375,
"weighted_score": 0.0,
"elo": 318,
"elo_unrounded": 318.23569578211266
},
"system_prompts": {
"cases": 60,
"passed": 1,
"pass_rate": 0.016666666666666666,
"weighted_points": 2.0250000000000004,
"possible_weighted_points": 128.25,
"weighted_score": 0.01578947368421053,
"elo": 516,
"elo_unrounded": 515.6348921626286
},
"recall_in_context": {
"cases": 30,
"passed": 0,
"pass_rate": 0.0,
"weighted_points": 0.0,
"possible_weighted_points": 68.875,
"weighted_score": 0.0,
"elo": 537,
"elo_unrounded": 537.1739454704327
},
"code": {
"cases": 15,
"passed": 0,
"pass_rate": 0.0,
"weighted_points": 0.0,
"possible_weighted_points": 38.0,
"weighted_score": 0.0,
"elo": 667,
"elo_unrounded": 666.6862941801066
}
},
"difficulties": {
"easy": {
"cases": 100,
"passed": 0,
"pass_rate": 0.0,
"weighted_points": 0.0,
"possible_weighted_points": 120.75,
"weighted_score": 0.0,
"elo": 192,
"elo_unrounded": 191.59059734804697
},
"medium": {
"cases": 100,
"passed": 1,
"pass_rate": 0.01,
"weighted_points": 2.0250000000000004,
"possible_weighted_points": 181.125,
"weighted_score": 0.011180124223602487,
"elo": 343,
"elo_unrounded": 343.3623736269484
},
"hard": {
"cases": 100,
"passed": 0,
"pass_rate": 0.0,
"weighted_points": 0.0,
"possible_weighted_points": 271.6875,
"weighted_score": 0.0,
"elo": 296,
"elo_unrounded": 295.84042114108263
}
},
"natural_eos_count": 51,
"generation_limit_hits": 249,
"context_overflows": 0
},
"verification": "All 300 ordered IDs, messages, item metadata, deterministic pass judgments, weights and aggregates recomputed against the pinned dataset and runner. No inference rerun.",
"overlap_caveat": "Instruct access was obtained after pretraining. This verification does not establish pretraining-set decontamination against Instruct Bench.",
"limitations": [
"Code checks use syntax and patterns, not program execution.",
"Comparison model scores are publisher self-reports, not a local rerun.",
"Base uses the official Alpaca fallback; SFT uses native chat. A before/after comparison includes prompt-format differences."
]
}