[{"data":1,"prerenderedAt":468},["ShallowReactive",2],{"docs-\u002Fdocs\u002Fv2\u002Fbench\u002Fbuilding\u002Fpublishing":3},{"id":4,"title":5,"body":6,"description":460,"extension":461,"meta":462,"navigation":463,"path":464,"seo":465,"stem":466,"__hash__":467},"docs\u002Fdocs\u002Fv2\u002Fbench\u002Fbuilding\u002Fpublishing.md","Publishing",{"type":7,"value":8,"toc":452},"minimark",[9,13,17,40,80,88,93,105,108,112,261,271,275,317,324,328,331,334,341,363,370,374,403,406,410,413,441,448],[10,11,5],"h1",{"id":12},"publishing",[14,15,16],"p",{},"A benchmark is a registry artifact like an agent, a module, or a cognet.",[18,19,24],"pre",{"className":20,"code":21,"language":22,"meta":23,"style":23},"language-bash shiki shiki-themes dark-plus","axon publish\n","bash","",[25,26,27],"code",{"__ignoreMap":23},[28,29,32,36],"span",{"class":30,"line":31},"line",1,[28,33,35],{"class":34},"sCudf","axon",[28,37,39],{"class":38},"sKc5r"," publish\n",[18,41,43],{"className":20,"code":42,"language":22,"meta":23,"style":23},"✓ Published @you\u002Fparser-bench@0.1.0 (public)\n  run it with `axon clone @you\u002Fparser-bench`\n",[25,44,45,60],{"__ignoreMap":23},[28,46,47,50,53,56],{"class":30,"line":31},[28,48,49],{"class":34},"✓",[28,51,52],{"class":38}," Published",[28,54,55],{"class":38}," @you\u002Fparser-bench@0.1.0",[28,57,59],{"class":58},"sTNBD"," (public)\n",[28,61,63,66,69,72,75,77],{"class":30,"line":62},2,[28,64,65],{"class":34},"  run",[28,67,68],{"class":38}," it",[28,70,71],{"class":38}," with",[28,73,74],{"class":38}," `",[28,76,35],{"class":34},[28,78,79],{"class":38}," clone @you\u002Fparser-bench`\n",[14,81,82,83,87],{},"What ships is the ",[84,85,86],"strong",{},"definition"," — config, tests, workspace. Not the results.",[89,90,92],"h2",{"id":91},"why-results-are-not-part-of-the-artifact","Why results are not part of the artifact",[14,94,95,96,100,101,104],{},"A run's observations belong to whoever spent the tokens producing them. Publishing a\nbenchmark says ",[97,98,99],"em",{},"here is an experiment you can run","; publishing a score says ",[97,102,103],{},"here is what I\nmeasured",". They are different claims, and bundling them would make the second one\nunavoidable.",[14,106,107],{},"It also keeps the artifact honest. A benchmark whose published form included its author's\nnumbers would quietly become a leaderboard where the author always ran first, on their\nhardware, with their models.",[89,109,111],{"id":110},"what-gets-included","What gets included",[18,113,115],{"className":20,"code":114,"language":22,"meta":23,"style":23},"my-bench\u002F\n├── workspace\u002F          ✓ the world, so the task is reproducible\n├── fixtures\u002F           ✓ rubrics and expected outputs\n├── tests\u002F              ✓ the scenarios\n├── bench.config.ts     ✓ matrix, schema, setup\n├── package.json        ✓ identity and engine dependencies\n└── .bench\u002Fruns\u002F        ✗ your run history stays local\n",[25,116,117,122,153,176,192,212,234],{"__ignoreMap":23},[28,118,119],{"class":30,"line":31},[28,120,121],{"class":34},"my-bench\u002F\n",[28,123,124,127,130,133,136,139,142,144,147,150],{"class":30,"line":62},[28,125,126],{"class":34},"├──",[28,128,129],{"class":38}," workspace\u002F",[28,131,132],{"class":38},"          ✓",[28,134,135],{"class":38}," the",[28,137,138],{"class":38}," world,",[28,140,141],{"class":38}," so",[28,143,135],{"class":38},[28,145,146],{"class":38}," task",[28,148,149],{"class":38}," is",[28,151,152],{"class":38}," reproducible\n",[28,154,156,158,161,164,167,170,173],{"class":30,"line":155},3,[28,157,126],{"class":34},[28,159,160],{"class":38}," fixtures\u002F",[28,162,163],{"class":38},"           ✓",[28,165,166],{"class":38}," rubrics",[28,168,169],{"class":38}," and",[28,171,172],{"class":38}," expected",[28,174,175],{"class":38}," outputs\n",[28,177,179,181,184,187,189],{"class":30,"line":178},4,[28,180,126],{"class":34},[28,182,183],{"class":38}," tests\u002F",[28,185,186],{"class":38},"              ✓",[28,188,135],{"class":38},[28,190,191],{"class":38}," scenarios\n",[28,193,195,197,200,203,206,209],{"class":30,"line":194},5,[28,196,126],{"class":34},[28,198,199],{"class":38}," bench.config.ts",[28,201,202],{"class":38},"     ✓",[28,204,205],{"class":38}," matrix,",[28,207,208],{"class":38}," schema,",[28,210,211],{"class":38}," setup\n",[28,213,215,217,220,223,226,228,231],{"class":30,"line":214},6,[28,216,126],{"class":34},[28,218,219],{"class":38}," package.json",[28,221,222],{"class":38},"        ✓",[28,224,225],{"class":38}," identity",[28,227,169],{"class":38},[28,229,230],{"class":38}," engine",[28,232,233],{"class":38}," dependencies\n",[28,235,237,240,243,246,249,252,255,258],{"class":30,"line":236},7,[28,238,239],{"class":34},"└──",[28,241,242],{"class":38}," .bench\u002Fruns\u002F",[28,244,245],{"class":38},"        ✗",[28,247,248],{"class":38}," your",[28,250,251],{"class":38}," run",[28,253,254],{"class":38}," history",[28,256,257],{"class":38}," stays",[28,259,260],{"class":38}," local\n",[14,262,263,266,267,270],{},[25,264,265],{},"workspace\u002F"," ships because the world ",[97,268,269],{},"is"," the task. A benchmark that told you to supply\nyour own repository would not be the same benchmark twice.",[89,272,274],{"id":273},"running-someone-elses","Running someone else's",[18,276,278],{"className":20,"code":277,"language":22,"meta":23,"style":23},"axon clone @you\u002Fparser-bench\ncd parser-bench\naxon bench prepare\naxon bench run\n",[25,279,280,290,298,308],{"__ignoreMap":23},[28,281,282,284,287],{"class":30,"line":31},[28,283,35],{"class":34},[28,285,286],{"class":38}," clone",[28,288,289],{"class":38}," @you\u002Fparser-bench\n",[28,291,292,295],{"class":30,"line":62},[28,293,294],{"class":34},"cd",[28,296,297],{"class":38}," parser-bench\n",[28,299,300,302,305],{"class":30,"line":155},[28,301,35],{"class":34},[28,303,304],{"class":38}," bench",[28,306,307],{"class":38}," prepare\n",[28,309,310,312,314],{"class":30,"line":178},[28,311,35],{"class":34},[28,313,304],{"class":38},[28,315,316],{"class":38}," run\n",[14,318,319,320,323],{},"Their scenarios, their workspace, their measurement schema — against whatever matrix you\ndeclare. Swap the models for the ones you care about, point the ",[25,321,322],{},"agent"," axis at your own\nproject, and the numbers you get are comparable with theirs because everything except what\nyou deliberately varied is identical.",[89,325,327],{"id":326},"comparability","Comparability",[14,329,330],{},"Results carry a manifest, and the manifest carries hashes: the benchmark's content, the\nmeasurement schema, the workspace, the harness version, and a pin for every axis — including\nthe ones held constant.",[14,332,333],{},"That is what decides whether two results can be compared at all. Same benchmark, same\nschema, same workspace, compatible harness: comparable. Anything else and the results\ndescribe different experiments that happen to share a name.",[14,335,336,337,340],{},"Which is why bumping the version matters when what you ",[97,338,339],{},"measure"," changes, not only when the\ncode does:",[18,342,344],{"className":20,"code":343,"language":22,"meta":23,"style":23},"npm version minor\naxon publish\n",[25,345,346,357],{"__ignoreMap":23},[28,347,348,351,354],{"class":30,"line":31},[28,349,350],{"class":34},"npm",[28,352,353],{"class":38}," version",[28,355,356],{"class":38}," minor\n",[28,358,359,361],{"class":30,"line":62},[28,360,35],{"class":34},[28,362,39],{"class":38},[14,364,365,366,369],{},"A benchmark that redefined ",[25,367,368],{},"resolved"," without a version bump would let old and new numbers\nland in the same aggregate. That is worse than having no aggregate — a wrong comparison is\nharder to notice than a missing one.",[89,371,373],{"id":372},"visibility","Visibility",[18,375,377],{"className":20,"code":376,"language":22,"meta":23,"style":23},"axon publish            # private — you and your org\naxon publish --public   # listed in the registry\n",[25,378,379,390],{"__ignoreMap":23},[28,380,381,383,386],{"class":30,"line":31},[28,382,35],{"class":34},[28,384,385],{"class":38}," publish",[28,387,389],{"class":388},"sOLPB","            # private — you and your org\n",[28,391,392,394,396,400],{"class":30,"line":62},[28,393,35],{"class":34},[28,395,385],{"class":38},[28,397,399],{"class":398},"scz_3"," --public",[28,401,402],{"class":388},"   # listed in the registry\n",[14,404,405],{},"Private is the default. A benchmark measuring something specific to your product is often\nexactly the benchmark you want, and exactly the one nobody else needs.",[89,407,409],{"id":408},"writing-a-readme","Writing a README",[14,411,412],{},"The registry shows it, and for a benchmark it carries weight the code cannot:",[414,415,416,423,429,435],"ul",{},[417,418,419,422],"li",{},[84,420,421],{},"What this measures",", in one sentence",[417,424,425,428],{},[84,426,427],{},"Why the task is representative"," — why this bug, this repo, this prompt",[417,430,431,434],{},[84,432,433],{},"Known limitations"," — what it does not tell you",[417,436,437,440],{},[84,438,439],{},"Suggested trials"," — how many runs before the numbers mean anything",[14,442,443,444,447],{},"The last one is worth stating explicitly. Someone cloning a benchmark will run it with\nwhatever ",[25,445,446],{},"trials"," you shipped, and a default of 1 published without comment quietly invites\na single-sample result to be treated as a finding.",[449,450,451],"style",{},"html pre.shiki code .sCudf, html code.shiki .sCudf{--shiki-default:#DCDCAA}html pre.shiki code .sKc5r, html code.shiki .sKc5r{--shiki-default:#CE9178}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html pre.shiki code .sTNBD, html code.shiki .sTNBD{--shiki-default:#D4D4D4}html pre.shiki code .sOLPB, html code.shiki .sOLPB{--shiki-default:#6A9955}html pre.shiki code .scz_3, html code.shiki .scz_3{--shiki-default:#569CD6}",{"title":23,"searchDepth":62,"depth":62,"links":453},[454,455,456,457,458,459],{"id":91,"depth":62,"text":92},{"id":110,"depth":62,"text":111},{"id":273,"depth":62,"text":274},{"id":326,"depth":62,"text":327},{"id":372,"depth":62,"text":373},{"id":408,"depth":62,"text":409},"Ship the experiment, not the score.","md",{},true,"\u002Fdocs\u002Fv2\u002Fbench\u002Fbuilding\u002Fpublishing",{"title":5,"description":460},"docs\u002Fv2\u002Fbench\u002Fbuilding\u002Fpublishing","s9fkDEjSUCL3BaVW3J3tMK_FaNg83m0jD4pNxiucsng",1785237008148]