Exact method excerpts from the frozen C01 implementation. These show timing boundaries and statistical estimation; they are not a standalone hardware runner. src/as_m6_bench/day0.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 50-70 def _run_request(mx:Any,model:Any,make_prompt_cache:Any,generate_step:Any,prompt:list[int],*,chunk_size:int)->dict[str,Any]: cache=make_prompt_cache(model);started=time.perf_counter_ns();chunks=[];last_logits=None for start in range(0,len(prompt),chunk_size): values=prompt[start:start+chunk_size];begun=time.perf_counter_ns();result=model(mx.array([values]),cache=cache) logits=result.logits if hasattr(result,'logits') else result;last_logits=logits[:,-1,:];mx.eval(last_logits);_eval_cache(mx,cache);ended=time.perf_counter_ns() chunks.append({'start_token':start,'token_count':len(values),'retained_tokens':start+len(values),'wall_seconds':(ended-begun)/1e9,'last_position_logits_shape':list(last_logits.shape)}) if last_logits is None:raise InvalidInput('C01 prompt must contain at least one token') prefill_done=time.perf_counter_ns();first=mx.argmax(last_logits,axis=-1);mx.eval(first) first_value=int(first.item() if hasattr(first,'item') else first);events=[{'index':0,'token_id':first_value,'monotonic_ns':time.perf_counter_ns()}];output=[first_value] iterator=generate_step(mx.array([first_value]),model,max_tokens=127,sampler=lambda logits:mx.argmax(logits,axis=-1),prompt_cache=cache,prefill_step_size=chunk_size) for index,(token,_) in enumerate(iterator,1): value=int(token.item() if hasattr(token,'item') else token);output.append(value);events.append({'index':index,'token_id':value,'monotonic_ns':time.perf_counter_ns()}) mx.synchronize();ended=time.perf_counter_ns() if len(output)!=128:raise IncompleteEvidence(f'fixed C01 request produced {len(output)} tokens instead of 128') decode_seconds=(events[-1]['monotonic_ns']-events[0]['monotonic_ns'])/1e9 return {'prompt_tokens':len(prompt),'generated_tokens':128,'timing_policy':TIMING_POLICY,'prefill_seconds':(prefill_done-started)/1e9, 'prefill_tokens_per_second':len(prompt)/max((prefill_done-started)/1e9,1e-12),'decode_seconds_first_to_last':decode_seconds, 'decode_tokens_per_second':127/max(decode_seconds,1e-12),'request_wall_seconds':(ended-started)/1e9, 'time_to_first_token_seconds':(events[0]['monotonic_ns']-started)/1e9, 'prefill_chunks':chunks,'token_events':events,'output_token_ids_sha256':hashlib.sha256(json.dumps(output,separators=(',',':')).encode()).hexdigest(), 'active_memory_bytes':int(mx.get_active_memory()),'peak_memory_bytes':int(mx.get_peak_memory())} src/as_m6_bench/day0.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 184-191 def _session_estimate(rows:list[dict[str,Any]],field:str)->dict[str,Any]: import numpy as np sessions={session:[float(row[field]) for row in rows if row['session']==session] for session in sorted({row['session'] for row in rows})} if len(sessions)<2 or any(not values for values in sessions.values()):raise IncompleteEvidence('day-zero estimate requires multiple complete sessions') session_means=[sum(values)/len(values) for values in sessions.values()];rng=np.random.default_rng(260913);draws=[] for _ in range(10_000): picked=rng.integers(0,len(session_means),len(session_means));draws.append(float(np.mean([session_means[int(index)] for index in picked]))) return {'mean':float(np.mean(session_means)),'ci95':[float(np.quantile(draws,.025)),float(np.quantile(draws,.975))],'session_means':{str(key):sum(value)/len(value) for key,value in sessions.items()},'samples':sum(map(len,sessions.values())),'sessions':len(sessions),'estimator':'equal-weight mean of session means; session bootstrap','bootstrap_resamples':10_000} src/as_m6_bench/day0.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 194-199 def _ratio_estimate(current:list[dict[str,Any]],control:list[dict[str,Any]],field:str)->dict[str,Any]: import numpy as np a=_session_estimate(current,field);b=_session_estimate(control,field);rng=np.random.default_rng(260914);av=list(a['session_means'].values());bv=list(b['session_means'].values());draws=[] for _ in range(10_000): am=float(np.mean([av[int(i)] for i in rng.integers(0,len(av),len(av))]));bm=float(np.mean([bv[int(i)] for i in rng.integers(0,len(bv),len(bv))]));draws.append(am/bm) return {'ratio':a['mean']/b['mean'],'ci95':[float(np.quantile(draws,.025)),float(np.quantile(draws,.975))],'current':a,'control':b,'estimator':'ratio of independently resampled device session means','bootstrap_resamples':10_000} src/as_m6_bench/mlx_runtime.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 78-81 def _eval_cache(mx: Any, cache: list[Any]) -> None: state = [array for entry in cache for array in _flatten_state(entry.state)] if state: mx.eval(state) src/as_m6_bench/mlx_runtime.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 84-91 def _flatten_state(value: Any) -> list[Any]: if hasattr(value, "shape") and hasattr(value, "dtype"): return [value] if isinstance(value, (list, tuple)): return [item for child in value for item in _flatten_state(child)] if isinstance(value, dict): return [item for child in value.values() for item in _flatten_state(child)] return []