Exact method excerpts from the frozen C01 implementation.
These show timing boundaries and statistical estimation; they are not a standalone hardware runner.


src/as_m6_bench/day0.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 50-70

def _run_request(mx:Any,model:Any,make_prompt_cache:Any,generate_step:Any,prompt:list[int],*,chunk_size:int)->dict[str,Any]:
    cache=make_prompt_cache(model);started=time.perf_counter_ns();chunks=[];last_logits=None
    for start in range(0,len(prompt),chunk_size):
        values=prompt[start:start+chunk_size];begun=time.perf_counter_ns();result=model(mx.array([values]),cache=cache)
        logits=result.logits if hasattr(result,'logits') else result;last_logits=logits[:,-1,:];mx.eval(last_logits);_eval_cache(mx,cache);ended=time.perf_counter_ns()
        chunks.append({'start_token':start,'token_count':len(values),'retained_tokens':start+len(values),'wall_seconds':(ended-begun)/1e9,'last_position_logits_shape':list(last_logits.shape)})
    if last_logits is None:raise InvalidInput('C01 prompt must contain at least one token')
    prefill_done=time.perf_counter_ns();first=mx.argmax(last_logits,axis=-1);mx.eval(first)
    first_value=int(first.item() if hasattr(first,'item') else first);events=[{'index':0,'token_id':first_value,'monotonic_ns':time.perf_counter_ns()}];output=[first_value]
    iterator=generate_step(mx.array([first_value]),model,max_tokens=127,sampler=lambda logits:mx.argmax(logits,axis=-1),prompt_cache=cache,prefill_step_size=chunk_size)
    for index,(token,_) in enumerate(iterator,1):
        value=int(token.item() if hasattr(token,'item') else token);output.append(value);events.append({'index':index,'token_id':value,'monotonic_ns':time.perf_counter_ns()})
    mx.synchronize();ended=time.perf_counter_ns()
    if len(output)!=128:raise IncompleteEvidence(f'fixed C01 request produced {len(output)} tokens instead of 128')
    decode_seconds=(events[-1]['monotonic_ns']-events[0]['monotonic_ns'])/1e9
    return {'prompt_tokens':len(prompt),'generated_tokens':128,'timing_policy':TIMING_POLICY,'prefill_seconds':(prefill_done-started)/1e9,
            'prefill_tokens_per_second':len(prompt)/max((prefill_done-started)/1e9,1e-12),'decode_seconds_first_to_last':decode_seconds,
            'decode_tokens_per_second':127/max(decode_seconds,1e-12),'request_wall_seconds':(ended-started)/1e9,
            'time_to_first_token_seconds':(events[0]['monotonic_ns']-started)/1e9,
            'prefill_chunks':chunks,'token_events':events,'output_token_ids_sha256':hashlib.sha256(json.dumps(output,separators=(',',':')).encode()).hexdigest(),
            'active_memory_bytes':int(mx.get_active_memory()),'peak_memory_bytes':int(mx.get_peak_memory())}


src/as_m6_bench/day0.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 184-191

def _session_estimate(rows:list[dict[str,Any]],field:str)->dict[str,Any]:
    import numpy as np
    sessions={session:[float(row[field]) for row in rows if row['session']==session] for session in sorted({row['session'] for row in rows})}
    if len(sessions)<2 or any(not values for values in sessions.values()):raise IncompleteEvidence('day-zero estimate requires multiple complete sessions')
    session_means=[sum(values)/len(values) for values in sessions.values()];rng=np.random.default_rng(260913);draws=[]
    for _ in range(10_000):
        picked=rng.integers(0,len(session_means),len(session_means));draws.append(float(np.mean([session_means[int(index)] for index in picked])))
    return {'mean':float(np.mean(session_means)),'ci95':[float(np.quantile(draws,.025)),float(np.quantile(draws,.975))],'session_means':{str(key):sum(value)/len(value) for key,value in sessions.items()},'samples':sum(map(len,sessions.values())),'sessions':len(sessions),'estimator':'equal-weight mean of session means; session bootstrap','bootstrap_resamples':10_000}


src/as_m6_bench/day0.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 194-199

def _ratio_estimate(current:list[dict[str,Any]],control:list[dict[str,Any]],field:str)->dict[str,Any]:
    import numpy as np
    a=_session_estimate(current,field);b=_session_estimate(control,field);rng=np.random.default_rng(260914);av=list(a['session_means'].values());bv=list(b['session_means'].values());draws=[]
    for _ in range(10_000):
        am=float(np.mean([av[int(i)] for i in rng.integers(0,len(av),len(av))]));bm=float(np.mean([bv[int(i)] for i in rng.integers(0,len(bv),len(bv))]));draws.append(am/bm)
    return {'ratio':a['mean']/b['mean'],'ci95':[float(np.quantile(draws,.025)),float(np.quantile(draws,.975))],'current':a,'control':b,'estimator':'ratio of independently resampled device session means','bootstrap_resamples':10_000}


src/as_m6_bench/mlx_runtime.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 78-81

def _eval_cache(mx: Any, cache: list[Any]) -> None:
    state = [array for entry in cache for array in _flatten_state(entry.state)]
    if state:
        mx.eval(state)


src/as_m6_bench/mlx_runtime.py, 71cbf5744ded6f97090c9afef826eb5596b44b68, lines 84-91

def _flatten_state(value: Any) -> list[Any]:
    if hasattr(value, "shape") and hasattr(value, "dtype"):
        return [value]
    if isinstance(value, (list, tuple)):
        return [item for child in value for item in _flatten_state(child)]
    if isinstance(value, dict):
        return [item for child in value.values() for item in _flatten_state(child)]
    return []
