#!/usr/bin/env python3 """Frozen synthetic ladder. No paid requests without --run. Never retry a reservation. Public export is allowlisted: no auth, account IDs, balances or reasoning traces. """ import argparse import datetime as dt import fcntl import hashlib import itertools import json import math import os from pathlib import Path import random import statistics import time import urllib.error import urllib.request MODELS = ['z-ai/glm-5.3-flash', 'deepseek/deepseek-v4.1-flash', 'tencent/hy4-preview', 'moonshotai/kimi-k3', 'openai/gpt-6-astra'] MAX_TOKENS, ROUNDS, CAP, FX = 2048, 3, 1400, 200 BASE = 'https://api.teai.io' SYSTEM = 'Solve the supplied synthetic task. Return only the requested JSON, without markdown or explanation. No tools are available.' def canonical(value): return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(',', ':')) def plan_score(jobs, order, switch): by_id = {j['id']: j for j in jobs} if len(order) != len(jobs) or set(order) != set(by_id): return None elapsed = score = 0 previous, seen = None, set() for name in order: job = by_id[name] if not set(job['after']) <= seen: return None elapsed += job['duration'] + (switch if previous is not None and previous != job['group'] else 0) score += elapsed * job['weight'] seen.add(name) previous = job['group'] return score def solve_plan(jobs, switch): # Subset DP: state must retain the last group because switching costs time. # Elapsed time also belongs to the state (different switch counts). states = {(0, None, 0): (0, ())} for _ in jobs: following = {} for (mask, previous, elapsed), (score, order) in states.items(): done = {jobs[i]['id'] for i in range(len(jobs)) if mask & (1 << i)} for i, job in enumerate(jobs): if mask & (1 << i) or not set(job['after']) <= done: continue end = elapsed + job['duration'] + (switch if previous is not None and previous != job['group'] else 0) key = (mask | (1 << i), job['group'], end) value = (score + end * job['weight'], order + (job['id'],)) if key not in following or value < following[key]: following[key] = value states = following score, order = min(states.values()) return {'order': list(order), 'score': score} def solve_ledger(invoices, events): latest, unique = {}, {} for row in invoices: if row['id'] not in latest or row['version'] > latest[row['id']]['version']: latest[row['id']] = row for row in events: unique.setdefault(row['id'], row) result = {c: {'customer': c, 'billed': 0, 'paid': 0, 'open': 0} for c in ['A', 'B', 'C', 'D']} for inv in latest.values(): if inv['status'] == 'void': continue paid = sum(e['amount'] for e in unique.values() if e['invoice'] == inv['id'] and e['status'] == 'posted' and e['day'] <= 20) row = result[inv['customer']] row['billed'] += inv['amount'] row['paid'] += paid row['open'] += inv['amount'] - paid return list(result.values()) def cases(): out = [] for level, size in enumerate([4, 6, 8, 10], 1): rng = random.Random(90290 + level) jobs = [dict(id=chr(65+i), duration=rng.randint(1, 8), weight=rng.randint(1, 9), group=rng.choice(['X', 'Y']), after=[]) for i in range(size)] for i in range(2, size): if i % 2 == 0: jobs[i]['after'] = [jobs[rng.randrange(i)]['id']] switch = 0 if level == 1 else level prompt = ('One worker executes all jobs without idle time, starting at time 0. Each job must start after all its after jobs finish. ' 'Before a job, add switch minutes when its group differs from the previous job. No setup before the first job. ' 'Minimize sum(weight * completion_time) over ALL jobs. Among tied optima choose the lexicographically smallest order of IDs. ' 'Return JSON {"order":[IDs],"score":integer}. ' + canonical({'switch': switch, 'jobs': jobs})) out.append(dict(id=f'plan-{level}', family='plan', level=level, input=dict(jobs=jobs, switch=switch), prompt=prompt, expected=solve_plan(jobs, switch))) invoices, events = [], [] for i in range([3, 6, 10, 16][level-1]): inv = dict(id=f'I{i:02}', version=1, customer='ABC'[i % 3], amount=1200+137*i, status='active') invoices.append(inv) if level >= 2 and i % 2 == 0: invoices.append(dict(inv, version=2, amount=inv['amount']+231)) if level >= 3 and i % 4 == 1: invoices.append(dict(inv, version=3, status='void')) events.append(dict(id=f'P{i:02}', invoice=inv['id'], amount=500+61*i, status='posted', day=10+i)) if level >= 2: events.append(dict(events[-1])) if level >= 3: events.append(dict(id=f'R{i:02}', invoice=inv['id'], amount=-79-i, status='posted', day=18)) if level == 4: events.append(dict(id=f'Q{i:02}', invoice=inv['id'], amount=400, status='pending', day=12)) rng.shuffle(invoices) rng.shuffle(events) prompt = ('Compute this synthetic ledger, all amounts integer minor units. For each invoice ID keep only the highest version. ' 'Discard void invoices AND all their payment events. Deduplicate events by event ID (duplicates are identical). ' 'Count only posted events with day <= 20; negative amounts are refunds. ' 'For each customer A,B,C,D in that order return billed=sum(active invoice amounts), paid=sum(eligible signed payments), ' 'open=billed-paid (do not clamp). Include customers with no invoices as zeros. ' 'JSON array only, each object has customer,billed,paid,open. ' + canonical(dict(invoices=invoices, events=events))) out.append(dict(id=f'ledger-{level}', family='ledger', level=level, input=dict(invoices=invoices, events=events), prompt=prompt, expected=solve_ledger(invoices, events))) return out def grade(case, response, identity): try: choice = response['choices'][0] if not identity: return 'identity_unverified' if choice['finish_reason'] != 'stop': return 'truncated_or_unfinished' value = json.loads(choice['message']['content']) # Canonical JSON equality distinguishes true from 1 and enforces keys/types. return 'pass' if canonical(value) == canonical(case['expected']) else 'wrong_answer' except (KeyError, IndexError, TypeError, ValueError): return 'invalid_json' def reserve_jpy(payload, price): # UTF-8 bytes upper-bound token count; 512 tokens cover message overhead. in_rate = max(price['sell_per_1m_tokens_in'], price['input_per_1m_tokens'] * 1.1) out_rate = max(price['sell_per_1m_tokens_out'], price['output_per_1m_tokens'] * 1.1) return math.ceil(((len(canonical(payload).encode()) + 512) * in_rate + MAX_TOKENS * out_rate) / 1e6 * FX) def payload(model, case): return dict(model=model, messages=[dict(role='system', content=SYSTEM), dict(role='user', content=case['prompt'])], max_tokens=MAX_TOKENS, temperature=0, stream=False) def export(rows): results = [r for r in rows if r['event'] == 'result'] summary = {} for model in MODELS: own = [r for r in results if r['model'] == model] summary[model] = dict(passed=sum(r['outcome'] == 'pass' for r in own), attempted=len(own), planned=24, estimated_sell_usd=sum(r.get('estimated_sell_usd', 0) for r in own), cost_unknown=sum('estimated_sell_usd' not in r for r in own), median_response_s=statistics.median([r['elapsed_s'] for r in own if r['http'] == 200]) if any(r['http'] == 200 for r in own) else None) return dict(manifest=rows[0], reserved_jpy=sum(r.get('reserve_jpy', 0) for r in rows), unresolved_reservations=sum(r['event'] == 'reserve' for r in rows)-len(results), summary=summary, results=results) def main(): ap = argparse.ArgumentParser(description=__doc__) ap.add_argument('--ledger', type=Path, required=True) ap.add_argument('--run', action='store_true') ap.add_argument('--export', type=Path) ap.add_argument('--prepare', action='store_true') args = ap.parse_args() if args.export: rows = [json.loads(s) for s in args.ledger.read_text().splitlines()] args.export.write_text(json.dumps(export(rows), ensure_ascii=False, indent=2)+'\n') print(canonical(export(rows)['summary'])) return if not (args.run or args.prepare): ap.error('Choose --prepare, --run or --export') suite = cases() config = dict(models=MODELS, cases=suite, system=SYSTEM, max_tokens=MAX_TOKENS, rounds=ROUNDS, temperature=0, cap_jpy=CAP, budget_fx=FX, base=BASE, timeout_s=120) fingerprint = hashlib.sha256(canonical(config).encode()).hexdigest() with args.ledger.open('a+') as out: fcntl.flock(out, fcntl.LOCK_EX | fcntl.LOCK_NB) out.seek(0) rows = [json.loads(s) for s in out.read().splitlines()] def append(row): out.write(canonical(row)+'\n') out.flush() os.fsync(out.fileno()) rows.append(row) if rows: assert rows[0]['fingerprint'] == fingerprint, 'Frozen conditions changed' else: with urllib.request.urlopen(BASE+'/v1/models/pricing', timeout=30) as response: catalog = json.load(response) prices = {m['id']: m for m in catalog['data'] if m['id'] in MODELS} assert set(prices) == set(MODELS) planned_reserve = ROUNDS * sum(reserve_jpy(payload(m, c), prices[m]) for m in MODELS for c in suite) assert planned_reserve <= CAP, f'Planned reservation {planned_reserve} exceeds cap' append(dict(event='manifest', config=config, fingerprint=fingerprint, prices=prices, started_at=dt.datetime.now(dt.timezone.utc).isoformat(), planned_reserve_jpy=planned_reserve, limitations=['Eight synthetic English tasks, three repeated calls per task; repetitions are not independent task samples.', 'Size increases by level, but empirical difficulty need not be monotonic.', 'No tools, temperature 0, 2048 output tokens; truncation is distinct from a wrong answer.', 'API-reported model and route identity, not independent attestation of model weights.', 'Selection labels predate this test; coding, tool use, context limits and diverse perspectives are not evaluated.', 'Latency is full response wall time, not TTFT. Costs are uncached token estimates, not settled bills.', 'No client retries; gateway-side routing/retries may occur. Unknown calls retain their full reservation.'])) print(canonical({k: rows[0][k] for k in ['fingerprint', 'planned_reserve_jpy']}), flush=True) if not args.run: return key = os.environ.get('TEAI_API_KEY') if not key: key = next(s.split('=', 1)[1].strip().strip('\"\'') for s in (Path.home()/'.config/teai/credentials').read_text().splitlines() if s.startswith('TEAI_API_KEY=')) prices = rows[0]['prices'] # Serial calls, rotating model order across cases/rounds. Results always append. for turn in range(ROUNDS): for ci, case in enumerate(suite): shift = (turn+ci) % len(MODELS) for model in MODELS[shift:]+MODELS[:shift]: ident = dict(model=model, round=turn, case=case['id']) if any(r['event'] == 'reserve' and all(r.get(k) == v for k, v in ident.items()) for r in rows): continue data = payload(model, case) reserve = reserve_jpy(data, prices[model]) assert sum(r.get('reserve_jpy', 0) for r in rows)+reserve <= CAP, 'Budget cap' append(dict(event='reserve', **ident, reserve_jpy=reserve)) req = urllib.request.Request(BASE+'/v1/chat/completions', data=canonical(data).encode(), headers={'Authorization': 'Bearer '+key, 'Content-Type': 'application/json'}) start = time.monotonic() status, body, route = 0, {}, '' try: with urllib.request.urlopen(req, timeout=120) as response: status = response.status body = json.load(response) route = response.headers.get('x-teai-route', '') fallback = response.headers.get('x-teai-fallback') or response.headers.get('x-teai-fallback-model') identity = body.get('model') == model and f'served={model};' in route and 'fallback_depth=0' in route and not fallback outcome = grade(case, body, identity) except urllib.error.HTTPError as error: status, identity, outcome = error.code, False, 'http_error' except (urllib.error.URLError, TimeoutError, ValueError, OSError): identity, outcome = False, 'transport_error' choice = (body.get('choices') or [{}])[0] usage = {k: v for k, v in (body.get('usage') or {}).items() if k in ['prompt_tokens', 'completion_tokens', 'total_tokens']} result = dict(event='result', **ident, http=status, outcome=outcome, identity_ok=bool(identity), elapsed_s=round(time.monotonic()-start, 3), response_model=body.get('model'), finish_reason=choice.get('finish_reason'), answer=(choice.get('message') or {}).get('content'), usage=usage) if all(type(usage.get(k)) is int and usage[k] >= 0 for k in ['prompt_tokens', 'completion_tokens']): p = prices[model] result['estimated_sell_usd'] = (usage['prompt_tokens']*p['sell_per_1m_tokens_in']+usage['completion_tokens']*p['sell_per_1m_tokens_out'])/1e6 append(result) print(canonical({k: result[k] for k in ['model', 'round', 'case', 'http', 'outcome', 'elapsed_s']}), flush=True) if status in [401, 402, 403]: raise SystemExit('Authentication/balance blocked; stop without retry') if __name__ == '__main__': main()