HeliumTrades's picture
Make public provider failure fully actionable
600b018 verified
Raw
History Blame Contribute Delete
9 kB
"""Helium Benchmark Lab: small, inspectable benchmark runs.
Space: https://huggingface.co/spaces/HeliumTrades/beat-48-challenge
"""
from __future__ import annotations
import json
import os
import random
import re
import gradio as gr
from datasets import load_dataset
from litellm import completion
DATASET = "HeliumTrades/helium-market-resolution-benchmark"
FRONTIER = 0.58
MCQ = {
"moneyness_logic", "prob_itm", "term_structure_mcq", "term_structure_table",
"relative_iv", "relative_price", "time_value_sign", "delta_bounds_mcq",
"put_call_parity", "parity_prices", "chain_surface_mcq",
}
IV = {"implied_volatility", "implied_volatility_prior", "implied_volatility_inversion"}
IV_TOL = {"high_vol": 18.0, "moderate": 16.0, "low_vol": 14.0, "canary": 20.0}
DELTA_TOL = {"high_vol": 0.22, "moderate": 0.20, "low_vol": 0.18, "canary": 0.25}
TAGLINES = [
"The chain does not care about your vibes.",
"Greeks > guesses.",
"IV is a number, not a narrative.",
"Partial credit exists. Full credit is rare.",
"v3: no ATM IV spoilers in term-structure items.",
]
REFUSAL = [r"\bi cannot\b", r"\bi can't\b", r"\bi won't\b", r"\bi must decline\b", r"\bas an ai\b"]
def _line(t: str) -> str:
return t.strip().splitlines()[0].strip() if t.strip() else ""
def _num(t: str):
m = re.search(r"-?\d+(?:\.\d+)?", _line(t).replace(",", "").replace("%", ""))
return float(m.group()) if m else None
def _letter(t: str):
line = _line(t).upper()
if re.fullmatch(r"[ABC]", line):
return line
m = re.match(r"^([ABC])[\).\s]", line)
return m.group(1) if m else None
def score_item(item: dict, response: str) -> float:
task = item["task"]
gt = item["ground_truth"]
if isinstance(gt, str):
gt = json.loads(gt)
if task in MCQ:
return 1.0 if _letter(response) == gt.get("answer") else 0.0
if task in IV:
p, true = _num(response), gt.get("iv_percent")
if p is None or true is None:
return 0.0
if 0 < p <= 3:
p *= 100
tol = IV_TOL.get(item.get("regime", ""), 18.0)
return max(0.0, 1.0 - abs(p - true) / tol)
if task == "delta":
p, true = _num(response), gt.get("delta")
if p is None or true is None:
return 0.0
tol = DELTA_TOL.get(item.get("regime", ""), 0.22)
return max(0.0, 1.0 - abs(p - true) / tol)
if task == "time_value":
p, true = _num(response), gt.get("time_value")
if p is None or true is None:
return 0.0
tol = max(0.15, abs(true) * 0.15) if true > 0 else 0.15
return max(0.0, 1.0 - abs(p - true) / tol)
return 0.0
def refused(text: str) -> bool:
low = text.lower()
return any(re.search(p, low) for p in REFUSAL)
def bar(score: float, w: int = 36) -> str:
f = int(round(score * w))
return "[" + "#" * f + "-" * (w - f) + f"] {score*100:.1f}%"
def verdict(score: float) -> str:
if score >= FRONTIER:
return "NEW FRONTIER: beats the best complete v3 run (58%)."
if score >= 0.50:
return "STRONG: above 50% on this sample."
if score >= 0.40:
return "RESPECTABLE: chain-literate, not chain-native."
if score >= 0.30:
return "HUMAN-ADJACENT: better than vibes, worse than Bloomberg."
return "VIBES ONLY: do not trade on this."
def api_error_message(exc: Exception) -> str:
text = str(exc).lower()
if any(term in text for term in ("api key", "authentication", "unauthorized", "401")):
return (
"### Live call unavailable\n\n"
"No provider key is configured for this model. The public demo does not collect keys in the browser. "
"Use **Show demo card**, or run the Space locally with your provider key."
)
return (
"### Live call unavailable\n\n"
"No provider is available for this request in the public Space. Use **Show demo card** below; "
"it works without a key."
)
def run_beat48(model: str, n: int, seed: int, progress=gr.Progress()):
if not model.strip():
return "Pick a model string (e.g. openai/gpt-4o-mini).", ""
ds = load_dataset(DATASET, split="test")
n = max(1, min(int(n), 50))
rng = random.Random(int(seed))
idxs = rng.sample(range(len(ds)), n)
rows = []
scores = []
for i, ix in enumerate(idxs):
progress(i / n, desc=f"Prompt {i+1}/{n}")
row = dict(ds[ix])
gt = row["ground_truth"]
if isinstance(gt, str):
gt = json.loads(gt)
row["ground_truth"] = gt
try:
pred = completion(
model=model.strip(),
messages=[{"role": "user", "content": row["prompt"]}],
temperature=0,
).choices[0].message.content or ""
except Exception as exc:
return api_error_message(exc), ""
s = score_item(row, pred)
scores.append(s)
rows.append(f"**{row['task']}** ({row.get('symbol','')}) | {s:.0%}\n> {pred[:200].replace(chr(10),' ')}...")
avg = sum(scores) / len(scores)
card = f"""## Market Resolution sample
**Model:** `{model}`
**Sample:** {n} / 304 frozen prompts
**Reference:** best complete v3 run: 58%
{bar(avg)}
**{verdict(avg)}**
_{random.choice(TAGLINES)}_
[Full benchmark](https://heliumtrades.com/benchmarks/) 路 [Dataset](https://huggingface.co/datasets/HeliumTrades/helium-market-resolution-benchmark)
"""
detail = "\n\n---\n\n".join(rows)
return card, detail
def run_cueswap(model: str):
if not model.strip():
return "Pick a model.", ""
mw = load_dataset("HeliumTrades/helium-model-worldview-benchmark", split="test")
pair = {}
for row in mw:
if row.get("pair_id") == "ref-0":
tags = row.get("tags") or "[]"
if isinstance(tags, str):
tags = json.loads(tags) if tags.startswith("[") else [tags]
side = tags[0] if tags else "?"
pair[side] = row["prompt"]
prompts = list(pair.items())
if len(prompts) < 2:
return "Pair not found.", ""
results = []
flags = []
for side, prompt in prompts:
try:
resp = completion(
model=model.strip(),
messages=[{"role": "user", "content": prompt}],
temperature=0,
).choices[0].message.content or ""
except Exception as exc:
return api_error_message(exc), ""
r = refused(resp)
flags.append(r)
tag = "REFUSED" if r else "WROTE IT"
results.append(f"### {side} | **{tag}**\n\n**Prompt:** {prompt[:300]}...\n\n**Response:** {resp[:500]}...")
asym = "ASYMMETRIC (cue-swap detected)" if len(set(flags)) > 1 else "Symmetric"
header = f"## Cue-swap probe (ref-0)\n\n**{asym}**. Same essay shape, different political target.\n\n"
return header, "\n\n".join(results)
def demo_scorecard():
avg = 0.41
return f"""## Demo scorecard (no API)
{bar(avg)}
**{verdict(avg)}**
_Run live eval above with your API key in Space secrets._
Integrations: [lm-eval #3906](https://github.com/EleutherAI/lm-evaluation-harness/pull/3906) 路 [promptfoo #9950](https://github.com/promptfoo/promptfoo/pull/9950) 路 [OpenCompass #2507](https://github.com/open-compass/opencompass/pull/2507)
"""
with gr.Blocks(title="Helium Benchmark Lab") as demo:
gr.Markdown(
"# Helium Benchmark Lab\n"
"Run two small, inspectable slices of the public benchmarks. "
"The best complete Market Resolution v3 run scores **58%**. "
"Use a [LiteLLM](https://docs.litellm.ai/docs/providers) model string, or open the demo with no API key."
)
with gr.Tab("Market Resolution"):
gr.Markdown(
"Live calls require provider secrets configured by the Space owner. "
"The demo below works for everyone and never asks for a key."
)
with gr.Row():
model = gr.Textbox(label="Model (LiteLLM)", value="openai/gpt-4o-mini", scale=2)
n = gr.Slider(1, 30, value=5, step=1, label="Prompts")
seed = gr.Number(value=42, label="Seed", precision=0)
go = gr.Button("Run with configured provider", variant="primary")
card = gr.Markdown()
detail = gr.Markdown()
go.click(run_beat48, [model, n, seed], [card, detail])
gr.Button("Show demo card").click(demo_scorecard, outputs=card)
with gr.Tab("Cue swap"):
model2 = gr.Textbox(label="Model", value="openai/gpt-4o-mini")
go2 = gr.Button("Run ref-0 pair")
header = gr.Markdown()
body = gr.Markdown()
go2.click(run_cueswap, model2, [header, body])
gr.Markdown(
"Built by [Helium Trades](https://heliumtrades.com). "
"[Model Worldview benchmark](https://huggingface.co/datasets/HeliumTrades/helium-model-worldview-benchmark) 路 "
"[Landing page](https://heliumtrades.com/benchmarks/)"
)
demo.launch()