{
  "schemaVersion": "1.0",
  "lastUpdated": "2026-08-18T02:33:58.678Z",
  "source": "Harpd benchmark json-extraction-v1",
  "methodology": "https://harpd.com/rank/methodology/",
  "license": "https://creativecommons.org/licenses/by/4.0/",
  "benchmark": "json-extraction-v1",
  "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
  "taskCount": 100,
  "isModeled": true,
  "methodologyNote": "MODELED ESTIMATE. Prices are public 2026 list prices; success/latency are assumed from published 2025–2026 behavior. Replace with live runs via the open-source runner (node runner/index.mjs) before citing.",
  "pricingSource": "src/lib/llm-pricing.ts (public 2026 list prices)",
  "benchmarkMethodologyUrl": "https://harpd.com/methodology/benchmarks/",
  "githubUrl": "https://github.com/harpd-dev/llm-cost-benchmark",
  "count": 56,
  "records": [
    {
      "id": "json-extraction-v1--gpt-4o-mini--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.88,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4o-mini--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 400,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4o-mini--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 900,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4o-mini--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4o-mini--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4o-mini--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.00096,
      "cost": 0.00096,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4o-mini--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4o-mini",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.001091,
      "cost": 0.001091,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.91,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 900,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 2100,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.00174,
      "cost": 0.00174,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--deepseek-v3--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "deepseek-v3",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.001912,
      "cost": 0.001912,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.9,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 350,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 800,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.0027,
      "cost": 0.0027,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-flash--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-flash",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.003,
      "cost": 0.003,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.84,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 700,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 1700,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.004048,
      "cost": 0.004048,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--llama-3.1-70b--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "llama-3.1-70b",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.004819,
      "cost": 0.004819,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.93,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 600,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 1400,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.0056,
      "cost": 0.0056,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-haiku-4--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-haiku-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.006022,
      "cost": 0.006022,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.95,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 1400,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 3000,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.011,
      "cost": 0.011,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gemini-2.5-pro--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gemini-2.5-pro",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.011579,
      "cost": 0.011579,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.96,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 1500,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 3200,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.0128,
      "cost": 0.0128,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--gpt-4.1--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "gpt-4.1",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.013333,
      "cost": 0.013333,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--success_rate",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "success_rate",
      "score": 0.97,
      "cost": null,
      "unit": "ratio",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--p50_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p50_latency_ms",
      "score": 1200,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--p95_latency_ms",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "p95_latency_ms",
      "score": 2600,
      "cost": null,
      "unit": "ms",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--avg_input_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_input_tokens",
      "score": 4000,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--avg_output_tokens",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "avg_output_tokens",
      "score": 600,
      "cost": null,
      "unit": "tokens",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--cost_per_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_task",
      "score": 0.021,
      "cost": 0.021,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    },
    {
      "id": "json-extraction-v1--claude-sonnet-4--cost_per_successful_task",
      "name": "JSON extraction — Claude vs GPT vs Gemini vs DeepSeek",
      "model": "claude-sonnet-4",
      "task": "Structured JSON extraction from messy natural-language input (en + ja + zh)",
      "metric": "cost_per_successful_task",
      "score": 0.021649,
      "cost": 0.021649,
      "unit": "usd",
      "timestamp": "2026-08-18T02:33:58.678Z",
      "source": "Harpd benchmark json-extraction-v1",
      "lastUpdated": "2026-08-18T02:33:58.678Z"
    }
  ]
}