🐦 Twitter Post Details

Viewing enriched Twitter post

@WolfBenchAI

GPT-5.5 takes over WolfBench! It’s now the #1 model, ahead of Claude Opus 4.7 and 4.6, GPT-5.4, Sonnet 4.6, Kimi K2.6, Gemini 3.1 Pro, and more. Notable findings after 30 runs (40h runtime, >1.7B tokens, ~$3K cost): - @OpenAI's GPT-5.5 is the best model we ever tested. - @cursor_ai's Agent CLI (CA) is the best agent we ever tested. - @NousResearch's Hermes Agent (HA) outperformed OpenClaw (OC). - With Hermes, going from medium to xhigh reasoning only improved consistency, not capability. Note: This is WolfBench, where we look at more than just the average score, because one metric is not enough. The golden βˆ… score is the actual 5-run average, which most other benchmarks report as their only score. β˜… shows the ceiling (what percentage of the full benchmark this model+agent combination solved at least once across all runs). β–  shows the solid base (what percentage of the full benchmark it solved consistently in every run).

Media 1

πŸ“Š Media Metadata

{
  "media": [
    {
      "url": "https://crmoxkoizveukayfjuyo.supabase.co/storage/v1/object/public/media/posts/2048876633105334375/media_0.jpg",
      "media_url": "https://crmoxkoizveukayfjuyo.supabase.co/storage/v1/object/public/media/posts/2048876633105334375/media_0.jpg",
      "type": "photo",
      "filename": "media_0.jpg"
    }
  ],
  "processed_at": "2026-05-11T22:25:25.394994",
  "pipeline_version": "2.0"
}

πŸ”§ Raw API Response

{
  "type": "tweet",
  "id": "2048876633105334375",
  "url": "https://x.com/WolfBenchAI/status/2048876633105334375",
  "twitterUrl": "https://twitter.com/WolfBenchAI/status/2048876633105334375",
  "text": "GPT-5.5 takes over WolfBench! It’s now the #1 model, ahead of Claude Opus 4.7 and 4.6, GPT-5.4, Sonnet 4.6, Kimi K2.6, Gemini 3.1 Pro, and more.\n\nNotable findings after 30 runs (40h runtime, >1.7B tokens, ~$3K cost):\n- @OpenAI's GPT-5.5 is the best model we ever tested.\n- @cursor_ai's Agent CLI (CA) is the best agent we ever tested.\n- @NousResearch's Hermes Agent (HA) outperformed OpenClaw (OC).\n- With Hermes, going from medium to xhigh reasoning only improved consistency, not capability.\n\nNote: This is WolfBench, where we look at more than just the average score, because one metric is not enough. The golden βˆ… score is the actual 5-run average, which most other benchmarks report as their only score. β˜… shows the ceiling (what percentage of the full benchmark this model+agent combination solved at least once across all runs). β–  shows the solid base (what percentage of the full benchmark it solved consistently in every run).",
  "source": "Twitter for iPhone",
  "retweetCount": 3,
  "replyCount": 2,
  "likeCount": 26,
  "quoteCount": 2,
  "viewCount": 2765,
  "createdAt": "Mon Apr 27 21:27:10 +0000 2026",
  "lang": "en",
  "bookmarkCount": 7,
  "isReply": false,
  "inReplyToId": null,
  "conversationId": "2048876633105334375",
  "displayTextRange": [
    0,
    275
  ],
  "inReplyToUserId": null,
  "inReplyToUsername": null,
  "author": {
    "type": "user",
    "userName": "WolfBenchAI",
    "url": "https://x.com/WolfBenchAI",
    "twitterUrl": "https://twitter.com/WolfBenchAI",
    "id": "2031113398151426049",
    "name": "WolfBench",
    "isVerified": false,
    "isBlueVerified": true,
    "verifiedType": "Business",
    "profilePicture": "https://pbs.twimg.com/profile_images/2036955371127136256/k-RW51zd_normal.jpg",
    "coverPicture": "https://pbs.twimg.com/profile_banners/2031113398151426049/1773932691",
    "description": "",
    "location": "",
    "followers": 269,
    "following": 18,
    "status": "",
    "canDm": false,
    "canMediaTag": true,
    "createdAt": "Mon Mar 09 21:03:49 +0000 2026",
    "entities": {
      "description": {
        "urls": []
      },
      "url": {}
    },
    "fastFollowersCount": 0,
    "favouritesCount": 114,
    "hasCustomTimelines": true,
    "isTranslator": false,
    "mediaCount": 8,
    "statusesCount": 28,
    "withheldInCountries": [],
    "affiliatesHighlightedLabel": {},
    "possiblySensitive": false,
    "pinnedTweetIds": [
      "2048876633105334375"
    ],
    "profile_bio": {
      "description": "https://t.co/IGlGiHLlnE // @WolframRvnwlf's new evaluation framework for models and agents: because one score is not enough! // brought to you by @CoreWeave/@wandb",
      "entities": {
        "description": {
          "urls": [
            {
              "display_url": "wolfbench.ai",
              "expanded_url": "http://wolfbench.ai",
              "indices": [
                0,
                23
              ],
              "url": "https://t.co/IGlGiHLlnE"
            }
          ],
          "user_mentions": [
            {
              "id_str": "",
              "indices": [
                27,
                41
              ],
              "name": "",
              "screen_name": "WolframRvnwlf"
            },
            {
              "id_str": "",
              "indices": [
                146,
                156
              ],
              "name": "",
              "screen_name": "CoreWeave"
            },
            {
              "id_str": "",
              "indices": [
                157,
                163
              ],
              "name": "",
              "screen_name": "wandb"
            }
          ]
        }
      }
    },
    "isAutomated": false,
    "automatedBy": null
  },
  "extendedEntities": {
    "media": [
      {
        "display_url": "pic.twitter.com/wRuVNURD8k",
        "expanded_url": "https://twitter.com/WolfBenchAI/status/2048876633105334375/photo/1",
        "ext_alt_text": "GPT-5.5 takes over WolfBench! It’s now the #1 model, ahead of Claude Opus 4.7 and 4.6, GPT-5.4, Sonnet 4.6, Kimi K2.6, Gemini 3.1 Pro, and more.",
        "ext_media_availability": {
          "status": "Available"
        },
        "features": {
          "large": {
            "faces": []
          },
          "orig": {
            "faces": []
          }
        },
        "id_str": "2048863511850418176",
        "indices": [
          276,
          299
        ],
        "media_key": "3_2048863511850418176",
        "media_results": {
          "id": "QXBpTWVkaWFSZXN1bHRzOgwAAQoAARxvBoI5GoAACgACHG8ScUDXgGcAAA==",
          "result": {
            "__typename": "ApiMedia",
            "id": "QXBpTWVkaWE6DAABCgABHG8GgjkagAAKAAIcbxJxQNeAZwAA",
            "media_key": "3_2048863511850418176"
          }
        },
        "media_url_https": "https://pbs.twimg.com/media/HG8GgjkagAAvFAp.jpg",
        "original_info": {
          "focus_rects": [
            {
              "h": 549,
              "w": 980,
              "x": 0,
              "y": 0
            },
            {
              "h": 980,
              "w": 980,
              "x": 0,
              "y": 0
            },
            {
              "h": 1117,
              "w": 980,
              "x": 0,
              "y": 0
            },
            {
              "h": 1350,
              "w": 675,
              "x": 236,
              "y": 0
            },
            {
              "h": 1350,
              "w": 980,
              "x": 0,
              "y": 0
            }
          ],
          "height": 1350,
          "width": 980
        },
        "sizes": {
          "large": {
            "h": 1350,
            "w": 980
          }
        },
        "type": "photo",
        "url": "https://t.co/wRuVNURD8k"
      }
    ]
  },
  "card": null,
  "place": {},
  "entities": {
    "hashtags": [],
    "symbols": [],
    "urls": [],
    "user_mentions": [
      {
        "id_str": "4398626122",
        "indices": [
          219,
          226
        ],
        "name": "OpenAI",
        "screen_name": "OpenAI"
      },
      {
        "id_str": "1695890961094909952",
        "indices": [
          273,
          283
        ],
        "name": "Cursor",
        "screen_name": "cursor_ai"
      },
      {
        "id_str": "1318419526132862976",
        "indices": [
          337,
          350
        ],
        "name": "Nous Research",
        "screen_name": "NousResearch"
      }
    ]
  },
  "quoted_tweet": null,
  "retweeted_tweet": null,
  "isLimitedReply": false,
  "communityInfo": null,
  "article": null
}