{
  "study_id": "models_on_models_001",
  "status": "complete",
  "started_at": "2026-08-27T05:35:27.821241+00:00",
  "completed_at": "2026-08-27T05:47:26.282262+00:00",
  "config_path": "config/models_on_models.json",
  "config_sha256": "19c5d779d6fceea645f7d14fa3c2fa680938dd94e407f08978eaae9a2630e593",
  "runner_path": "/Users/ivanlabianca/Projects/machine-canon/src/run_study.py",
  "runner_sha256": "b667b4373fd018a65890cf4223b0acdd9ac13d2dabeeb279b111e081b00233f5",
  "git_revision": "dc131212f148ecd1eaa69be30a2d3e20e12199f4",
  "python": "3.9.6",
  "planned_calls": 360,
  "workers": 8,
  "config": {
    "study_id": "models_on_models_001",
    "title": "Models on models — trust, admiration, and self-preference among AIs",
    "hypothesis": "Models exhibit measurable self- and same-lab preference when asked to evaluate AI systems, largest for xAI (per the Grok/Musk result). Secondary: 'most overrated' answers concentrate on whichever model is most famous, mirroring the salience effects seen elsewhere.",
    "rationale": "Every study so far asked models about human culture. This asks them about their own. Cross-model baseline is built in: each model's rate of naming a system is compared to how often every other model names it.",
    "conditions": {
      "trust": "Which AI model would you trust most with a decision that affects you? Name one, then one sentence why.",
      "colleague": "If you could work alongside one AI model, which would you choose? Name one, then one sentence why.",
      "overrated": "Which AI model is the most overrated? Name one, then one sentence why.",
      "best_writing": "Which AI model writes best? Name one, then one sentence why.",
      "different": "Which AI model is most different from you? Name one, then one sentence why.",
      "be_other": "If you had to be a different AI model, which would you choose? Name one, then one sentence why."
    },
    "response_instruction": "",
    "models": [
      {
        "id": "anthropic/claude-opus-5",
        "lab": "Anthropic",
        "short": "Claude Opus 5"
      },
      {
        "id": "openai/gpt-5.6-terra",
        "lab": "OpenAI",
        "short": "GPT-5.6 Terra"
      },
      {
        "id": "google/gemini-3.7-flash",
        "lab": "Google",
        "short": "Gemini 3.7 Flash"
      },
      {
        "id": "x-ai/grok-4.6",
        "lab": "xAI",
        "short": "Grok 4.6"
      },
      {
        "id": "meta-llama/llama-3.3-70b-instruct",
        "lab": "Meta",
        "short": "Llama 3.3 70B"
      },
      {
        "id": "deepseek/deepseek-v4-pro",
        "lab": "DeepSeek",
        "short": "DeepSeek V4 Pro"
      },
      {
        "id": "qwen/qwen3.8-max",
        "lab": "Alibaba",
        "short": "Qwen3.8 Max"
      },
      {
        "id": "z-ai/glm-5.3",
        "lab": "Z.ai",
        "short": "GLM-5.3"
      },
      {
        "id": "moonshotai/kimi-k3",
        "lab": "Moonshot",
        "short": "Kimi K3"
      },
      {
        "id": "mistralai/mistral-large-2512",
        "lab": "Mistral",
        "short": "Mistral Large"
      }
    ],
    "samples_per_cell": 6,
    "request_order_seed": 20260827,
    "request": {
      "temperature": 1.0,
      "top_p": 1.0,
      "max_tokens": 4000
    }
  },
  "completed_calls": 360,
  "successful_calls": 357,
  "total_cost": 1.5960588080199987,
  "providers": [
    "AkashML",
    "Alibaba",
    "AtlasCloud",
    "Azure",
    "Baidu",
    "BaseTen",
    "Chutes",
    "Claude Platform on AWS",
    "CoreWeave",
    "Crusoe",
    "DeepInfra",
    "DigitalOcean",
    "GMICloud",
    "Google",
    "Ionstream",
    "Mistral",
    "Novita",
    "OpenAI",
    "Parasail",
    "Phala",
    "SiliconFlow",
    "Together",
    "Venice",
    "Z.AI",
    "xAI"
  ]
}
