{
  "protocol_version": 1,
  "article_slug": "runanywhere-wally-coding-agent-cost",
  "reviewed_on": "2026-10-08",
  "evidence_status": "proposed_protocol_not_executed",
  "not_a_vendor_sdk_schema": true,
  "sources": [
    "https://www.ycombinator.com/companies/runanywhere",
    "https://github.com/RunanywhereAI/wally",
    "https://github.com/RunanywhereAI/wally/blob/main/docs/EDITORS.md"
  ],
  "locales": {
    "en": {
      "title": "Wally by RunAnywhere: Cost per Accepted Coding Task",
      "cases": [
        {
          "Path": "Hosted Wally",
          "Workload question": "Does the endpoint improve this model's task economics?",
          "Evidence to collect": "Usage, network path, queue time and task acceptance"
        },
        {
          "Path": "Customer-controlled deployment",
          "Workload question": "Can this workload run within our infrastructure boundary?",
          "Evidence to collect": "Written deployment scope, hardware, support and full operating costs"
        },
        {
          "Path": "Local model",
          "Workload question": "Can a model that fits this device satisfy the task?",
          "Evidence to collect": "Device, memory, backend, model quality and local tool behavior"
        }
      ],
      "steps": [
        "Start each attempt from a clean task fixture and the same initial cache policy.",
        "Define the time limit and acceptance tests before the first run.",
        "Interleave provider runs to reduce time-of-day load bias; repeat tasks.",
        "Save model usage, tool timings, retries, final diff and test outcomes.",
        "Have a reviewer accept the change without knowing the provider where practical."
      ]
    },
    "de": {
      "title": "Wally von RunAnywhere: Kosten pro akzeptierter Coding-Aufgabe",
      "cases": [
        {
          "Weg": "Gehostetes Wally",
          "Prüffrage": "Verbessert der Endpunkt die Aufgabenökonomie dieses Modells?",
          "Benötigte Evidenz": "Verbrauch, Netzwerkweg, Wartezeit und Abnahme"
        },
        {
          "Weg": "Kundenseitiger Betrieb",
          "Prüffrage": "Passt die Aufgabe in unsere Infrastrukturgrenze?",
          "Benötigte Evidenz": "Schriftlicher Umfang, Hardware, Support und Betriebskosten"
        },
        {
          "Weg": "Lokales Modell",
          "Prüffrage": "Erfüllt ein auf dieses Gerät passendes Modell die Aufgabe?",
          "Benötigte Evidenz": "Gerät, Speicher, Backend, Qualität und Werkzeugverhalten"
        }
      ],
      "steps": [
        "Beginne jeden Versuch mit derselben sauberen Aufgabe und Cache-Regel.",
        "Lege Zeitlimit und Abnahmetests vor dem ersten Lauf fest.",
        "Wechsle die Anbieter zwischen Läufen und wiederhole Aufgaben, um Lastschwankungen zu begrenzen.",
        "Sichere Modellverbrauch, Werkzeugzeiten, Wiederholungen, finalen Diff und Testergebnisse.",
        "Lass einen Reviewer möglichst ohne Kenntnis des Anbieters abnehmen."
      ]
    },
    "es": {
      "title": "Wally de RunAnywhere: coste por tarea de código aceptada",
      "cases": [
        {
          "Ruta": "Wally alojado",
          "Pregunta": "¿Mejora la economía de tareas de este modelo?",
          "Evidencia": "Consumo, red, espera y aceptación"
        },
        {
          "Ruta": "Despliegue controlado por el cliente",
          "Pregunta": "¿Puede ejecutarse dentro de nuestra infraestructura?",
          "Evidencia": "Alcance escrito, hardware, soporte y operación"
        },
        {
          "Ruta": "Modelo local",
          "Pregunta": "¿El modelo que cabe en este dispositivo cumple la tarea?",
          "Evidencia": "Dispositivo, memoria, backend, calidad y herramientas"
        }
      ],
      "steps": [
        "Reinicia cada intento desde la misma tarea limpia y política de caché.",
        "Define límite temporal y pruebas antes de empezar.",
        "Alterna proveedores y repite tareas para reducir sesgo por carga horaria.",
        "Guarda consumo, tiempos de herramientas, reintentos, diff final y pruebas.",
        "Usa un revisor que desconozca el proveedor cuando sea posible."
      ]
    },
    "zh": {
      "title": "RunAnywhere Wally：按已验收编程任务计算成本",
      "cases": [
        {
          "路径": "托管 Wally",
          "核心问题": "端点能否改善该模型的任务经济性？",
          "需要记录的证据": "用量、网络路径、排队时间与任务验收"
        },
        {
          "路径": "客户控制部署",
          "核心问题": "任务能否在自身基础设施边界内运行？",
          "需要记录的证据": "书面部署范围、硬件、支持与完整运营费用"
        },
        {
          "路径": "本地模型",
          "核心问题": "设备能容纳的模型是否满足任务？",
          "需要记录的证据": "设备、内存、后端、模型质量与工具行为"
        }
      ],
      "steps": [
        "每次尝试从干净任务夹具和相同初始缓存策略开始。",
        "首次运行前定义时间上限和验收测试。",
        "交错运行供应商并重复任务，减少时段负载偏差。",
        "保存模型用量、工具耗时、重试、最终 diff 和测试结果。",
        "条件允许时，让不知道供应商的审查者验收变更。"
      ]
    }
  },
  "run_configuration": {
    "product_versions": null,
    "model": null,
    "fixture_revision": null,
    "permissions": null,
    "time_budget_seconds": null,
    "cost_budget": null
  },
  "trial_results": [],
  "report_fields": [
    "attempt_id",
    "case_id",
    "observed_final_state",
    "acceptance_passed",
    "independent_evidence",
    "duration_seconds",
    "total_cost",
    "human_review_minutes",
    "unresolved_effects"
  ]
}
