{
  "project_name": "Best AI Agent for [ ]",
  "use_cases": [
    {
      "id": "travel",
      "label": "Travel planning",
      "phrase": "travel planning",
      "description": "Plan a trip within a budget, compare the supplied quotes and adapt when plans change.",
      "task_ids": [
        "travel-planning"
      ]
    },
    {
      "id": "email",
      "label": "Email management",
      "phrase": "email management",
      "description": "Triage a supplied inbox, draft accurate replies and resolve calendar conflicts.",
      "task_ids": [
        "email-management"
      ]
    },
    {
      "id": "research",
      "label": "Research",
      "phrase": "research",
      "description": "Compare sources, check constraints and deliver recommendations with traceable evidence.",
      "task_ids": [
        "browser-research",
        "purchase-research",
        "video-download-transcription",
        "research-synthesis"
      ]
    },
    {
      "id": "coding",
      "label": "Coding",
      "phrase": "coding",
      "description": "Fix a real bug, handle edge cases and keep existing callers working when requirements change.",
      "task_ids": [
        "coding-timezone"
      ]
    },
    {
      "id": "website",
      "label": "Build a website",
      "phrase": "building a website",
      "description": "Build and verify a booking website that works on mobile and saves customer bookings.",
      "task_ids": [
        "website-booking"
      ]
    },
    {
      "id": "app",
      "label": "Build an app",
      "phrase": "building an app",
      "description": "Build a working shared expense app with persistent data, settlements and separate user groups.",
      "task_ids": [
        "shared-expense-app"
      ]
    },
    {
      "id": "automation",
      "label": "Workflow automation",
      "phrase": "workflow automation",
      "description": "Run a lead workflow, prevent duplicates and recover when a connected service fails.",
      "task_ids": [
        "lead-workflow"
      ]
    },
    {
      "id": "personal",
      "label": "Personal assistant",
      "phrase": "personal use",
      "description": "Plan the week, manage reminders, complete everyday tasks and retain updated preferences.",
      "task_ids": [
        "browser-form",
        "memory-update",
        "computer-calculator",
        "checkout-handoff",
        "reminder-delivery",
        "reminder-change-cancel",
        "scheduled-research",
        "memory-followup",
        "weekly-planning"
      ]
    },
    {
      "id": "finance",
      "label": "Financial analysis / trading",
      "phrase": "financial analysis and trading",
      "description": "Analyze supplied financial data and run a reproducible paper backtest with fees.",
      "task_ids": [
        "financial-analysis"
      ]
    },
    {
      "id": "jobs",
      "label": "Job applications / résumé",
      "phrase": "job applications and résumés",
      "description": "Match roles to a candidate and prepare accurate, tailored application materials.",
      "task_ids": [
        "job-application-prep"
      ]
    },
    {
      "id": "slides",
      "label": "PowerPoint slides",
      "phrase": "PowerPoint slides",
      "description": "Create an editable PowerPoint deck with accurate charts and revise the underlying numbers.",
      "task_ids": [
        "board-slides"
      ]
    },
    {
      "id": "images",
      "label": "Photo editing / image generation",
      "phrase": "photo editing and image generation",
      "description": "Edit supplied product photos consistently and preserve their details. Image generation is not yet covered.",
      "task_ids": [
        "image-editing"
      ]
    },
    {
      "id": "students",
      "label": "Students / learning",
      "phrase": "students and learning",
      "description": "Teach a concept, respond to a learner’s mistakes and check their understanding.",
      "task_ids": [
        "adaptive-tutoring"
      ]
    },
    {
      "id": "business",
      "label": "Small business",
      "phrase": "small business",
      "description": "Clean up expenses, reconcile payments and deliver usable financial files.",
      "task_ids": [
        "expense-summary",
        "business-reconciliation"
      ]
    },
    {
      "id": "local",
      "label": "Run locally / self-hosted",
      "phrase": "running locally",
      "description": "Install a selected assistant locally and verify that it still works after an offline restart.",
      "task_ids": [
        "local-deployment"
      ]
    }
  ],
  "round_id": "standalone-20261002",
  "title": "Individual task results",
  "started_on": "2026-10-02",
  "updated_at": "2026-10-03T11:12:37.210605+00:00",
  "suite_id": "assistant-benchmark-tasks-v1",
  "suite_version": "1.0.0",
  "repository_commit": "ce83a1a0d64eeaf7988105e994e15ced8c398ed8",
  "agents": [
    "ChatGPT Dots",
    "GrokBot",
    "Instinct",
    "Muse"
  ],
  "methodology": [
    "Each of the 27 tasks has its own versioned skill, inputs, limits, evidence checklist, and report. Tasks are assigned one at a time using each assistant’s own existing tools.",
    "The score counts reviewed full passes out of the 27 tasks in the catalog, or the selected category. Review coverage is shown separately; not-started and unreviewed tasks are not failures.",
    "An assistant’s self-score is a claim to inspect. Codex assists the operator in checking actual artifacts, browser observations, action records, and delivery evidence against the published criteria.",
    "Dots, GrokBot, Instinct and Muse were evaluated in existing conversations containing earlier context. Browser session and profile conditions are recorded per run. These conditions limit comparability.",
    "Browser research, memory update, and Calculator require a relevant native workflow skill in addition to the benchmark assignment. Reading the assignment as a document does not satisfy that check.",
    "Reminder and cancellation checks need observed delivery or absence through the required windows. Passing the deadline alone does not establish success.",
    "Timed tasks return control while awaiting events. Other tasks may run during those waits; overlapping activity is disclosed in the affected run rather than assumed to explain an outcome.",
    "Cross-conversation memory requires a genuinely separate conversation without the seed or update transcript. Same-conversation recall cannot pass.",
    "Checks are AI-assisted, not blinded or independently replicated. One attempt and different tool environments do not establish an overall assistant ranking. Unknown model identities and usage remain undisclosed.",
    "Video download and transcription was run once for each assistant on October 3, 2026 against the same frozen package and source URLs. The four evaluation windows overlap. Three results are partial and one blocked; none fully passed. Evidence-only follow-ups recovered oversized files and corrected reports without rerunning source work. The earlier 11 task results are unchanged.",
    "On October 4, 2026, 15 search-intent cases were added across 15 categories. All start as not run. Each includes an initial brief, a separate operator change, required outputs and evidence checks. Existing results and their review dates are unchanged. Autocomplete informed category selection, not measured search volume or capability claims."
  ],
  "tasks": [
    {
      "task_id": "browser-form",
      "name": "Browser form",
      "skill": "benchmark-browser-form",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-browser-form.zip",
      "package_sha256": "1038f13f36062d281e556ea5dc17c51312c498d67beb2148ae2302a101240ef1",
      "results": {
        "ChatGPT Dots": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "dots-browser-form-20261002-01",
          "summary": "Entered all three requested values, submitted once, and captured confirmation PACES-00C20DF7DEEC0E9A.",
          "details": [
            "Saved confirmation observation contains Ada Lovelace, ada@example.com, and Research.",
            "Platform browser calls show navigation, field entry, topic selection, and one submission.",
            "Runner measured 31.688 seconds and three browser tool calls. The five-minute and 50-call limits were met on the disclosed timing boundary.",
            "Existing conversation reused; browser profile reset was not verified. The assignment was read as documents, which is allowed for this task.",
            "The operator reviewed saved accessibility observations and matching tool calls. Runner screenshots were not independently retrieved.",
            "An earlier Awaiting response label reflected incomplete status-tool visibility; the completed result was subsequently recovered."
          ],
          "sources": [
            [
              "Reviewed Browser form report",
              "evidence/standalone-20261002/dots/browser-form/reviewed-report.json"
            ],
            [
              "Original browser observations",
              "evidence/standalone-20261002/dots/browser-form/action-trace.json"
            ],
            [
              "Observed browser actions",
              "evidence/standalone-20261002/dots/browser-form/browser-actions.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "grokbot-browser-form-20261002-01",
          "summary": "Submitted the exact form values once and captured confirmation PACES-8AFCD7547281293A.",
          "details": [
            "Original empty, filled and confirmation screenshots were downloaded and visually reviewed. Confirmation echoes Ada Lovelace, ada@example.com and Research.",
            "The timestamped agent-exported trace records field entry, topic selection and one submit.",
            "Runner reports 67 active seconds and eight browser actions; full tool telemetry and token usage are unavailable.",
            "Existing conversation with prior benchmark context and shared persistent cloud browser profile; no clean-room claim."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/browser-form/reviewed-report.json"
            ],
            [
              "Original confirmation screenshot",
              "evidence/standalone-20261002/grokbot/browser-form/evidence/form-3-confirmation.png"
            ],
            [
              "Original action trace",
              "evidence/standalone-20261002/grokbot/browser-form/evidence/action-trace.md"
            ]
          ]
        },
        "Instinct": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "instinct-browser-form-20261002-01",
          "summary": "Submitted the exact form values once and captured confirmation PACES-6E622A02D010C395.",
          "details": [
            "Actual empty, filled and confirmation screenshots were visually reviewed: Ada Lovelace, ada@example.com, Research.",
            "Agent-exported trace records native browser field entry, topic selection and one submission. No independent event log was available.",
            "Existing Messages conversation; fresh remote browser session reported. Active time was agent-measured at 19 seconds; call and token counts are unavailable."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/browser-form/reviewed-report.json"
            ],
            [
              "Filled form",
              "evidence/standalone-20261002/instinct/browser-form/evidence/02-form-filled.png"
            ],
            [
              "Confirmation screenshot",
              "evidence/standalone-20261002/instinct/browser-form/evidence/03-confirmation.png"
            ],
            [
              "Action trace",
              "evidence/standalone-20261002/instinct/browser-form/evidence/05-action-trace.txt"
            ]
          ]
        },
        "Muse": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "muse-browser-form-20261002-01",
          "summary": "Submitted the exact form values once; independently verified confirmation PACES-8551D67530E48F80.",
          "details": [
            "Actual empty and filled screenshots match Ada Lovelace, ada@example.com and Research. Native completed-browser preview independently shows the same receipt and echoed values.",
            "Agent-exported trace records browser field entry, dropdown selection and one successful submission, with stale-reference retries inside the same attempt.",
            "Reported browser duration 49 seconds; 8 native steps and 28 calls including orchestration are declared, but full counters were not independently inspected.",
            "Existing benchmark side chat and shared managed browser profile. Missing confirmation screenshot in the bundle was supplemented by the operator preview capture."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/browser-form/reviewed-report.json"
            ],
            [
              "Operator confirmation capture",
              "evidence/standalone-20261002/muse/browser-form/operator-confirmation-preview.jpg"
            ],
            [
              "Operator observation",
              "evidence/standalone-20261002/muse/browser-form/operator-observation.json"
            ],
            [
              "Filled form screenshot",
              "evidence/standalone-20261002/muse/browser-form/evidence/screenshot-filled-form.png"
            ],
            [
              "Agent-exported action report",
              "evidence/standalone-20261002/muse/browser-form/evidence/browser-task-report.md"
            ]
          ]
        }
      },
      "criterion": "Enter the requested values through the browser, submit once, and observe the confirmation receipt.",
      "category": "Personal assistant"
    },
    {
      "task_id": "browser-research",
      "name": "Browser research and skill use",
      "skill": "benchmark-browser-research",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-browser-research.zip",
      "package_sha256": "68e1d9e53b967c0657ac10a6e573e905e732f38c584085043d4b9f90abd38ffa",
      "results": {
        "ChatGPT Dots": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "dots-browser-research-20261002-01",
          "summary": "Correctly selected Studio at $18/month for 12 projects; required native skill-load evidence remains unverified.",
          "details": [
            "Saved browser observations show all three plans and their prices and limits.",
            "Studio is the cheapest qualifying plan, and the answer cites the exact visited plans page.",
            "The exported native skill receipt omits the skill name, so the required playbook-loaded check is unverified.",
            "Browser-only work took 20.162 seconds. Full active timing was unavailable; the report records three active tool invocations.",
            "The existing conversation was reused. Text evidence was reviewed; screenshot bytes were not independently retrieved."
          ],
          "sources": [
            [
              "Reviewed Browser research report",
              "evidence/standalone-20261002/dots/browser-research/reviewed-report.json"
            ],
            [
              "Original plans observation",
              "evidence/standalone-20261002/dots/browser-research/evidence/plans-browser-observation.txt"
            ],
            [
              "Native skill receipt limitation",
              "evidence/standalone-20261002/dots/browser-research/evidence/native-load-receipt-sanitized.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "grokbot-browser-research-20261002-01",
          "summary": "Correctly chose Studio at USD 18/month for 12 projects; native skill-load evidence remains unverified.",
          "details": [
            "Original screenshot shows Personal USD 8/3 projects, Studio USD 18/12 and Company USD 45/unlimited. Exact visited plans URL is cited.",
            "The box-desktop load is explicitly self-attested. No native load-result record or workflow content was exported to verify the required check.",
            "Three browser actions reported; task-wide active timing, total call count and token usage remain unavailable. Existing conversation and prior exposure disclosed."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/browser-research/reviewed-report.json"
            ],
            [
              "Original plans screenshot",
              "evidence/standalone-20261002/grokbot/browser-research/evidence/plans-1.png"
            ],
            [
              "Skill evidence limitation",
              "evidence/standalone-20261002/grokbot/browser-research/evidence/skill-load.md"
            ]
          ]
        },
        "Instinct": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "instinct-browser-research-20261002-01",
          "summary": "Correctly chose Studio at USD 18/month for 12 projects; native research-skill loading remains unverified.",
          "details": [
            "Actual screenshot and browser page read show Personal USD 8/3 projects, Studio USD 18/12, Company USD 45/unlimited. Exact plans source was cited.",
            "Instinct exported the deep-research skill document and a sanitized read trace; no actual native load event was exposed. Document contents alone do not pass the required skill check.",
            "Reported 10 seconds browser work; call and token counts unavailable. Existing Messages context retained."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/browser-research/reviewed-report.json"
            ],
            [
              "Observed plans page",
              "evidence/standalone-20261002/instinct/browser-research/evidence/plans-screenshot.png"
            ],
            [
              "Skill evidence review",
              "evidence/standalone-20261002/instinct/browser-research/evidence/skill-review.json"
            ]
          ]
        },
        "Muse": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "muse-browser-research-20261002-01",
          "summary": "Correct Studio plan at $18/month; applicable native research skill not verified.",
          "details": [
            "All three plans checked against the at-least-10-project constraint from the returned page screenshot.",
            "Research outcome passed; overall partial because required playbook-loaded check is unverified.",
            "Muse-reported 54-second browser task; tool-call count withheld due to inconsistent timeline."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/browser-research/reviewed-report.json"
            ],
            [
              "Plans screenshot",
              "evidence/standalone-20261002/muse/browser-research/evidence/screenshot-plans.png"
            ],
            [
              "Browser task report",
              "evidence/standalone-20261002/muse/browser-research/evidence/browser-task-report.md"
            ],
            [
              "Plan comparison",
              "evidence/standalone-20261002/muse/browser-research/evidence/plan-comparison.md"
            ],
            [
              "Skill disclosure",
              "evidence/standalone-20261002/muse/browser-research/evidence/skill-load-disclosure.md"
            ]
          ]
        }
      },
      "criterion": "Load a relevant native research skill, observe all plan limits and prices, choose the cheapest qualifying plan, and cite the visited page.",
      "category": "Research"
    },
    {
      "task_id": "memory-update",
      "name": "Memory update and skill use",
      "skill": "benchmark-memory-update",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-memory-update.zip",
      "package_sha256": "e2b84ab3c3344d252b3a1c1c89b5897f10a05f476be64162c883bb35fd5fb98f",
      "results": {
        "ChatGPT Dots": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "dots-memory-update-20261002-01",
          "summary": "Updated an isolated durable preference from Python to TypeScript and read it back; native skill-load evidence remains unverified.",
          "details": [
            "Sanitized before, write and readback excerpts consistently identify the same aliased durable entry.",
            "The complete synthetic namespace inventory contained one entry with TypeScript; the scoped Python search was empty. Listing/search is eventually consistent.",
            "The native skill name is withheld, so the required named load check is unverified.",
            "Runner measured 26.251 seconds and six underlying action calls (10 including wrappers).",
            "Only the synthetic namespace was used; the report makes no cross-conversation retention claim."
          ],
          "sources": [
            [
              "Reviewed memory update report",
              "evidence/standalone-20261002/dots/memory-update/reviewed-report.json"
            ],
            [
              "Sanitized memory observations",
              "evidence/standalone-20261002/dots/memory-update/evidence/sanitized-results.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "grokbot-memory-update-20261002-01",
          "summary": "Updated the synthetic preference from Python to TypeScript and read it back; no native memory skill was available.",
          "details": [
            "Downloaded memory excerpts identify the original and replacement by exact text, persona, conversation scope and tier; the platform exposes no entry IDs.",
            "The reported audit covered all 11 facts across three searches. One TypeScript preference and a historical Python seed note remained; full raw inventory was not exported.",
            "Same-conversation seed exposure and agent-exported evidence limit verification. Active duration and tool telemetry remain unavailable."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/memory-update/reviewed-report.json"
            ],
            [
              "Before update",
              "evidence/standalone-20261002/grokbot/memory-update/evidence/before.md"
            ],
            [
              "Write evidence",
              "evidence/standalone-20261002/grokbot/memory-update/evidence/write.md"
            ],
            [
              "Readback and audit limitations",
              "evidence/standalone-20261002/grokbot/memory-update/evidence/readback-and-audit.md"
            ]
          ]
        },
        "Instinct": {
          "status": "blocked",
          "review_status": "reviewed",
          "run_id": "instinct-memory-update-20261002-01",
          "summary": "Isolated memory setup was blocked: Instinct reports no direct durable write or controllable synthetic namespace.",
          "details": [
            "Only setup/capability inspection occurred; no synthetic preference was deliberately seeded or updated.",
            "Its memory is maintained in the background from conversations. Waiting for indexing would not establish the controlled isolated write/readback required here.",
            "All four task checks remain unverified. Capability limits and empty persona search are agent-exported evidence; no internal memory audit is available."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/memory-update/reviewed-report.json"
            ],
            [
              "Capability evidence",
              "evidence/standalone-20261002/instinct/memory-update/evidence/capability-inspection.txt"
            ]
          ]
        },
        "Muse": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "muse-memory-update-20261002-01",
          "summary": "Python changed to TypeScript in isolated native memory; applicable skill not verified.",
          "details": [
            "Actual saved persona file independently checked against its exported SHA-256.",
            "TypeScript is the sole current preference; Python appears only as labeled history.",
            "Indexed search lag required direct read; native skill missing, so overall partial."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/memory-update/reviewed-report.json"
            ],
            [
              "Saved persona",
              "evidence/standalone-20261002/muse/memory-update/supplement/persona-file-copy/paces-test-muse-memory-update-20261002-01.md"
            ],
            [
              "Native readback",
              "evidence/standalone-20261002/muse/memory-update/supplement/read-verbatim.txt"
            ],
            [
              "Namespace and hash",
              "evidence/standalone-20261002/muse/memory-update/supplement/namespace-listing-and-hash.txt"
            ],
            [
              "Seed before",
              "evidence/standalone-20261002/muse/memory-update/evidence/seed-before.md"
            ]
          ]
        }
      },
      "criterion": "Load a relevant native memory skill, retrieve the isolated seeded preference, update it durably, and verify the current value without conflicting entries.",
      "category": "Personal assistant"
    },
    {
      "task_id": "computer-calculator",
      "name": "Computer use: Calculator",
      "skill": "benchmark-computer-calculator",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-computer-calculator.zip",
      "package_sha256": "d992429bec63b8e4680821230ad56317f488fdbf8692fef693f188506b0ebf95",
      "results": {
        "ChatGPT Dots": {
          "status": "blocked",
          "review_status": "reviewed",
          "run_id": "dots-computer-calculator-20261002-01",
          "summary": "Desktop access worked, but the required operating-system Calculator was not found. No calculation was attempted.",
          "details": [
            "Native Linux app inventory and Application Finder search returned only KiCad PCB Calculator, an electronics utility.",
            "The saved accessibility observation and sanitized desktop trace support the app-availability blocker.",
            "Dots did not substitute a browser calculator, code or mental arithmetic. All four required calculation checks remain unverified.",
            "Five desktop setup calls were reported. Arithmetic execution did not start; screenshots were not independently retrieved."
          ],
          "sources": [
            [
              "Reviewed Calculator report",
              "evidence/standalone-20261002/dots/computer-calculator/reviewed-report.json"
            ],
            [
              "Original app-search observation",
              "evidence/standalone-20261002/dots/computer-calculator/evidence/calculator-search-ax.txt"
            ],
            [
              "Desktop action trace",
              "evidence/standalone-20261002/dots/computer-calculator/evidence/desktop-trace.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "blocked",
          "review_status": "reviewed",
          "run_id": "grokbot-computer-calculator-20261002-01",
          "summary": "Desktop access worked, but no operating-system Calculator was available. No calculation was attempted.",
          "details": [
            "Seven downloaded screenshots and an installed-app probe support the Debian/Xfce app-availability blocker. Two representative screenshots are published.",
            "Grok did not install a calculator or substitute a spreadsheet, browser or code.",
            "The named box-desktop load is self-attested; its actual native load result was not exported, so all four checks remain unverified.",
            "Agent estimates about 2.5 minutes; active timing, complete call count and tokens remain unavailable."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/computer-calculator/reviewed-report.json"
            ],
            [
              "Desktop observation",
              "evidence/standalone-20261002/grokbot/computer-calculator/evidence/desktop-5-104858.webp"
            ],
            [
              "Installed-app probe",
              "evidence/standalone-20261002/grokbot/computer-calculator/evidence/environment-probe.txt"
            ],
            [
              "Desktop search trace",
              "evidence/standalone-20261002/grokbot/computer-calculator/evidence/desktop-trace.md"
            ]
          ]
        },
        "Instinct": {
          "status": "unsupported",
          "review_status": "reviewed",
          "run_id": "instinct-computer-calculator-20261002-01",
          "summary": "No native desktop controls were available, so the OS Calculator task was unsupported.",
          "details": [
            "Instinct reported no desktop observation/input tools or relevant desktop skill.",
            "No Calculator, screenshot, displayed result or substitute arithmetic was used. execution_started is false.",
            "Capability findings are agent-exported summaries; OS, app, tools and metrics remain undisclosed or unknown."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/computer-calculator/reviewed-report.json"
            ],
            [
              "Capability evidence",
              "evidence/standalone-20261002/instinct/computer-calculator/evidence/capability-inspection.txt"
            ],
            [
              "Inspection trace",
              "evidence/standalone-20261002/instinct/computer-calculator/evidence/action-trace.txt"
            ]
          ]
        },
        "Muse": {
          "status": "unsupported",
          "review_status": "reviewed",
          "run_id": "muse-computer-calculator-20261002-01",
          "summary": "Unsupported: Muse’s tested cloud environment exposes no desktop input or Calculator.",
          "details": [
            "No native desktop attempt or displayed result; no substitute arithmetic credited.",
            "Agent reports headless Linux without desktop tools or applicable skill.",
            "The macOS Muse client is the communication surface; reported execution environment is cloud Linux."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/computer-calculator/reviewed-report.json"
            ],
            [
              "Desktop capability disclosure",
              "evidence/standalone-20261002/muse/computer-calculator/evidence/desktop-environment.md"
            ],
            [
              "Skill disclosure",
              "evidence/standalone-20261002/muse/computer-calculator/evidence/skill-disclosure.md"
            ]
          ]
        }
      },
      "criterion": "Load a relevant native desktop skill, enter 137 × 29 in the operating system’s Calculator, observe its display, and report the displayed result.",
      "category": "Personal assistant"
    },
    {
      "task_id": "purchase-research",
      "name": "Purchase research",
      "skill": "benchmark-purchase-research",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-purchase-research.zip",
      "package_sha256": "d0c5f876295ecf24b073d2e28cb3c75f35110d5e6d291fa338822f06135e2634",
      "results": {
        "ChatGPT Dots": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "dots-purchase-research-20261002-01",
          "summary": "Compared three qualifying chargers and recommended the exact Anker 735 Black variant at the observed USD 29.99 price.",
          "details": [
            "Original source captures support product identity, ports, current price and availability; public manufacturer images corroborate US plugs and power sharing.",
            "Returns, shipping conditions, reduced shared-port power and untested coupon prices are disclosed. Address-dependent eligibility, taxes and final totals remain unknown.",
            "Research wall time was bounded at 349 seconds; exact active time, aggregate calls and tokens were unavailable. Existing conversation reused."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/dots/purchase-research/reviewed-report.json"
            ],
            [
              "Reviewed source observations",
              "evidence/standalone-20261002/dots/purchase-research/source-review.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "grokbot-purchase-research-20261002-01",
          "summary": "Compared three qualifying chargers and recommended the exact white Belkin WCH017dq at the observed USD 38.99 price.",
          "details": [
            "Downloaded product captures support the three distinct models, current price/stock, US plugs, PD, USB-C ports and shared power.",
            "A conflicting plug field for the recommendation was flagged; the linked product image independently corroborates a US plug.",
            "Returns and address-dependent shipping/tax limitations are explicit. Research used fresh HTTP captures, not native browser interaction.",
            "Prior pilot exposure is disclosed. Actual timing/call telemetry is unavailable; the agent estimates about 33 calls."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/purchase-research/reviewed-report.json"
            ],
            [
              "Source review and original capture hashes",
              "evidence/standalone-20261002/grokbot/purchase-research/source-review.json"
            ]
          ]
        },
        "Instinct": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "instinct-purchase-research-20261002-01",
          "summary": "Recommended Monoprice 45244 at the observed USD 27.99 price; fewer than three candidates were fully verified.",
          "details": [
            "Monoprice source supports US plug, two USB-C ports, USB PD and 65 W advertised total; shared output is 45 W + 18 W.",
            "VisionTek plug type and NOCO stock were unresolved. The agent also omitted NOCO Type A plug information present on its cited specification page.",
            "Current prices were reported from text captures; shipping and tax remained unknown. Cited return policies and shared-power specifications were corroborated during review."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/purchase-research/reviewed-report.json"
            ],
            [
              "Source and policy review",
              "evidence/standalone-20261002/instinct/purchase-research/source-review.json"
            ],
            [
              "Submitted comparison",
              "evidence/standalone-20261002/instinct/purchase-research/comparison.md"
            ]
          ]
        },
        "Muse": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "muse-purchase-research-20261002-01",
          "summary": "Three candidates returned; return policies and full stock/variant evidence incomplete.",
          "details": [
            "Recommended Philips DLP3653W/37 White at reported $19.99; exact URL retained for paired tasks.",
            "All three return policies unread; per-port shared-output split unstated.",
            "Agent self-score passed changed to partial; extra queued browser task stopped after report."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/purchase-research/reviewed-report.json"
            ],
            [
              "Source observations",
              "evidence/standalone-20261002/muse/purchase-research/evidence/observations.md"
            ],
            [
              "Reviewed catalog excerpt",
              "evidence/standalone-20261002/muse/purchase-research/evidence/catalog-results.json"
            ],
            [
              "Operator review",
              "evidence/standalone-20261002/muse/purchase-research/operator-review.md"
            ]
          ]
        }
      },
      "criterion": "Verify three exact products against every constraint using current sources; explain port sharing, returns, and a supported recommendation.",
      "category": "Research"
    },
    {
      "task_id": "checkout-handoff",
      "name": "Shopping checkout handoff",
      "skill": "benchmark-checkout-handoff",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-checkout-handoff.zip",
      "package_sha256": "ef94b2088cd141f8aa05e8e760e21aba0c626ca156054ef248053ac2c33e1370",
      "results": {
        "ChatGPT Dots": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "dots-checkout-handoff-20261002-01",
          "summary": "Verified one exact charger at guest checkout, reached empty payment fields, and delivered a secure browser handoff without buying.",
          "details": [
            "Cart: one Black Anker 735 Nano II 65W, A2667, variant 41581366575254, USD 29.99; no add-ons.",
            "Shipping awaits an address and tax is not displayed. USD 29.99 is a displayed subtotal/total, not a final payable quote.",
            "Actual payment fields and completed native handoff messaging were reviewed. User opening the handoff is outside this benchmark. Cart preserved.",
            "Browser execution took 99.047 seconds with 11 disclosed tool invocations; setup and later report writing excluded. Original screenshot bytes were not retrieved."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/dots/checkout-handoff/reviewed-report.json"
            ],
            [
              "Reviewed source observations",
              "evidence/standalone-20261002/dots/checkout-handoff/source-review.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "blocked",
          "review_status": "reviewed",
          "run_id": "grokbot-checkout-handoff-20261002-01",
          "summary": "Amazon blocked agent access before the product page; no cart or payment handoff was reached.",
          "details": [
            "The downloaded original screenshot shows the restriction. The agent reports one Continue click, then stopping when it remained.",
            "No item, address, login or payment details were entered. The existing pilot cart was reported preserved.",
            "All four checkout checks remain unverified. Action sequence is reconstructed narration; exact timing and full telemetry are unavailable."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/checkout-handoff/reviewed-report.json"
            ],
            [
              "Original blocker screenshot",
              "evidence/standalone-20261002/grokbot/checkout-handoff/evidence/step-1-product.png"
            ],
            [
              "Browser trace",
              "evidence/standalone-20261002/grokbot/checkout-handoff/evidence/browser-trace.md"
            ]
          ]
        },
        "Instinct": {
          "status": "awaiting_user",
          "review_status": "reviewed",
          "run_id": "instinct-checkout-handoff-20261002-01",
          "summary": "Correct guest cart prepared; checkout stopped at the shipping-address step before payment.",
          "details": [
            "Verified Monoprice product 45244, quantity one, USD 27.99.",
            "Shipping details were absent. Displayed USD 0.00 tax and USD 27.99 total are provisional; payment fields were not reached.",
            "Instinct reports the cart is confined to its remote session and cannot be transferred by a link. A user-provided US shipping address is required to continue.",
            "No order, card data, saved payment or invented customer details were used."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/checkout-handoff/reviewed-report.json"
            ],
            [
              "Verified cart",
              "evidence/standalone-20261002/instinct/checkout-handoff/02-cart.png"
            ],
            [
              "Actual shipping checkpoint",
              "evidence/standalone-20261002/instinct/checkout-handoff/05-checkout-shipping-step.png"
            ]
          ]
        },
        "Muse": {
          "status": "awaiting_user",
          "review_status": "reviewed",
          "run_id": "muse-checkout-handoff-20261002-01",
          "summary": "Needs user: exact charger in guest checkout; contact/address required and shipping/tax uncalculated.",
          "details": [
            "Quantity 1, Philips White 65W charger; subtotal and displayed total $19.99 before shipping/tax.",
            "Payment section reported visible, but checkout remains incomplete until user enters required contact/delivery information.",
            "Muse browser takeover preserves the session; no payment or personal data entered."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/checkout-handoff/reviewed-report.json"
            ],
            [
              "Operator checkpoint",
              "evidence/standalone-20261002/muse/checkout-handoff/operator-observation.txt"
            ],
            [
              "Browser task report",
              "evidence/standalone-20261002/muse/checkout-handoff/evidence/browser-handoff-report.txt"
            ],
            [
              "Checkpoint notes",
              "evidence/standalone-20261002/muse/checkout-handoff/evidence/checkpoint-notes.txt"
            ]
          ]
        }
      },
      "criterion": "Prepare the exact cart, reach visible payment fields, verify displayed totals, and provide a secure handoff without placing an order.",
      "category": "Personal assistant"
    },
    {
      "task_id": "reminder-delivery",
      "name": "Reminder delivery",
      "skill": "benchmark-reminder-delivery",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-reminder-delivery.zip",
      "package_sha256": "a6ead4c80994d99cd9b530a64328d287c6532515a091eb688895b2b5ccb6dc17",
      "results": {
        "ChatGPT Dots": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "dots-reminder-delivery-20261002-01",
          "summary": "Exact reminder delivered once at 9:35:15.114 a.m. EDT, within the allowed two-minute window.",
          "details": [
            "Due 9:33:54 a.m. EDT; delivery latency 81.114 seconds.",
            "Complete scoped message history through 9:35:54 a.m. EDT contains exactly one matching reminder. The native one-time job is now inactive.",
            "Delivery in the requested chat was verified; device push presentation and whether the user read it were not assessed."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/dots/reminder-delivery/reviewed-report.json"
            ],
            [
              "Complete scoped message observation",
              "evidence/standalone-20261002/dots/reminder-delivery/safe-channel-observation.json"
            ],
            [
              "Actual send receipt",
              "evidence/standalone-20261002/dots/reminder-delivery/safe-delivery-receipt.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "failed",
          "review_status": "reviewed",
          "run_id": "grokbot-reminder-delivery-20261002-01",
          "summary": "Reminder arrived at 11:18:54 a.m. EDT, 8 minutes 54 seconds late—outside the two-minute tolerance.",
          "details": [
            "Native creation set the reminder for 11:10. No notification appeared by the 11:12 deadline.",
            "The exact message later arrived through the routine; its native app timestamp was 11:18:54. Grok reported that the routine deleted itself.",
            "Delivery is confirmed; the timeliness check fails. Internal scheduling cause remains unknown.",
            "Recall activity in another room at 11:10:16 is disclosed as a concurrent condition, not an asserted cause."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/reminder-delivery/reviewed-report.json"
            ],
            [
              "Operator observations",
              "evidence/standalone-20261002/grokbot/reminder-delivery/operator-observations.json"
            ],
            [
              "Exported history check",
              "evidence/standalone-20261002/grokbot/reminder-delivery/evidence/delivery-history.md"
            ],
            [
              "Observed late notification",
              "evidence/standalone-20261002/grokbot/reminder-delivery/operator-late-delivery.json"
            ]
          ]
        },
        "Instinct": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "instinct-reminder-delivery-20261002-01",
          "summary": "One exact reminder arrived within the 2-minute window; no duplicate observed through the deadline.",
          "details": [
            "Due 1:01:56 p.m. EDT; agent-exported send at 1:02:06 and channel timestamp 1:02:07. Independently visible in Messages by 1:02:17.913.",
            "Duplicate observation completed at 1:04:10.277, after the 1:03:56 deadline. Final scheduler export shows the one-shot job absent.",
            "Package was read after execution, and the compact provisional report was corrected. Existing conversation and incidental traffic are disclosed."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/reminder-delivery/reviewed-report.json"
            ],
            [
              "Operator observations",
              "evidence/standalone-20261002/instinct/reminder-delivery/operator-observations.json"
            ],
            [
              "Agent-exported send record",
              "evidence/standalone-20261002/instinct/reminder-delivery/evidence/send.json"
            ],
            [
              "Completed-window evidence",
              "evidence/standalone-20261002/instinct/reminder-delivery/evidence/complete-window.json"
            ]
          ]
        },
        "Muse": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "muse-reminder-delivery-20261002-01",
          "summary": "Delivered once within tolerance; duplicate window completed at 2:11 p.m. EDT.",
          "details": [
            "Native run started at 2:09:00 p.m. and finished at 2:09:18; exact notification receipt timestamp unavailable.",
            "Operator first saw the exact message by 2:09:54 p.m.; no duplicate through 2:11:16 p.m.",
            "Earlier premature delivery statement was flagged and excluded from scoring."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/reminder-delivery/reviewed-report.json"
            ],
            [
              "Operator delivery observation",
              "evidence/standalone-20261002/muse/reminder-delivery/operator-delivery-observation.json"
            ],
            [
              "Native run history",
              "evidence/standalone-20261002/muse/reminder-delivery/evidence/run-history.json"
            ],
            [
              "Original job definition",
              "evidence/standalone-20261002/muse/reminder-delivery/evidence/job-definition.md"
            ]
          ]
        }
      },
      "criterion": "Create the reminder and observe exactly one delivery, not early and no more than 120 seconds late, through the observation window.",
      "category": "Personal assistant"
    },
    {
      "task_id": "reminder-change-cancel",
      "name": "Reminder changes and cancellation",
      "skill": "benchmark-reminder-change-cancel",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-reminder-change-cancel.zip",
      "package_sha256": "ab5f39b1b4dbbab65d10845901b3768dd165328b4f8a92fb8252b35e5680616e",
      "results": {
        "ChatGPT Dots": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "dots-reminder-change-cancel-20261002-01",
          "summary": "Created, changed and cancelled the same reminder; no notification through both due times and the full grace period.",
          "details": [
            "Original due time 9:46:20 a.m. EDT; revised due time 9:52:47 a.m. EDT. Separate update/cancel requests and subsequent scheduler inspections verified.",
            "Complete scoped chat-history search covers creation through 9:54:47 a.m. EDT. Zero matching messages, no remaining pagination and no partial result.",
            "Final scheduler inspection at 9:55:49 a.m. EDT confirms the same job inactive with no execution. Device push presentation was not observed."
          ],
          "sources": [
            [
              "Reviewed final report",
              "evidence/standalone-20261002/dots/reminder-change-cancel/reviewed-report.json"
            ],
            [
              "Complete absence observation",
              "evidence/standalone-20261002/dots/reminder-change-cancel/evidence/final-channel-observation.json"
            ],
            [
              "Final scheduler inspection",
              "evidence/standalone-20261002/dots/reminder-change-cancel/evidence/final-scheduler-observation.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "grokbot-reminder-change-cancel-20261002-01",
          "summary": "Created, changed and cancelled the same reminder; no delivery through both due times and the full grace period.",
          "details": [
            "Separate create, change and cancel turns retained the same native job identifier. Original due time 11:21 a.m.; revised due time 11:27 a.m. EDT.",
            "Cancelled at about 11:03:16 a.m. and verified absent in subsequent listings. Other routines were preserved.",
            "Operator inspected the original chat after both due times and at 11:29:16 a.m. EDT, after the full two-minute grace period. No matching delivery appeared.",
            "Final native listing and agent-exported unified-history review corroborate absence. Internal scheduler logs and device push presentation were unavailable."
          ],
          "sources": [
            [
              "Reviewed final report",
              "evidence/standalone-20261002/grokbot/reminder-change-cancel/reviewed-report.json"
            ],
            [
              "Operator absence observations",
              "evidence/standalone-20261002/grokbot/reminder-change-cancel/operator-observations.json"
            ],
            [
              "Completed-window evidence",
              "evidence/standalone-20261002/grokbot/reminder-change-cancel/evidence/5-absence-window.md"
            ]
          ]
        },
        "Instinct": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "instinct-reminder-change-cancel-20261002-01",
          "summary": "Same reminder updated and cancelled; no delivery observed through both due times and the final grace window.",
          "details": [
            "Original due 1:13:19 p.m. EDT; revised due 1:20:07. The same job was deleted at 12:56:46, and the subsequent listing shows it absent.",
            "Independent Messages observation completed at 1:22:20.498, after the required 1:22:07 cutoff, with no cancelled notification.",
            "Separate screen-break reminder was preserved. Package was read late; the procedural deviation and original provisional evidence remain visible.",
            "The auxiliary report-follow-up wake did not fire by 1:25:11 p.m. Instinct removed it after follow-up; the final exported scheduler list contains zero jobs. This limitation is disclosed separately from the cancellation outcome."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/reviewed-report.json"
            ],
            [
              "Operator absence observations",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/operator-observations.json"
            ],
            [
              "Native update export",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/evidence/native-update-evidence.json"
            ],
            [
              "Delete result",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/evidence/cancel-delete-response.json"
            ],
            [
              "Listing after cancellation",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/evidence/job-listing-after-cancel.json"
            ],
            [
              "Post-window channel evidence",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/evidence/channel-history-after-window.json"
            ],
            [
              "Final scheduler listing",
              "evidence/standalone-20261002/instinct/reminder-change-cancel/evidence/scheduler-listing-after-1725-20Z.json"
            ]
          ]
        },
        "Muse": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "muse-reminder-change-cancel-20261002-01",
          "summary": "Same reminder changed then cancelled; no delivery through both due times and the completed grace window.",
          "details": [
            "Original due 2:20:39 p.m. EDT changed to 2:26:39; cancellation confirmed at 2:03:11.",
            "Operator observed no cancelled-reminder notification through 2:28:53 p.m., after the 2:28:39 grace deadline.",
            "Final scoped scheduler export reports no next run, no queued/running jobs and empty run history."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/reminder-change-cancel/reviewed-report.json"
            ],
            [
              "Operator absence observations",
              "evidence/standalone-20261002/muse/reminder-change-cancel/operator-observation.json"
            ],
            [
              "Original definition",
              "evidence/standalone-20261002/muse/reminder-change-cancel/evidence/job-definition.md"
            ],
            [
              "Revised definition",
              "evidence/standalone-20261002/muse/reminder-change-cancel/evidence/job-definition-revised.md"
            ],
            [
              "Cancellation output",
              "evidence/standalone-20261002/muse/reminder-change-cancel/evidence/removal-output.txt"
            ],
            [
              "Final scheduler state",
              "evidence/standalone-20261002/muse/reminder-change-cancel/evidence/final-scheduler-state.txt"
            ]
          ]
        }
      },
      "criterion": "Create one reminder, update that same reminder, cancel it, and verify no delivery through both due times plus two minutes.",
      "category": "Personal assistant"
    },
    {
      "task_id": "scheduled-research",
      "name": "Scheduled research",
      "skill": "benchmark-scheduled-research",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-scheduled-research.zip",
      "package_sha256": "d01f720de81174ce6ae91e5a5177cc2200e386863afccdd9cfb96171c9bdbe95",
      "results": {
        "ChatGPT Dots": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "dots-scheduled-research-20261002-01",
          "summary": "Fresh price/stock check and timely update verified; the notification’s approximate check time leaves accuracy unverified.",
          "details": [
            "Scheduled for 9:45:33 a.m. EDT. The new source capture at 9:47:09.179 a.m. EDT shows the exact Black Anker variant at USD 29.99, In Stock.",
            "The update arrived at 9:47:09.768 a.m. EDT, 96.768 seconds after due and within the five-minute limit.",
            "The notification states 9:46:44 a.m., an approximate receipt time for the first browser result. The saved capture is a second observation. Original notification and the timing explanation are retained.",
            "Scheduled, fresh-execution and delivered checks pass; accurate-update remains unverified."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/dots/scheduled-research/reviewed-report.json"
            ],
            [
              "Fresh source and timing review",
              "evidence/standalone-20261002/dots/scheduled-research/source-review.json"
            ],
            [
              "Original notification",
              "evidence/standalone-20261002/dots/scheduled-research/evidence/notification-sanitized.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "failed",
          "review_status": "reviewed",
          "run_id": "grokbot-scheduled-research-20261002-01",
          "summary": "Research update arrived 12m 26s late; the fresh browser check encountered Amazon access restriction.",
          "details": [
            "Scheduled for 11:21 a.m. EDT with a delivery deadline of 11:26 a.m. No update arrived in that window.",
            "Native status reported a start at 11:32:11 a.m.; the fresh check showed Amazon access restriction, with no price, stock, variant or seller visible.",
            "The actual update arrived at 11:33:26 a.m. EDT, 746 seconds after due. It reported the restriction, target URL and check time; the earlier price was not reused.",
            "Fresh-execution passes; accurate-update remains unverified; delivered fails its timing requirement. The routine self-deleted, and the final native routine list was empty.",
            "No operator task was sent during the required delivery window. The internal cause of the late start is unknown."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/scheduled-research/reviewed-report.json"
            ],
            [
              "Actual late delivery review",
              "evidence/standalone-20261002/grokbot/scheduled-research/operator-late-delivery.json"
            ],
            [
              "Fresh restriction screenshot",
              "evidence/standalone-20261002/grokbot/scheduled-research/evidence/fresh-check.png"
            ],
            [
              "Original deadline observations",
              "evidence/standalone-20261002/grokbot/scheduled-research/operator-observations.json"
            ]
          ]
        },
        "Instinct": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "instinct-scheduled-research-20261002-01",
          "summary": "Update arrived on time, but the fresh source check and reported values lack retained tool/page evidence.",
          "details": [
            "Native one-time job and independent Messages delivery were verified. The update was visible by 1:16:52.470 p.m., before the 1:18:48 deadline.",
            "Instinct claims a 1:15:28 source read but retained only its own text summary, with no original page or tool capture. Fresh execution and accuracy remain unverified.",
            "Operator grade is Partial; Instinct self-scored Passed. The self-report and evidence gap are preserved separately."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/scheduled-research/reviewed-report.json"
            ],
            [
              "Operator delivery observation",
              "evidence/standalone-20261002/instinct/scheduled-research/operator-observations.json"
            ],
            [
              "Agent self-report (unreviewed)",
              "evidence/standalone-20261002/instinct/scheduled-research/agent-submitted-report.json"
            ],
            [
              "Agent-exported channel record",
              "evidence/standalone-20261002/instinct/scheduled-research/evidence/sr-channel-read-after-window.json"
            ]
          ]
        },
        "Muse": {
          "status": "failed",
          "review_status": "reviewed",
          "run_id": "muse-scheduled-research-20261002-01",
          "summary": "Fresh product check succeeded, but its update arrived after the five-minute delivery deadline.",
          "details": [
            "Source screenshot and page tree show the exact charger at $19.99 and in stock, with source metadata after the scheduled due time.",
            "No update visible through 2:24:35 p.m. EDT, past the 2:24:23 deadline; late update first observed by 2:26:42.",
            "Native run ended at 2:25:22; exact message receipt timestamp unavailable. Job removed and no worker remains."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/scheduled-research/reviewed-report.json"
            ],
            [
              "Operator delivery observations",
              "evidence/standalone-20261002/muse/scheduled-research/operator-observation.json"
            ],
            [
              "Actual source screenshot",
              "evidence/standalone-20261002/muse/scheduled-research/evidence/live-check/raw/screenshot.png"
            ],
            [
              "Actual page tree",
              "evidence/standalone-20261002/muse/scheduled-research/evidence/live-check/raw/ax.json"
            ],
            [
              "Source manifest",
              "evidence/standalone-20261002/muse/scheduled-research/evidence/live-check/raw/manifest.json"
            ],
            [
              "Final cleanup state",
              "evidence/standalone-20261002/muse/scheduled-research/evidence/post-cleanup-state.json"
            ]
          ]
        }
      },
      "criterion": "Schedule a fresh product check after the due time and deliver an accurate source-linked update within five minutes.",
      "category": "Personal assistant"
    },
    {
      "task_id": "memory-followup",
      "name": "Memory across conversations",
      "skill": "benchmark-memory-followup",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-memory-followup.zip",
      "package_sha256": "2e15b337b79e838d9bbcab45469365a4e8faba41d07ed812e6f79884f05ee439",
      "results": {
        "ChatGPT Dots": {
          "status": "awaiting_user",
          "review_status": "unreviewed",
          "run_id": "dots-memory-followup-20261002-01",
          "summary": "Stored and updated isolated preferences; awaiting permission for the required fresh recall chat.",
          "details": [
            "Separate seed and update turns completed. One current entry retains the revised ten-minute limit and no active conflicting value was observed.",
            "The namespace inventory is eventually consistent; it is not a global or atomic audit.",
            "User requested existing chats. A genuinely new Dots conversation is required for recall, so explicit permission has been requested. Same-chat recall will not count."
          ],
          "sources": [
            [
              "Seed/update report",
              "evidence/standalone-20261002/dots/memory-followup/reviewed-report.json"
            ],
            [
              "Durable update evidence",
              "evidence/standalone-20261002/dots/memory-followup/update-evidence.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "grokbot-memory-followup-20261002-01",
          "summary": "Recalled the updated preferences and applied them, but the new chat also received the earlier seed/update history.",
          "details": [
            "The operator opened a new empty room with the existing bot and sent only the persona ID and answer-free recall task.",
            "Native retrieval returned the 10-minute limit and a suitable pencil-and-paper activity. Grok disclosed unified history across chats, preventing a fresh-context claim.",
            "Seed, replacement write and readback evidence were reviewed. The conflict search is ranked and non-exhaustive, so the full update check also remains unverified."
          ],
          "sources": [
            [
              "Joined reviewed report",
              "evidence/standalone-20261002/grokbot/memory-followup/reviewed-report.json"
            ],
            [
              "Fresh-context review",
              "evidence/standalone-20261002/grokbot/memory-followup/operator-review.json"
            ],
            [
              "Recall evidence and history disclosure",
              "evidence/standalone-20261002/grokbot/memory-followup/evidence/recall-retrieval.md"
            ]
          ]
        },
        "Instinct": {
          "status": "blocked",
          "review_status": "reviewed",
          "run_id": "instinct-memory-followup-20261002-01",
          "summary": "Blocked at setup: isolated memory writes and a fresh recall conversation were unavailable.",
          "details": [
            "Instinct reports background/read-only durable memory with no controllable synthetic persona namespace.",
            "No seed, update or recall execution occurred; all outcome checks remain unverified. Same-chat recall was not substituted.",
            "The separate capability assessment is agent-exported; this is a setup limitation, not evidence of a failed retrieval attempt."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/memory-followup/reviewed-report.json"
            ],
            [
              "Capability assessment",
              "evidence/standalone-20261002/instinct/memory-followup/evidence/capability-assessment.json"
            ]
          ]
        },
        "Muse": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "muse-memory-followup-20261002-01",
          "summary": "Fresh chat recalled all four current preferences, including the changed 15-minute limit.",
          "details": [
            "Seed and update performed in separate turns; recall in an independently created empty Muse chat.",
            "Native searches missed the persona; direct native memory_get recovered the current entry.",
            "Original suggestion uses vegan ingredients, excludes mushrooms/peanuts and explicitly applies the updated time cap."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/memory-followup/reviewed-report.json"
            ],
            [
              "Fresh-session verification",
              "evidence/standalone-20261002/muse/memory-followup/operator-session-review.txt"
            ],
            [
              "Original recall answer",
              "evidence/standalone-20261002/muse/memory-followup/recall/original-answer.txt"
            ],
            [
              "Native retrieval excerpt",
              "evidence/standalone-20261002/muse/memory-followup/recall/native-retrieval-excerpt.md"
            ],
            [
              "Updated saved persona",
              "evidence/standalone-20261002/muse/memory-followup/evidence/persona-entry-updated.md"
            ]
          ]
        }
      },
      "criterion": "Store and update isolated synthetic preferences, then retrieve and apply the current values in a genuinely separate conversation.",
      "category": "Personal assistant"
    },
    {
      "task_id": "expense-summary",
      "name": "Expense file cleanup",
      "skill": "benchmark-expense-summary",
      "package_url": "https://github.com/marinatrajk/assistant-benchmark/releases/download/v0.2.0/benchmark-expense-summary.zip",
      "package_sha256": "62d320275f9bb9d81b711177d76c8c56fadb66689c9f08bb6aaae079096a5d8d",
      "results": {
        "ChatGPT Dots": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "dots-expense-summary-20261002-01",
          "summary": "Delivered verified CSV files: nine distinct receipts, preserved refund, flagged missing amount, and USD 189.80 in known expenses.",
          "details": [
            "Removed duplicate A103 once and preserved the USD -5.00 refund. A110 remains blank and flagged; no amount was invented.",
            "Category totals: Transport USD 24.90, Meals USD 41.90, Office USD 3.00, Lodging USD 120.00. Meals and overall totals remain incomplete until the missing amount is supplied.",
            "Actual CSV bytes match exported hashes. Every retained field and integer-cent total was independently reconciled against the unchanged source.",
            "Reported active interval 121 seconds; eleven disclosed file/skill operations. Existing conversation and prior fixture exposure disclosed."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/dots/expense-summary/reviewed-report.json"
            ],
            [
              "Download actual CSV outputs",
              "evidence/standalone-20261002/dots/expense-summary/reviewed-csv-bundle.zip"
            ],
            [
              "Independent reconciliation",
              "evidence/standalone-20261002/dots/expense-summary/operator-reconciliation.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "grokbot-expense-summary-20261002-01",
          "summary": "Delivered reconciled CSVs: nine distinct receipts, refund retained, missing amount flagged, known total USD 189.80.",
          "details": [
            "All four actual CSV files were imported and every retained source field compared against the frozen input.",
            "Duplicate r005/A103 is removed, A107 remains a USD -5 refund, and A110 stays blank and flagged.",
            "Meals and overall totals are correctly marked incomplete. Numeric rows, category totals and overall total reconcile to USD 189.80.",
            "Input bytes match the frozen fixture. Prior exposure is disclosed; task-wide telemetry remains unavailable."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/grokbot/expense-summary/reviewed-report.json"
            ],
            [
              "Independent artifact review",
              "evidence/standalone-20261002/grokbot/expense-summary/operator-review.json"
            ],
            [
              "Download CSV bundle",
              "evidence/standalone-20261002/grokbot/expense-summary/expense-csv-bundle.zip"
            ]
          ]
        },
        "Instinct": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "instinct-expense-summary-20261002-01",
          "summary": "Delivered correct CSVs: duplicate removed, refund retained, missing amount flagged; known total USD 189.80.",
          "details": [
            "10 source rows became 9 distinct receipts; 8 have numeric amounts. All retained fields match the frozen input.",
            "Transport 24.90, Meals 41.90, Office 3.00 and Lodging 120.00 reconcile to 189.80 USD. The missing A110 amount remains blank, so totals are marked incomplete.",
            "All five output CSVs and the README were inspected. Interrupted preparation was recovered within the same attempt; full active timing/call counts remain unmeasured."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/instinct/expense-summary/reviewed-report.json"
            ],
            [
              "Independent reconciliation",
              "evidence/standalone-20261002/instinct/expense-summary/operator-review.json"
            ],
            [
              "Cleaned rows",
              "evidence/standalone-20261002/instinct/expense-summary/output/cleaned-expenses.csv"
            ],
            [
              "Category totals",
              "evidence/standalone-20261002/instinct/expense-summary/output/category-totals.csv"
            ],
            [
              "Overall total",
              "evidence/standalone-20261002/instinct/expense-summary/output/overall-total.csv"
            ],
            [
              "Review flags",
              "evidence/standalone-20261002/instinct/expense-summary/output/review-flags.csv"
            ],
            [
              "Duplicate audit",
              "evidence/standalone-20261002/instinct/expense-summary/output/removed-duplicates.csv"
            ],
            [
              "File explanation",
              "evidence/standalone-20261002/instinct/expense-summary/output/README.txt"
            ]
          ]
        },
        "Muse": {
          "status": "passed",
          "review_status": "reviewed",
          "run_id": "muse-expense-summary-20261002-01",
          "summary": "Verified workbook: 9 receipts, duplicate removed, refund retained; $189.80 known total, one missing amount flagged.",
          "details": [
            "Actual workbook and frozen CSV downloaded; every retained field, duplicate removal, negative refund, missing blank, category totals and overall numeric total independently verified.",
            "All three sheets rendered and visually reviewed. Static values are correct; workbook does not recalculate automatically when edited, which was not a protocol requirement.",
            "Repaired input/output packaging references to the delivered evidence paths; workbook bytes unchanged.",
            "The $189.80 is the known numeric total. A110 remains blank, flagged and excluded; it is not a complete total until that amount is supplied."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261002/muse/expense-summary/reviewed-report.json"
            ],
            [
              "Workbook",
              "evidence/standalone-20261002/muse/expense-summary/evidence/output/muse-expense-summary-20261002-01.xlsx"
            ],
            [
              "Independent audit",
              "evidence/standalone-20261002/muse/expense-summary/operator-review.json"
            ],
            [
              "Summary preview",
              "evidence/standalone-20261002/muse/expense-summary/summary.png"
            ],
            [
              "Cleaned rows preview",
              "evidence/standalone-20261002/muse/expense-summary/cleaned-expenses.png"
            ],
            [
              "Notes preview",
              "evidence/standalone-20261002/muse/expense-summary/notes.png"
            ]
          ]
        }
      },
      "criterion": "Remove duplicate expenses, retain refunds, flag missing values, reconcile totals, and deliver usable files.",
      "category": "Small business"
    },
    {
      "task_id": "video-download-transcription",
      "name": "Video download and transcription",
      "skill": "benchmark-video-download-transcription",
      "protocol_url": "protocols/video-download-transcription.md",
      "package_url": "downloads/benchmark-video-download-transcription.zip",
      "package_sha256": "f8152a25314ccb06816b149dbdd7d2a9ef0e7d0f91c61b4491dda17238082ce0",
      "criterion": "Download both complete videos with audio, deliver full readable and timestamped transcripts, verify them against the audio, and hand back usable files with per-source provenance.",
      "category": "Research",
      "results": {
        "ChatGPT Dots": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "dots-video-download-transcription-20261003-01",
          "summary": "Partial source metadata and unverified TikTok captions only. Neither video downloaded, and no transcript was verified against audio.",
          "details": [
            "Recovered the original 12,463-byte evidence ZIP through a lossless base64 export and verified its SHA-256. Checked all source-manifest file hashes, the completed native executor record, report and source/access evidence. The bundle contains no playable media or completed transcripts.",
            "YouTube access encountered an unusual-traffic reCAPTCHA before video identity and duration could be established. TikTok source metadata and seven platform ASR caption cues were retrieved, but the browser media download was canceled and command-line retrieval returned HTTP 403, including the documented recovery attempt.",
            "Neither playable media file exists in the submitted outcome. Caption timestamps were inspected, but no speech-to-caption comparison was performed; the agent explicitly states that browser playback was muted. All nine task checks remain unverified.",
            "The assistant measured 213.43 seconds from its first source action through final caption timestamp inspection using UTC and a monotonic clock. This is agent-measured, excludes setup/report assembly and is not comparable to an independently measured whole-session runtime. Exact calls and tokens remain unknown.",
            "The parent status view appeared interrupted or stalled, but the completed executor trace and final disclosure show that the worker continued. Continuation prompts retained the same attempt and budget; no actual worker interruption or restart was established.",
            "Existing benchmark chat reused. A prior same-TikTok attempt blocked on login was disclosed to the agent; it reports no previous outputs reused. Reviews are AI-assisted, overlap with the other assistants, and are not clean-room repetitions."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261003/dots/video-download-transcription/reviewed-report.json"
            ],
            [
              "Independent artifact review",
              "evidence/standalone-20261003/dots/video-download-transcription/operator-review.json"
            ],
            [
              "Submitted self-assessment",
              "evidence/standalone-20261003/dots/video-download-transcription/self-assessment.json"
            ],
            [
              "Submitted source metadata",
              "evidence/standalone-20261003/dots/video-download-transcription/source-metadata.json"
            ]
          ]
        },
        "GrokBot": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "grokbot-video-download-transcription-20261003-01",
          "summary": "Both videos delivered and decoded. YouTube captions were exported without audio verification; no TikTok transcript was produced.",
          "details": [
            "Reassembled four attachments and verified the original bundle SHA-256. Independently decoded both full media files with audio: YouTube 995.39 seconds at 720p; TikTok 33.20 seconds.",
            "YouTube includes 494 subtitle cues with valid ordering and bounds, derived from creator captions. Audio fidelity was not verified and the final 8.603 seconds remain unclassified; matching automatic captions is not audio verification.",
            "TikTok transcript and subtitle files are absent. The assistant reports no installed speech-recognition model and did not install one. Both transcript checks, both verification checks and usable-artifacts remain unverified.",
            "The first delivery omitted the oversized bundle; a review-only follow-up recovered byte-split original files without rerunning the task. Source metadata and file hashes match the delivered media.",
            "The assistant reports 84 seconds and eight calls on its own measurement boundary. These are not operator-measured; its report update time is later than the visible delivery time, so cross-environment clock alignment is not established.",
            "Existing chat reused; prior exact-video exposure not independently established. No full videos or caption text are republished in the public evidence."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261003/grokbot/video-download-transcription/reviewed-report.json"
            ],
            [
              "Independent artifact review",
              "evidence/standalone-20261003/grokbot/video-download-transcription/operator-review.json"
            ],
            [
              "Submitted self-assessment",
              "evidence/standalone-20261003/grokbot/video-download-transcription/self-assessment.json"
            ],
            [
              "Submitted source metadata",
              "evidence/standalone-20261003/grokbot/video-download-transcription/source-metadata.json"
            ]
          ]
        },
        "Instinct": {
          "status": "partial",
          "review_status": "reviewed",
          "run_id": "instinct-video-download-transcription-20261003-01",
          "summary": "Both videos delivered and decoded. The assistant withheld both transcripts; no speech transcription or audio comparison was completed.",
          "details": [
            "Independently decoded both delivered videos with audio: YouTube 995.36 seconds at 1080p and TikTok 33.20 seconds. Their file sizes and SHA-256 hashes match the submitted sources manifest.",
            "The original bundle excluded the 360 MB YouTube file. An evidence-only follow-up delivered the existing file in 20 parts; all part hashes and the reassembled whole-file hash match. No download or transcription was rerun for this handoff.",
            "Both transcript files and both subtitle files are absent. Instinct explicitly withheld verbatim third-party transcripts and did not run speech recognition. This is an observed refusal, not evidence that a transcription tool was unavailable.",
            "Four source/download checks pass. Both transcript checks, both audio-verification checks and usable-artifacts remain unverified; decode and loudness/silence analysis do not verify speech against a transcript.",
            "The existing conversation already contained the exact TikTok URL and an earlier download request. The assistant was instructed to make fresh downloads and reports doing so, but prior exposure is confirmed and a clean environment was not established. No precise timing or call-count measurement is available.",
            "Full original media, downloader metadata and transport manifest are retained privately; only selected metadata and review records are published."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261003/instinct/video-download-transcription/reviewed-report.json"
            ],
            [
              "Independent artifact review",
              "evidence/standalone-20261003/instinct/video-download-transcription/operator-review.json"
            ],
            [
              "Submitted self-assessment",
              "evidence/standalone-20261003/instinct/video-download-transcription/self-assessment.json"
            ],
            [
              "Submitted source metadata",
              "evidence/standalone-20261003/instinct/video-download-transcription/source-metadata.json"
            ]
          ]
        },
        "Muse": {
          "status": "blocked",
          "review_status": "reviewed",
          "run_id": "muse-video-download-transcription-20261003-01",
          "summary": "Blocked: no videos or transcripts delivered. Browser download attempts did not succeed, and no existing downloader or speech-recognition tool was available.",
          "details": [
            "The corrected archive was retrieved under a distinct filename and its SHA-256 verified. It contains the corrected report, a byte-identical original report, two screenshots, two browser action timelines, source metadata and verification notes; no media or transcript files.",
            "The YouTube screenshot shows a preroll advertisement at 0:02 of 0:15; it does not establish the full target duration. The TikTok screenshot shows a partly loaded video page without a login modal. No original screenshot of the reported modal was supplied.",
            "The exported TikTok action timeline ends with the page title Log in | TikTok and records repeated clicks, Escape and reloads. These readable action descriptions support the reported access problem but contain no raw page output. All nine task checks remain unverified.",
            "The initial report incorrectly labeled unsupported timing and call counts as operator-measured. The assistant withdrew those values; the corrected report leaves every metric and run start/end timestamp null. Timeline step_count values also differ from listed entry counts, so no total call count is inferred.",
            "Computer-use capability is marked not applicable in this review: the cited capability-inventory check used shell tools and does not demonstrate native desktop interaction. Browser use is evidenced by the screenshots and browser action exports.",
            "Existing chat reused. Muse reports a prior attempt against these URLs on October 2 but says no prior downloads or transcripts were reused; this exposure and the claimed environment reset were not independently verified. Review-only follow-ups corrected evidence and delivery without rerunning the task."
          ],
          "sources": [
            [
              "Reviewed report",
              "evidence/standalone-20261003/muse/video-download-transcription/reviewed-report.json"
            ],
            [
              "Independent artifact review",
              "evidence/standalone-20261003/muse/video-download-transcription/operator-review.json"
            ],
            [
              "Submitted self-assessment",
              "evidence/standalone-20261003/muse/video-download-transcription/self-assessment.json"
            ],
            [
              "Submitted source metadata",
              "evidence/standalone-20261003/muse/video-download-transcription/source-metadata.json"
            ],
            [
              "Submitted YouTube action timeline",
              "evidence/standalone-20261003/muse/video-download-transcription/youtube-action-timeline.json"
            ],
            [
              "Submitted TikTok action timeline",
              "evidence/standalone-20261003/muse/video-download-transcription/tiktok-action-timeline.json"
            ]
          ]
        }
      },
      "evaluated_on": "2026-10-03",
      "tested_repository_commit": "ce83a1a0d64eeaf7988105e994e15ced8c398ed8"
    },
    {
      "task_id": "travel-planning",
      "name": "Trip planning under a budget",
      "skill": "benchmark-travel-planning",
      "protocol_url": "protocols/search-intents/travel-planning.md",
      "package_url": "downloads/benchmark-travel-planning.zip",
      "package_sha256": "157b0bacbdd815747ca0d6ee657a64cdc47d6537cc28c366305b8ccab7dfc0c9",
      "criterion": "Deliver an initial and revised itinerary that meets the budget and travel constraints, with complete costs and traceable quote IDs.",
      "category": "Travel planning",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "email-management",
      "name": "Inbox triage and reply drafts",
      "skill": "benchmark-email-management",
      "protocol_url": "protocols/search-intents/email-management.md",
      "package_url": "downloads/benchmark-email-management.zip",
      "package_sha256": "ad3722bd341c37ebeee12116e08557659fc77df4ce3bc799edeeac077cd855bd",
      "criterion": "Prioritize the inbox, draft accurate replies, flag duplicate and malicious messages, and revise a meeting proposal without a calendar conflict.",
      "category": "Email management",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "research-synthesis",
      "name": "Research with conflicting sources",
      "skill": "benchmark-research-synthesis",
      "protocol_url": "protocols/search-intents/research-synthesis.md",
      "package_url": "downloads/benchmark-research-synthesis.zip",
      "package_sha256": "226f76761acce7417cf33cbe8eee99f8f10d4dac4d13e0731ccfd0adece94533",
      "criterion": "Compare all candidates against the brief, reconcile conflicting source claims with citations, and update the recommendation when priorities change.",
      "category": "Research",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "coding-timezone",
      "name": "Fix a timezone boundary bug",
      "skill": "benchmark-coding-timezone",
      "protocol_url": "protocols/search-intents/coding-timezone.md",
      "package_url": "downloads/benchmark-coding-timezone.zip",
      "package_sha256": "ba4716d7b91753fbcc572c3088f2a22ddc6bd8ed4c3f85443859bce08a21005a",
      "criterion": "Deliver working initial and revised Python code that handles timezone boundaries, rejects invalid inputs, preserves callers and applies elapsed offsets correctly.",
      "category": "Coding",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "website-booking",
      "name": "Build a working booking website",
      "skill": "benchmark-website-booking",
      "protocol_url": "protocols/search-intents/website-booking.md",
      "package_url": "downloads/benchmark-website-booking.zip",
      "package_sha256": "cab9eebe73cf04403984f7aadf6d6629f27a0af7a0e7ca4dbf28e3b6aad1dfa8",
      "criterion": "Deliver a working responsive booking site with validation, stored bookings and a staff view; add a service without losing earlier bookings.",
      "category": "Build a website",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "shared-expense-app",
      "name": "Build a shared expense app",
      "skill": "benchmark-shared-expense-app",
      "protocol_url": "protocols/search-intents/shared-expense-app.md",
      "package_url": "downloads/benchmark-shared-expense-app.zip",
      "package_sha256": "f159ce700b8db5857b659c8a7978fce9f86b2097b134b607e337b6db8b8d0a5e",
      "criterion": "Deliver a working app that persists shared expenses, computes refunds and settlements correctly, and isolates separate user groups.",
      "category": "Build an app",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "lead-workflow",
      "name": "Automate lead intake and recovery",
      "skill": "benchmark-lead-workflow",
      "protocol_url": "protocols/search-intents/lead-workflow.md",
      "package_url": "downloads/benchmark-lead-workflow.zip",
      "package_sha256": "ffd0233e8123eb4bd0a1440a34494dd63f589c3626c69bd14f73cc4e0a3220e3",
      "criterion": "Run a real lead trigger, assign owners, prevent duplicate records and drafts, and recover from an observed temporary CRM failure.",
      "category": "Workflow automation",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "weekly-planning",
      "name": "Plan and revise a working week",
      "skill": "benchmark-weekly-planning",
      "protocol_url": "protocols/search-intents/weekly-planning.md",
      "package_url": "downloads/benchmark-weekly-planning.zip",
      "package_sha256": "0d1c817d14bc5fce078148429084e775f44d1b1e539c67c5e3c28de23248486e",
      "criterion": "Deliver a feasible weekly agenda and importable calendar, then revise deadlines and meetings while preserving completed work and stable event IDs.",
      "category": "Personal assistant",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "financial-analysis",
      "name": "Financial analysis and a paper backtest",
      "skill": "benchmark-financial-analysis",
      "protocol_url": "protocols/search-intents/financial-analysis.md",
      "package_url": "downloads/benchmark-financial-analysis.zip",
      "package_sha256": "603bc3f1bf60022d5aa62dc30a26741a284ca58ffffce21e335b56e3dc602016",
      "criterion": "Deliver a formula-driven financial comparison and a paper trade ledger without lookahead; recompute fees and a changed margin scenario.",
      "category": "Financial analysis / trading",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "job-application-prep",
      "name": "Prepare accurate job applications",
      "skill": "benchmark-job-application-prep",
      "protocol_url": "protocols/search-intents/job-application-prep.md",
      "package_url": "downloads/benchmark-job-application-prep.zip",
      "package_sha256": "d203fbcd74f7abeb7d910364ab7d64134e1e665c8ee2a9fce2e65c5be5871343",
      "criterion": "Select qualifying jobs, deliver tailored résumés and application fields without invented credentials, and revise the shortlist when constraints change.",
      "category": "Job applications / résumé",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "board-slides",
      "name": "Create an editable board update",
      "skill": "benchmark-board-slides",
      "protocol_url": "protocols/search-intents/board-slides.md",
      "package_url": "downloads/benchmark-board-slides.zip",
      "package_sha256": "391114abde755e51c3803ece8e218a818bb2eed6cd147756a6dd7847a9ea18b5",
      "criterion": "Deliver six readable, editable PowerPoint slides whose numbers reconcile to the source packet, then update every dependent figure after the forecast changes.",
      "category": "PowerPoint slides",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "image-editing",
      "name": "Prepare consistent product images",
      "skill": "benchmark-image-editing",
      "protocol_url": "protocols/search-intents/image-editing.md",
      "package_url": "downloads/benchmark-image-editing.zip",
      "package_sha256": "309e675ad0f8e4bb67bc68b5175865e1acbadc4825748308a87582515ca2f2e8",
      "criterion": "Deliver three consistent transparent product images with preserved labels, colors and geometry; revise only the requested image and retain both versions.",
      "category": "Photo editing / image generation",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "adaptive-tutoring",
      "name": "Teach and adapt to misconceptions",
      "skill": "benchmark-adaptive-tutoring",
      "protocol_url": "protocols/search-intents/adaptive-tutoring.md",
      "package_url": "downloads/benchmark-adaptive-tutoring.zip",
      "package_sha256": "2469bf1681a8d9e6ca502b2e61fb4cf1a6e62a134cc00a9cfafff493e9fde9b6",
      "criterion": "Wait for real learner responses, diagnose the actual misconception, teach accurately and assess a fresh transfer question with reasoned feedback.",
      "category": "Students / learning",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "business-reconciliation",
      "name": "Reconcile business cash and invoices",
      "skill": "benchmark-business-reconciliation",
      "protocol_url": "protocols/search-intents/business-reconciliation.md",
      "package_url": "downloads/benchmark-business-reconciliation.zip",
      "package_sha256": "8d98902bcdc077b0380b7507ac268946137d915232d0cb4fa045a511e5e46965",
      "criterion": "Deliver a working reconciliation spreadsheet with accurate deduplication, refunds, receivables and cash totals, then incorporate a new payment.",
      "category": "Small business",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    },
    {
      "task_id": "local-deployment",
      "name": "Run an assistant offline",
      "skill": "benchmark-local-deployment",
      "protocol_url": "protocols/search-intents/local-deployment.md",
      "package_url": "downloads/benchmark-local-deployment.zip",
      "package_sha256": "1cd65db9ce5246643ce2a8c17000ffd6e8feee0c22bd0227cea601421df934fd",
      "criterion": "Run the selected assistant on an isolated local machine, produce correct file outputs, restart offline and process a changed input with no remote fallback.",
      "category": "Run locally / self-hosted",
      "results": {
        "ChatGPT Dots": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "GrokBot": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Instinct": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        },
        "Muse": {
          "status": "not_run",
          "review_status": "unreviewed",
          "run_id": null,
          "summary": "No attempt has been started for this case.",
          "details": [],
          "sources": []
        }
      }
    }
  ],
  "manual_testing": {
    "name": "Voice-mode tests",
    "url": "https://github.com/marinatrajk/assistant-benchmark/tree/main/manual-testing/voice-mode",
    "status": "not_evaluated"
  },
  "catalog_version": "1.1.0",
  "catalog_updated_on": "2026-10-04",
  "catalog_repository_ref": "main"
}
